Files
archivdms/internal/pagesplit/pagesplit.go
T
patrick 9a24ea29e1 FDN-01: repository & projektgerüst
Git-Repository für bestehenden archivdms-Code initialisiert, Branch-/Commit-Konvention (feature/<ticket>-<slug>-Branches, Ticket-Prefix in Commit-Nachricht) etabliert.
2026-08-11 21:27:53 +02:00

515 lines
18 KiB
Go

// Package pagesplit implements barcode separator-page splitting for
// multi-page PDF ingest ("Trennseiten-Split", inspired by Paperless-ngx's
// ASN/separator barcode feature, adapted to archivdms's ingest pipeline).
//
// Idea: a scanner operator interleaves printed separator sheets carrying a
// well-known barcode (default value "ARCHIVDMS-SPLIT") between the individual
// receipts of one long scan run. At ingest the PDF is checked page by page for
// that barcode; where it is found, the document is cut, and the separator page
// itself is dropped (it is a control sheet, not content — same behaviour as
// Paperless-ngx). Each resulting part then runs through the completely normal
// staging path (own WORM file, own hash/duplicate check, own processing job).
//
// Design constraints this package follows, all inherited from the existing
// codebase:
//
// - No CGO, no PDF library: everything is done by shelling out to the
// poppler-utils binaries that are already a service dependency of the OCR
// pipeline (pdfinfo, pdftoppm, pdfseparate, pdfunite) plus zbarimg via
// internal/barcode. Deliberately NOT qpdf/pdftk — those would be a new
// package dependency for something poppler already covers.
// - Best-effort, fail-safe: every error path returns "no split" rather than
// failing the upload. A scanner run that cannot be analysed must still be
// archived, unsplit, rather than rejected. The one thing that is never
// silently swallowed is a *partially* produced split — Split either yields
// a complete set of parts or nothing at all.
// - Off by default (Detector.Enabled), per the project's conservative rule
// for new preprocessing behaviour (cf. the Otsu binarize switch in
// internal/ocr).
//
// Scope note: only application/pdf is handled. Multi-page TIFF is a
// theoretically possible scanner output but is not currently produced by any
// archivdms ingest path (HTTP upload and the SFTP watcher both hand single
// images or PDFs to the pipeline), so it is intentionally out of scope here
// rather than half-supported.
package pagesplit
import (
"bytes"
"context"
"crypto/rand"
"encoding/hex"
"fmt"
"log/slog"
"os"
"os/exec"
"path/filepath"
"regexp"
"sort"
"strconv"
"strings"
"time"
"archivdms/internal/barcode"
)
// DefaultMarker is the barcode payload that marks a separator page when no
// other value is configured. Chosen to be unambiguous and unlikely to collide
// with a taxonomy barcode (internal/storage/taxonomy.go barcode_value) or with
// anything printed on a real invoice.
const DefaultMarker = "ARCHIVDMS-SPLIT"
// defaultRasterDPI is the resolution separator detection rasterizes at. Much
// lower than the 300 dpi the OCR pipeline uses: a separator sheet carries one
// large, high-contrast barcode, and 150 dpi decodes those reliably while
// keeping the extra pdftoppm pass cheap on long scan runs.
const defaultRasterDPI = 150
// defaultTimeout bounds each individual poppler subprocess call.
const defaultTimeout = 120 * time.Second
// defaultMaxPages caps how many pages are analysed. A scan run beyond this is
// treated as "not analysable" (no split, archived as one document) instead of
// spending unbounded time rasterizing — the same bounded-worst-case reasoning
// as internal/ocr's maxImagePixels clamp.
const defaultMaxPages = 200
// Detector performs separator-page detection and PDF splitting.
//
// Construct via New and set the optional fields afterwards; the zero value is
// disabled and therefore a safe no-op.
type Detector struct {
// Enabled turns the whole feature on. False (zero value) => Split always
// reports "no split".
Enabled bool
// Marker is the barcode payload identifying a separator page. Empty
// defaults to DefaultMarker. Compared case-insensitively after trimming.
Marker string
// MarkerPrefix switches the comparison from "equals Marker" to "starts
// with Marker", so operators can encode extra data on the separator sheet
// (e.g. "ARCHIVDMS-SPLIT-2026-INVOICES") with a single configured value.
MarkerPrefix bool
// PdftoppmPath/PdfinfoPath/PdfseparatePath/PdfunitePath name the poppler
// binaries. Empty values fall back to the plain command names.
PdftoppmPath string
PdfinfoPath string
PdfseparatePath string
PdfunitePath string
// TmpDir is the scratch base directory (config.StorageConfig.OCRTmpPath()).
// Every Split call gets its own subdirectory, removed by Result.Cleanup.
TmpDir string
// RasterDPI overrides defaultRasterDPI. MaxPages overrides defaultMaxPages.
RasterDPI int
MaxPages int
// Timeout bounds each subprocess call. Zero => defaultTimeout.
Timeout time.Duration
// Logger receives best-effort diagnostics. Optional (nil = silent).
Logger *slog.Logger
}
// New builds a Detector from the resolved config values.
func New(enabled bool, marker string, markerPrefix bool, pdftoppmPath, tmpDir string) *Detector {
return &Detector{
Enabled: enabled,
Marker: marker,
MarkerPrefix: markerPrefix,
PdftoppmPath: pdftoppmPath,
TmpDir: tmpDir,
}
}
// Result describes a completed split.
type Result struct {
// Parts holds the absolute paths of the produced part PDFs, in original
// page order. Always at least one entry when Split reports split == true.
Parts []string
// PartPageRanges[i] holds the 1-based [first,last] page numbers of Parts[i]
// within the original document — audit-log material, so the aggregation of
// pages into parts stays reconstructible after the original is gone.
PartPageRanges [][2]int
// SeparatorPages holds the 1-based page numbers that carried the marker
// barcode and were therefore dropped.
SeparatorPages []int
// PageCount is the original document's total page count.
PageCount int
// Cleanup removes the scratch directory holding Parts. Never nil when
// Split returned split == true; callers must defer it.
Cleanup func()
}
// Split analyses pdfPath for separator pages and, if any are found, produces
// one part PDF per content segment.
//
// Returns split == false (with a nil Result) for every "carry on normally"
// outcome: detector disabled, poppler/zbarimg missing, fewer than two pages,
// page count above MaxPages, no separator barcode found, or every page being a
// separator page. Only genuinely unexpected failures return an error, and even
// those are meant to be treated by the caller as "archive unsplit" plus an
// audit entry — never as an upload failure.
//
// pdfPath must be a scratch/inbox file: it is only ever read, but the whole
// point of this function is that it runs BEFORE the file becomes a WORM
// archive object, so it must never be pointed at store/.
func (d *Detector) Split(ctx context.Context, pdfPath string) (res *Result, split bool, err error) {
if d == nil || !d.Enabled {
return nil, false, nil
}
for _, bin := range []string{d.pdfinfoPath(), d.pdftoppmPath(), d.pdfseparatePath(), d.pdfunitePath()} {
if _, lookErr := exec.LookPath(bin); lookErr != nil {
d.log(slog.LevelWarn, "pagesplit skipped: poppler binary not found in PATH",
"binary", bin, "err", lookErr)
return nil, false, nil
}
}
if _, lookErr := exec.LookPath("zbarimg"); lookErr != nil {
d.log(slog.LevelWarn, "pagesplit skipped: zbarimg not found in PATH", "err", lookErr)
return nil, false, nil
}
pageCount, err := d.pageCount(ctx, pdfPath)
if err != nil {
return nil, false, fmt.Errorf("pagesplit: page count: %w", err)
}
if pageCount < 2 {
return nil, false, nil
}
if pageCount > d.maxPages() {
d.log(slog.LevelWarn, "pagesplit skipped: page count above limit",
"file", pdfPath, "pages", pageCount, "max_pages", d.maxPages())
return nil, false, nil
}
jobDir := filepath.Join(d.tmpDir(), "split-"+randomID())
if mkErr := os.MkdirAll(jobDir, 0o750); mkErr != nil {
return nil, false, fmt.Errorf("pagesplit: create scratch dir: %w", mkErr)
}
cleanup := func() { os.RemoveAll(jobDir) }
// Anything below that returns without a successful split must not leak the
// scratch directory; the success path hands cleanup to the caller instead.
ok := false
defer func() {
if !ok {
cleanup()
}
}()
sepPages, err := d.detectSeparatorPages(ctx, pdfPath, jobDir, pageCount)
if err != nil {
return nil, false, fmt.Errorf("pagesplit: separator detection: %w", err)
}
if len(sepPages) == 0 {
return nil, false, nil
}
ranges := contentRanges(pageCount, sepPages)
if len(ranges) == 0 {
// Pathological upload: only separator sheets, no content at all. Do not
// silently discard it — archive the original unsplit so the operator
// sees what was scanned.
d.log(slog.LevelWarn, "pagesplit skipped: document consists of separator pages only",
"file", pdfPath, "pages", pageCount)
return nil, false, nil
}
partsDir := filepath.Join(jobDir, "parts")
if mkErr := os.MkdirAll(partsDir, 0o750); mkErr != nil {
return nil, false, fmt.Errorf("pagesplit: create parts dir: %w", mkErr)
}
var parts []string
for i, rg := range ranges {
partPath, perr := d.extractRange(ctx, pdfPath, partsDir, i+1, rg[0], rg[1])
if perr != nil {
// Partial split is never handed out — the caller falls back to
// archiving the unsplit original.
return nil, false, fmt.Errorf("pagesplit: extract pages %d-%d: %w", rg[0], rg[1], perr)
}
parts = append(parts, partPath)
}
d.log(slog.LevelInfo, "pagesplit produced parts",
"file", pdfPath, "pages", pageCount, "separator_pages", sepPages, "parts", len(parts))
ok = true
return &Result{
Parts: parts,
PartPageRanges: ranges,
SeparatorPages: sepPages,
PageCount: pageCount,
Cleanup: cleanup,
}, true, nil
}
// IsSeparatorValue reports whether a decoded barcode payload marks a separator
// page under this detector's marker configuration.
func (d *Detector) IsSeparatorValue(value string) bool {
v := strings.ToUpper(strings.TrimSpace(value))
m := strings.ToUpper(strings.TrimSpace(d.marker()))
if v == "" || m == "" {
return false
}
if d.MarkerPrefix {
return strings.HasPrefix(v, m)
}
return v == m
}
var pdfinfoPagesRegex = regexp.MustCompile(`(?m)^Pages:\s+(\d+)`)
// pageCount reads the page count via `pdfinfo`.
func (d *Detector) pageCount(ctx context.Context, pdfPath string) (int, error) {
cctx, cancel := context.WithTimeout(ctx, d.timeout())
defer cancel()
cmd := exec.CommandContext(cctx, d.pdfinfoPath(), pdfPath)
var out, stderr bytes.Buffer
cmd.Stdout = &out
cmd.Stderr = &stderr
if err := cmd.Run(); err != nil {
return 0, fmt.Errorf("pdfinfo failed: %w (%s)", err, strings.TrimSpace(stderr.String()))
}
m := pdfinfoPagesRegex.FindStringSubmatch(out.String())
if m == nil {
return 0, fmt.Errorf("pdfinfo output had no Pages line")
}
n, err := strconv.Atoi(m[1])
if err != nil {
return 0, fmt.Errorf("pdfinfo page count unparseable: %w", err)
}
return n, nil
}
// pageNumRegex pulls the page number out of the filenames pdftoppm/pdfseparate
// generate (page-01.png, page-1.png, seg-12.pdf, ...). Sorting on that number
// rather than lexically matters as soon as a run crosses 9 or 99 pages.
var pageNumRegex = regexp.MustCompile(`(\d+)\D*$`)
// detectSeparatorPages rasterizes every page once and decodes barcodes on it,
// returning the 1-based page numbers that carry the marker.
func (d *Detector) detectSeparatorPages(ctx context.Context, pdfPath, jobDir string, pageCount int) ([]int, error) {
rasterDir := filepath.Join(jobDir, "raster")
if err := os.MkdirAll(rasterDir, 0o750); err != nil {
return nil, fmt.Errorf("create raster dir: %w", err)
}
cctx, cancel := context.WithTimeout(ctx, d.timeout())
defer cancel()
prefix := filepath.Join(rasterDir, "page")
cmd := exec.CommandContext(cctx, d.pdftoppmPath(),
"-r", strconv.Itoa(d.rasterDPI()), "-png", pdfPath, prefix)
var stderr bytes.Buffer
cmd.Stderr = &stderr
if err := cmd.Run(); err != nil {
return nil, fmt.Errorf("pdftoppm failed: %w (%s)", err, strings.TrimSpace(stderr.String()))
}
pages, err := sortedNumberedFiles(rasterDir, ".png")
if err != nil {
return nil, err
}
if len(pages) != pageCount {
// Mismatch means the page-number mapping below cannot be trusted, and a
// wrong mapping would cut the document in the wrong place — refuse.
return nil, fmt.Errorf("rasterized %d pages but pdfinfo reported %d", len(pages), pageCount)
}
var sep []int
for i, page := range pages {
codes, decErr := barcode.DecodeBarcodes(ctx, page)
if decErr != nil {
// Best-effort per page, exactly as in internal/ocr: a page whose
// barcode pass errored is simply treated as a content page.
continue
}
for _, code := range codes {
if d.IsSeparatorValue(code) {
sep = append(sep, i+1)
break
}
}
}
return sep, nil
}
// contentRanges turns a page count plus the separator page numbers into the
// 1-based inclusive page ranges of the content segments, dropping the
// separator pages themselves and any empty segment (two adjacent separator
// sheets, or one at the very start/end).
func contentRanges(pageCount int, sepPages []int) [][2]int {
isSep := make(map[int]bool, len(sepPages))
for _, p := range sepPages {
isSep[p] = true
}
var ranges [][2]int
start := 0
for p := 1; p <= pageCount; p++ {
if isSep[p] {
if start != 0 {
ranges = append(ranges, [2]int{start, p - 1})
start = 0
}
continue
}
if start == 0 {
start = p
}
}
if start != 0 {
ranges = append(ranges, [2]int{start, pageCount})
}
return ranges
}
// extractRange writes pages [first,last] of pdfPath into one PDF under
// partsDir, using pdfseparate (per-page extraction) plus pdfunite (re-merge)
// — the poppler-only equivalent of `qpdf --pages`.
func (d *Detector) extractRange(ctx context.Context, pdfPath, partsDir string, index, first, last int) (string, error) {
segDir := filepath.Join(partsDir, fmt.Sprintf("seg-%03d", index))
if err := os.MkdirAll(segDir, 0o750); err != nil {
return "", fmt.Errorf("create segment dir: %w", err)
}
sepCtx, cancelSep := context.WithTimeout(ctx, d.timeout())
defer cancelSep()
pattern := filepath.Join(segDir, "p-%d.pdf")
cmd := exec.CommandContext(sepCtx, d.pdfseparatePath(),
"-f", strconv.Itoa(first), "-l", strconv.Itoa(last), pdfPath, pattern)
var stderr bytes.Buffer
cmd.Stderr = &stderr
if err := cmd.Run(); err != nil {
return "", fmt.Errorf("pdfseparate failed: %w (%s)", err, strings.TrimSpace(stderr.String()))
}
pageFiles, err := sortedNumberedFiles(segDir, ".pdf")
if err != nil {
return "", err
}
want := last - first + 1
if len(pageFiles) != want {
return "", fmt.Errorf("pdfseparate produced %d pages, expected %d", len(pageFiles), want)
}
if len(pageFiles) == 1 {
// Single-page segment: the extracted page already IS the part.
return pageFiles[0], nil
}
uniteCtx, cancelUnite := context.WithTimeout(ctx, d.timeout())
defer cancelUnite()
outPath := filepath.Join(partsDir, fmt.Sprintf("part-%03d.pdf", index))
args := append(append([]string{}, pageFiles...), outPath)
uniteCmd := exec.CommandContext(uniteCtx, d.pdfunitePath(), args...)
var uniteErr bytes.Buffer
uniteCmd.Stderr = &uniteErr
if err := uniteCmd.Run(); err != nil {
os.Remove(outPath)
return "", fmt.Errorf("pdfunite failed: %w (%s)", err, strings.TrimSpace(uniteErr.String()))
}
if fi, statErr := os.Stat(outPath); statErr != nil || fi.Size() == 0 {
os.Remove(outPath)
return "", fmt.Errorf("pdfunite produced empty/missing output: %v", statErr)
}
return outPath, nil
}
// sortedNumberedFiles lists dir's files with the given extension, sorted by
// the trailing number in their name (numeric, not lexical).
func sortedNumberedFiles(dir, ext string) ([]string, error) {
entries, err := os.ReadDir(dir)
if err != nil {
return nil, fmt.Errorf("read dir %s: %w", dir, err)
}
type numbered struct {
path string
num int
}
var found []numbered
for _, entry := range entries {
if entry.IsDir() || !strings.HasSuffix(entry.Name(), ext) {
continue
}
num := 0
base := strings.TrimSuffix(entry.Name(), ext)
if m := pageNumRegex.FindStringSubmatch(base); m != nil {
num, _ = strconv.Atoi(m[1])
}
found = append(found, numbered{path: filepath.Join(dir, entry.Name()), num: num})
}
sort.Slice(found, func(i, j int) bool {
if found[i].num != found[j].num {
return found[i].num < found[j].num
}
return found[i].path < found[j].path
})
paths := make([]string, 0, len(found))
for _, f := range found {
paths = append(paths, f.path)
}
return paths, nil
}
// randomID returns a random hex string for scratch directory names. Kept
// dependency-free, same approach as internal/ocr.randomID.
func randomID() string {
b := make([]byte, 16)
if _, err := rand.Read(b); err != nil {
return fmt.Sprintf("job-%d", time.Now().UnixNano())
}
return hex.EncodeToString(b)
}
func (d *Detector) log(level slog.Level, msg string, args ...any) {
if d == nil || d.Logger == nil {
return
}
d.Logger.Log(context.Background(), level, msg, args...)
}
func (d *Detector) marker() string {
if strings.TrimSpace(d.Marker) == "" {
return DefaultMarker
}
return d.Marker
}
func (d *Detector) pdftoppmPath() string { return orDefault(d.PdftoppmPath, "pdftoppm") }
func (d *Detector) pdfinfoPath() string { return orDefault(d.PdfinfoPath, "pdfinfo") }
func (d *Detector) pdfseparatePath() string { return orDefault(d.PdfseparatePath, "pdfseparate") }
func (d *Detector) pdfunitePath() string { return orDefault(d.PdfunitePath, "pdfunite") }
func orDefault(v, def string) string {
if strings.TrimSpace(v) == "" {
return def
}
return v
}
func (d *Detector) tmpDir() string {
if strings.TrimSpace(d.TmpDir) == "" {
return os.TempDir()
}
return d.TmpDir
}
func (d *Detector) rasterDPI() int {
if d.RasterDPI <= 0 {
return defaultRasterDPI
}
return d.RasterDPI
}
func (d *Detector) maxPages() int {
if d.MaxPages <= 0 {
return defaultMaxPages
}
return d.MaxPages
}
func (d *Detector) timeout() time.Duration {
if d.Timeout <= 0 {
return defaultTimeout
}
return d.Timeout
}