// Package ocr wraps the system tesseract/poppler-utils binaries as a // best-effort text extraction sidecar for uploaded documents. Deliberately no // Go OCR binding is used — see dms-featureliste-prompt.md — this shells out // via os/exec, matching how the rest of archivdms avoids heavyweight native // dependencies. // // OCR failures (binary missing, timeout, non-zero exit) are never fatal to an // upload: Extract returns an error, and callers are expected to store the // document anyway with an empty ocr_text plus an audit warning. // // Extract also opportunistically decodes barcodes (via internal/barcode, // itself a zbarimg os/exec sidecar) from the same rasterized page images / // source image used for OCR, so ingest gets both text and barcode payloads // out of a single pass over the document. package ocr import ( "bytes" "context" "crypto/rand" "encoding/hex" "errors" "fmt" "image" "image/jpeg" "image/png" "log/slog" "os" "os/exec" "path/filepath" "regexp" "sort" "strconv" "strings" "time" "archivdms/internal/barcode" ) // Result is the outcome of Extract: the extracted text plus any barcode // payloads found along the way (raw values, unfiltered/unmatched — matching // against tags/document_types/correspondents happens in the ingest layer, // internal/api/document_handlers.go). type Result struct { Text string Barcodes []string // Words holds word-level bounding boxes for search-highlight/overlay use // (Phase 1 datengrundlage only — no persistence/API/frontend wiring yet, // see coords.go). Best-effort like the rest of this package: nil if TSV // extraction failed or produced nothing, even when Text is populated. Words []WordBox } // randomID returns a random hex string suitable as a scratch-directory name. // Kept dependency-free (no google/uuid in go.mod) — same approach as // cmd/archivdms/main.go's randomPassword(). func randomID() string { b := make([]byte, 16) if _, err := rand.Read(b); err != nil { // Extremely unlikely; fall back to a fixed-ish name, still unique // enough within a single process run given the deferred cleanup. return fmt.Sprintf("job-%d", time.Now().UnixNano()) } return hex.EncodeToString(b) } // Extractor extracts text from uploaded documents via tesseract (images) and // pdftotext/pdftoppm+tesseract (PDFs). type Extractor struct { TesseractPath string PdftoppmPath string ConvertPath string // SofficePath is the LibreOffice headless binary used to convert Office // documents to PDF before OCR (see convert.go). Empty defaults to "soffice". SofficePath string Languages string Timeout time.Duration // Binarize enables the Otsu auto-threshold step (binarizeImage) at the end // of the preprocessing pipeline in runTesseract, after deskew/contrast/OSD // rotation. Off by default (zero value) — set from // config.OCRConfig.BinarizeOCR (cmd/archivdms/main.go). Literature reports // 15-30% accuracy gains from Otsu thresholding on plain text scans, but it // is a lossy, irreversible step (hard black/white split) that can hurt // documents with color stamps/signatures or photos rather than flat text — // keep this switchable rather than hard-wired until it's been validated // against a broader corpus than the deskew/contrast tuning above. Binarize bool // DeskewMethod selects between "imagemagick" (default/empty — existing // deskewImage behavior) and "hough" (angle detection via the // hough_deskew.py sidecar, applied with a plain `convert -rotate`). // Set from config.OCRConfig.ResolvedDeskewMethod() (cmd/archivdms/main.go). // See runTesseract's deskew branch and houghDeskewAngle below. DeskewMethod string // PythonPath is the python3 binary used to run hough_deskew.py. Empty // defaults to "python3". Only consulted when DeskewMethod == "hough". PythonPath string // HoughDeskewScriptPath is the path to hough_deskew.py. Set from // config.OCRConfig.ResolvedHoughDeskewScriptPath(). Only consulted when // DeskewMethod == "hough". HoughDeskewScriptPath string // TmpDir is the base scratch directory for pdftoppm rasterization jobs // (config.StorageConfig.OCRTmpPath()). Each job gets its own // TmpDir// subdirectory, removed after use. TmpDir string // Logger receives best-effort preprocessing diagnostics (e.g. deskew // success/angle or the reason a step was skipped). Optional — nil disables // this logging; the OCR pipeline behaves identically either way. Set by // callers after New (cmd/archivdms/main.go). Logger *slog.Logger } // log emits at the given level via e.Logger, tolerating a nil logger (no-op). // Keeps every call site free of nil checks. func (e *Extractor) log(level slog.Level, msg string, args ...any) { if e.Logger == nil { return } e.Logger.Log(context.Background(), level, msg, args...) } // New creates an Extractor with the given settings. Callers typically build // this from config.OCRConfig + config.StorageConfig.OCRTmpPath(). func New(tesseractPath, pdftoppmPath, languages string, timeout time.Duration, tmpDir string) *Extractor { return &Extractor{ TesseractPath: tesseractPath, PdftoppmPath: pdftoppmPath, Languages: languages, Timeout: timeout, TmpDir: tmpDir, } } // minTextLenPDF is the threshold below which an already-existing PDF text // layer (from pdftotext) is considered "empty/too short" and we fall back to // rasterize+tesseract (i.e. the PDF is likely a pure scan with no text layer). const minTextLenPDF = 20 // Extract extracts text (and, best-effort, barcode payloads) from filePath, // dispatching on mimeType: // - image/* -> tesseract directly (+ barcode decode on the // original image) // - application/pdf -> pdftotext first, pdftoppm+tesseract fallback // (+ barcode decode on each rasterized page, PDF-text-layer path has no // images to decode from) // // Any other mimeType returns an error (unsupported), tolerated by callers. func (e *Extractor) Extract(ctx context.Context, filePath, mimeType string) (*Result, error) { switch { case strings.HasPrefix(mimeType, "image/"): return e.ocrImage(ctx, filePath) case mimeType == "application/pdf": return e.ocrPDF(ctx, filePath) case mimeType == "message/rfc822": // E-Mail: parse text directly, no rasterization/OCR needed. return e.extractEML(filePath) case isOfficeMime(mimeType): // Office document: convert to PDF via LibreOffice, then OCR that PDF. pdfPath, cleanup, err := e.officeToPDF(ctx, filePath) if err != nil { return nil, err } defer cleanup() return e.ocrPDF(ctx, pdfPath) default: return nil, fmt.Errorf("ocr: unsupported mime type %q", mimeType) } } // sofficePath returns the configured LibreOffice binary, defaulting to "soffice". func (e *Extractor) sofficePath() string { if strings.TrimSpace(e.SofficePath) == "" { return "soffice" } return e.SofficePath } // ocrImage runs tesseract directly against an image file and decodes // barcodes from the same original image (no rasterization needed). func (e *Extractor) ocrImage(ctx context.Context, filePath string) (*Result, error) { if _, err := exec.LookPath(e.tesseractPath()); err != nil { return nil, fmt.Errorf("ocr: tesseract not found in PATH: %w", err) } text, words, err := e.runTesseract(ctx, filePath) if err != nil { return nil, err } for i := range words { words[i].Page = 1 } // Wortboxen liegen jetzt im ROH-Pixelraum von filePath. Der Browser // rendert die Datei aber EXIF-orientiert (image-orientation: from-image // ist überall Default) — ohne diesen Schritt liegt das Overlay bei jedem // Handyfoto mit Orientation != 1 um 90/180 Grad verdreht bzw. komplett // außerhalb des Bildes. Siehe exif.go für die vollständige Herleitung. if len(words) > 0 { if orientation, oerr := jpegEXIFOrientation(filePath); orientation > 1 { rawW, rawH, derr := decodeImageDims(filePath) if derr == nil { applyEXIFOrientation(words, orientation, rawW, rawH) e.log(slog.LevelInfo, "ocr word boxes mapped into exif-oriented display space", "file", filePath, "exif_orientation", orientation, "raw_width", rawW, "raw_height", rawH, "words", len(words)) } else { e.log(slog.LevelWarn, "ocr exif orientation known but image dims unreadable, word boxes stay in raw space", "file", filePath, "exif_orientation", orientation, "err", derr) } } else if oerr != nil && !errors.Is(oerr, errNoEXIFOrientation) { e.log(slog.LevelWarn, "ocr exif orientation probe failed, assuming 1", "file", filePath, "err", oerr) } } codes, err := barcode.DecodeBarcodes(ctx, filePath) if err != nil { // Barcode decoding is best-effort only; never fails the OCR result. codes = nil } return &Result{Text: text, Barcodes: codes, Words: words}, nil } // ocrPDF first tries the cheap path (pdftotext, works when the PDF already // has a text layer) and only falls back to rasterize+tesseract when that // yields too little text (i.e. the PDF is a pure scan). Barcode decoding // only happens on the raster fallback path, since that's the only point a // page image exists to decode from. func (e *Extractor) ocrPDF(ctx context.Context, filePath string) (*Result, error) { if text, err := e.pdftotext(ctx, filePath); err == nil && len(strings.TrimSpace(text)) >= minTextLenPDF { return &Result{Text: text}, nil } if _, err := exec.LookPath(e.pdftoppmPath()); err != nil { return nil, fmt.Errorf("ocr: pdftoppm not found in PATH: %w", err) } if _, err := exec.LookPath(e.tesseractPath()); err != nil { return nil, fmt.Errorf("ocr: tesseract not found in PATH: %w", err) } return e.pdfRasterOCR(ctx, filePath) } // pdftotext tries to extract an existing text layer via `pdftotext -layout`. func (e *Extractor) pdftotext(ctx context.Context, filePath string) (string, error) { if _, err := exec.LookPath("pdftotext"); err != nil { return "", fmt.Errorf("ocr: pdftotext not found in PATH: %w", err) } cctx, cancel := context.WithTimeout(ctx, e.timeout()) defer cancel() cmd := exec.CommandContext(cctx, "pdftotext", "-layout", filePath, "-") var out, stderr bytes.Buffer cmd.Stdout = &out cmd.Stderr = &stderr if err := cmd.Run(); err != nil { return "", fmt.Errorf("ocr: pdftotext failed: %w (%s)", err, strings.TrimSpace(stderr.String())) } return out.String(), nil } // pdfRasterOCR rasterizes each PDF page to PNG via pdftoppm, then runs // tesseract over each page image, concatenating the results with a page // separator. func (e *Extractor) pdfRasterOCR(ctx context.Context, filePath string) (*Result, error) { jobID := randomID() jobDir := filepath.Join(e.tmpDir(), jobID) if err := os.MkdirAll(jobDir, 0o750); err != nil { return nil, fmt.Errorf("ocr: create job tmp dir: %w", err) } defer os.RemoveAll(jobDir) cctx, cancel := context.WithTimeout(ctx, e.timeout()) defer cancel() prefix := filepath.Join(jobDir, "page") cmd := exec.CommandContext(cctx, e.pdftoppmPath(), "-r", "300", "-png", filePath, prefix) var stderr bytes.Buffer cmd.Stderr = &stderr if err := cmd.Run(); err != nil { return nil, fmt.Errorf("ocr: pdftoppm failed: %w (%s)", err, strings.TrimSpace(stderr.String())) } entries, err := os.ReadDir(jobDir) if err != nil { return nil, fmt.Errorf("ocr: read job tmp dir: %w", err) } var pages []string for _, entry := range entries { if !strings.HasSuffix(entry.Name(), ".png") { continue } pages = append(pages, filepath.Join(jobDir, entry.Name())) } sort.Strings(pages) if len(pages) == 0 { return nil, fmt.Errorf("ocr: pdftoppm produced no page images") } var sb strings.Builder var barcodes []string var allWords []WordBox for i, page := range pages { if i > 0 { sb.WriteString("\n\f\n") // form-feed page separator } text, words, err := e.runTesseract(ctx, page) if err != nil { return nil, fmt.Errorf("ocr: tesseract on page %d: %w", i+1, err) } sb.WriteString(text) // Words here are in the rasterized page PNG's pixel space (mapped // back through this page's own preprocessing chain already), NOT // further mapped into PDF point/MediaBox coordinate space — see the // "PDF scope note" in coords.go's package doc comment. for j := range words { words[j].Page = i + 1 } allWords = append(allWords, words...) // Decode barcodes from the already-rasterized page PNG, before // jobDir is removed by the deferred cleanup above. if codes, err := barcode.DecodeBarcodes(ctx, page); err == nil { barcodes = append(barcodes, codes...) } } return &Result{Text: sb.String(), Barcodes: barcodes, Words: allWords}, nil } // runTesseract runs tesseract against filePath, first attempting to detect // and physically correct page rotation, then falling back to `--psm 1` // (automatic layout + OSD) and finally plain default-psm recognition. // // Why the explicit rotate-then-OCR step exists (dms doc id 7, a sideways // phone photo of a thermal receipt): --psm 1 asks tesseract to detect // orientation *and* rotate internally in one pass, but tesseract only // applies that internal rotation when its own OSD confidence is high. // Real-world phone-camera photos of small/low-contrast receipts routinely // come back with "Weak margin" / confidence well under 1.0 — tesseract logs // the guess (e.g. "Rotate: 90") but then silently keeps reading the // unrotated pixels, producing garbled text instead of an error we could // catch. Explicitly running OSD (--psm 0), physically rotating the image // ourselves with the stdlib image package (no CGO, no extra sidecar // binary), and only then running normal recognition on the corrected image // sidesteps that internal confidence gate entirely. func (e *Extractor) runTesseract(ctx context.Context, filePath string) (string, []WordBox, error) { origPath := filePath chain := &geomChain{log: e.log} if clampedPath, cleanup, ok := e.clampImageSize(ctx, filePath); ok { defer cleanup() chain.recordScale("clamp", filePath, clampedPath) filePath = clampedPath } if strings.EqualFold(e.DeskewMethod, "hough") { if deskewedPath, cleanup, angleDeg, ok := e.deskewImageHough(ctx, filePath); ok { defer cleanup() chain.recordRotation("deskew-hough", filePath, deskewedPath, angleDeg, false) filePath = deskewedPath } } else if deskewedPath, cleanup, angleDeg, ok := e.deskewImage(ctx, filePath); ok { defer cleanup() chain.recordRotation("deskew-imagemagick", filePath, deskewedPath, angleDeg, false) filePath = deskewedPath } if contrastPath, cleanup, ok := e.normalizeContrast(ctx, filePath); ok { defer cleanup() // -colorspace Gray -normalize should never change pixel dimensions, // but that is verified rather than assumed (recordScale drops the // identity case for free and catches a surprising resize instead of // silently misplacing every word box). chain.recordScale("contrast", filePath, contrastPath) filePath = contrastPath } if rotatedPath, cleanup, degreesApplied, ok := e.rotateForOSD(ctx, filePath); ok { defer cleanup() chain.recordRotation("osd-rotate", filePath, rotatedPath, float64(degreesApplied), true) filePath = rotatedPath } if e.Binarize { if binarizedPath, cleanup, ok := e.binarizeImage(ctx, filePath); ok { defer cleanup() // -auto-threshold Otsu should never change pixel dimensions, same // verify-don't-assume reasoning as normalizeContrast above. chain.recordScale("otsu", filePath, binarizedPath) filePath = binarizedPath } } text, err := e.runTesseractWithFallback(ctx, filePath) if err == nil && strings.TrimSpace(text) != "" { transforms, ok := chain.transforms() if !ok { // Chain incomplete: emitting boxes now would put a visibly wrong // overlay on the document. No boxes is the safer failure mode. return text, nil, nil } words := e.extractWords(ctx, filePath, transforms) return text, words, nil } // Layered text recovery (Docspell-Muster): the hardened preprocessing // pipeline above (clamp/deskew/contrast/rotate) either errored or yielded // zero characters. Before giving up, retry OCR once on the raw, // unpreprocessed file — a preprocessing step (e.g. an over-aggressive // deskew/rotate on a marginal image, or a `convert` producing a technically // valid but degenerate output) can occasionally turn a readable source into // an empty/failed OCR result, in which case the untouched original still // recovers text. If no preprocessing actually ran (filePath unchanged), // there is nothing new to try and we return the primary result as-is (no // redundant second pass over the identical file). if filePath == origPath { return text, nil, err } rawText, rawErr := e.runTesseractWithFallback(ctx, origPath) if rawErr == nil && strings.TrimSpace(rawText) != "" { // Ran directly against the untouched original: no transforms to invert. words := e.extractWords(ctx, origPath, nil) return rawText, words, nil } // Raw fallback did not improve on the primary result: preserve the original // outcome (empty-but-no-error, or the more informative primary error). return text, nil, err } // extractWords runs TSV word-box extraction against filePath and maps the // result back into the pre-preprocessing coordinate space via transforms. // Best-effort: any failure just logs a warning and yields no word boxes, // same tolerance as the rest of the preprocessing pipeline — a missing // Result.Words must never fail an otherwise-successful OCR run. func (e *Extractor) extractWords(ctx context.Context, filePath string, transforms []geomTransform) []WordBox { words, err := e.wordBoxesWithFallback(ctx, filePath) if err != nil { e.log(slog.LevelWarn, "ocr tsv word-box extraction failed", "file", filePath, "err", err) return nil } mapWordsToOriginal(words, transforms) return words } // convertStep runs `convert ` for one preprocessing step // and is the single implementation of the boilerplate every ImageMagick-based // step in this file used to repeat verbatim: binary lookup, extension- // preserving scratch filename, timeout context, stderr capture, removal of a // partial output on failure, empty-output check, and a cleanup closure. // // It exists because there are now five such steps (clamp, deskew, hough-rotate, // contrast, otsu) whose only real differences are the argument list and how // loudly they log — before this helper each one carried its own slightly // divergent copy of the same 35 lines, which is exactly how the "skipped step // but kept the coordinate transform" class of bug creeps in. // // Contract, identical for every caller: ok == false means "this step did not // happen, proceed with the unmodified input file" — never an OCR abort. quiet // suppresses the warn-level failure logs for steps that are routine no-ops // (clamp, contrast) rather than something an operator should see. func (e *Extractor) convertStep(ctx context.Context, filePath, namePrefix, stepLabel string, quiet bool, args ...string) (dstPath string, cleanup func(), stdout string, ok bool) { if _, err := exec.LookPath(e.convertPath()); err != nil { if !quiet { e.log(slog.LevelWarn, "ocr "+stepLabel+" skipped: convert (ImageMagick) binary not found in PATH", "file", filePath, "convert", e.convertPath(), "err", err) } return "", nil, "", false } ext := filepath.Ext(filePath) if ext == "" { ext = ".png" } dst := filepath.Join(e.tmpDir(), namePrefix+"-"+randomID()+ext) cctx, cancel := context.WithTimeout(ctx, e.timeout()) defer cancel() cmdArgs := append([]string{filePath}, args...) cmdArgs = append(cmdArgs, dst) cmd := exec.CommandContext(cctx, e.convertPath(), cmdArgs...) var out, stderr bytes.Buffer cmd.Stdout = &out cmd.Stderr = &stderr if err := cmd.Run(); err != nil { os.Remove(dst) // best-effort cleanup of any partial output reason := "failed" if cctx.Err() == context.DeadlineExceeded { reason = "timed out" } if !quiet { e.log(slog.LevelWarn, "ocr "+stepLabel+" "+reason, "file", filePath, "err", err, "stderr", strings.TrimSpace(stderr.String())) } return "", nil, "", false } if fi, err := os.Stat(dst); err != nil || fi.Size() == 0 { os.Remove(dst) if !quiet { e.log(slog.LevelWarn, "ocr "+stepLabel+" produced empty/missing output, skipping", "file", filePath, "dst", dst, "stat_err", err) } return "", nil, "", false } return dst, func() { os.Remove(dst) }, strings.TrimSpace(out.String()), true } // maxImagePixels is a safety ceiling (in total pixels) above which images // are downscaled before OCR, analogous to Paperless-ngx's // OCR_MAX_IMAGE_PIXELS guard. Deliberately generous (well above the ~12MP // 4000x3000 phone photos that make up the current test corpus, dms doc ids // 4-9, all of which are left untouched by this guard) — this exists purely // to bound worst-case tesseract runtime/memory on pathological uploads // (e.g. a 48MP+ phone photo or an accidentally huge scan), not as an // accuracy tweak. Routine downscaling to a fixed target size (tested at // ~2500px longest edge against the same corpus) was tried and rejected: it // measurably lost small receipt-text detail (a whole header line vanished // on doc id 6) and reduced tesseract output length on most docs — the // current resolution is already helping, not hurting. const maxImagePixels = "30000000@>" // clampImageSize runs `convert -resize 30000000@> `, which is a // no-op unless the image exceeds ~30 megapixels (ImageMagick's `>` modifier // only shrinks, never enlarges). Best-effort like the other preprocessing // steps: failures just skip this step. Placed before deskewImage/ // normalizeContrast so any downscaling happens once, up front, rather than // each step operating on full-resolution pixels unnecessarily. func (e *Extractor) clampImageSize(ctx context.Context, filePath string) (dstPath string, cleanup func(), ok bool) { dst, cleanupFn, _, ok := e.convertStep(ctx, filePath, "clamp", "clamp resize", true, "-resize", maxImagePixels) return dst, cleanupFn, ok } // Despeckle/blur-correction and unpaper were evaluated against the same // test corpus (dms doc ids 4-9, tenant 3, all 4000x3000 phone photos of the // same receipt) and deliberately NOT added as pipeline steps: // // - ImageMagick -despeckle: fixed a wrong postal code on doc id 6 // ("33116" -> correct "39116") but caused catastrophic garbling // (mirrored/upside-down-looking output) on doc ids 5, 8 and 9 — tesseract's // internal --psm 1 orientation detection got confused by the added // noise-removal artifacts on already-marginal images. Net effect across // the corpus is negative; the risk of turning a mediocre result into a // garbage one outweighs the occasional improvement. // - ImageMagick -unsharp 0x1.0: similarly destabilized layout/orientation // detection on doc id 6 (scrambled reading order, duplicated content) // without a clear win elsewhere. // - ImageMagick -median 1: neutral to slightly negative, no measurable // benefit over normalizeContrast alone. // - unpaper (installed and tested on 192.168.1.204, version 7.0.0): built // for scanned book/document pages (border/margin detection, black-area // removal) rather than already-tightly-cropped phone photos of small // receipts; even with border/mask/black filters disabled it reduced // tesseract output length on every test doc (e.g. doc id 4: 832 -> 521 // chars) and is not installed as a service dependency. // // If OCR quality issues resurface on genuinely blurry/noisy source material // (as opposed to skew/contrast/orientation, already handled above), re-run // this comparison against the specific failing documents before reaching // for these again — do not re-add them blindly based on this note alone. // deskewThreshold is the ImageMagick `-deskew` threshold. ImageMagick finds // the smallest rotation that makes the image's background pixels (within // this percentage of full white/black) fall on straight rows/columns, then // rotates by that amount to straighten it. 40% was verified against several // of the known-problematic phone-photographed receipts (dms doc ids 2/7/8): // it visibly straightens text lines and measurably improves tesseract // output, while 80% over-rotated a low-contrast receipt and made it worse. const deskewThreshold = "40%" // deskewImage runs `convert -deskew 40% ` to correct small-angle // skew (a few degrees off horizontal, typical of hand-scanned or // phone-photographed documents) BEFORE the OSD-based 90/180-degree rotation // pass below: OSD only ever reports rotation in 90-degree steps, so it // cannot fix fine skew, and deskewing first also improves OSD's own // orientation-confidence reading on marginal pages. Best-effort like the // rest of this package — if `convert` is missing, times out, or fails, ok is // false and callers proceed with the original (non-deskewed) file rather // than aborting the OCR pass. Never touches the archived original: filePath // here is always an already-copied working file (the source image itself, // or a pdftoppm-rasterized page under TmpDir), consistent with the WORM // rule that OCR only ever reads from store/. func (e *Extractor) deskewImage(ctx context.Context, filePath string) (dstPath string, cleanup func(), angleDeg float64, ok bool) { // `-print "%[deskew:angle]\n"` makes ImageMagick emit the rotation angle it // computed (in degrees) to stdout while processing, so we can log which // correction was actually applied and, more importantly, invert it in // coords.go. Best-effort: if the build/version does not populate that // property, stdout is empty. dst, cleanupFn, angleStr, ok := e.convertStep(ctx, filePath, "deskew", "deskew", false, "-deskew", deskewThreshold, "-print", "%[deskew:angle]\n") if !ok { return "", nil, 0, false } // A missing or unparseable angle (older ImageMagick build not populating // the property) is reported as 0 degrees. geomChain.recordRotation then // detects "angle 0 but dimensions changed" and suppresses this document's // word boxes entirely rather than emitting a misaligned overlay — see // coords.go. The OCR text itself is unaffected. angle, parseErr := strconv.ParseFloat(angleStr, 64) if parseErr != nil { angle = 0 } angleLog := angleStr if angleLog == "" { angleLog = "unknown" } e.log(slog.LevelInfo, "ocr deskew applied", "file", filePath, "threshold", deskewThreshold, "angle_deg", angleLog) return dst, cleanupFn, angle, true } // houghDeskewAngle runs the hough_deskew.py sidecar (python3 + OpenCV) against // filePath and returns the detected skew angle in degrees (ImageMagick // `-rotate` sign convention). Best-effort like every other preprocessing step // in this package: a missing python3/script, a missing opencv-python // dependency, a timeout, or a script that found nothing all result in ok == // false, angle 0 — callers must proceed with the unrotated image rather than // aborting OCR. Never touches the archived original; filePath is always an // already-copied working file under TmpDir, same WORM reasoning as // deskewImage. func (e *Extractor) houghDeskewAngle(ctx context.Context, filePath string) (angleDeg float64, ok bool) { pythonPath := e.pythonPath() if _, err := exec.LookPath(pythonPath); err != nil { e.log(slog.LevelWarn, "ocr hough deskew skipped: python3 binary not found in PATH", "file", filePath, "python", pythonPath, "err", err) return 0, false } scriptPath := e.houghDeskewScriptPath() if _, err := os.Stat(scriptPath); err != nil { e.log(slog.LevelWarn, "ocr hough deskew skipped: hough_deskew.py not found", "file", filePath, "script", scriptPath, "err", err) return 0, false } cctx, cancel := context.WithTimeout(ctx, e.timeout()) defer cancel() cmd := exec.CommandContext(cctx, pythonPath, scriptPath, filePath) var stdout, stderr bytes.Buffer cmd.Stdout = &stdout cmd.Stderr = &stderr if err := cmd.Run(); err != nil { reason := "hough_deskew.py failed" if cctx.Err() == context.DeadlineExceeded { reason = "hough_deskew.py timed out" } e.log(slog.LevelWarn, "ocr "+reason, "file", filePath, "script", scriptPath, "err", err, "stderr", strings.TrimSpace(stderr.String())) return 0, false } angleStr := strings.TrimSpace(stdout.String()) angle, parseErr := strconv.ParseFloat(angleStr, 64) if parseErr != nil { e.log(slog.LevelWarn, "ocr hough deskew produced unparseable angle, skipping", "file", filePath, "stdout", angleStr, "err", parseErr) return 0, false } e.log(slog.LevelInfo, "ocr hough deskew angle detected", "file", filePath, "angle_deg", angle) return angle, true } // deskewImageHough is the "hough" DeskewMethod counterpart to deskewImage: it // detects the skew angle via houghDeskewAngle (Python/OpenCV sidecar, angle // detection only) and then applies that rotation with a plain // `convert -rotate ` — ImageMagick used purely as a rotation tool here, // never its own `-deskew` angle-detection heuristic. This split is the whole // point of the hough method: `-deskew`'s peak/valley background-projection // analysis needs page margin to find background rows/columns and fails on // tightly-cropped phone photos (no margin context); minAreaRect/HoughLinesP // in hough_deskew.py degrades more gracefully when content fills the frame. // Best-effort like deskewImage: any failure (angle detection failed, convert // missing/failed/timed out, empty output) yields ok == false and callers // proceed with the original, unrotated file. func (e *Extractor) deskewImageHough(ctx context.Context, filePath string) (dstPath string, cleanup func(), angleDeg float64, ok bool) { angle, angleOk := e.houghDeskewAngle(ctx, filePath) if !angleOk { return "", nil, 0, false } // A near-zero angle means the image is already straight — skip the // rotate round-trip (and the log noise) entirely, same treatment // deskewImage's ImageMagick angle gets implicitly (a 0-degree // `-deskew` result is a no-op rotation). if angle > -0.05 && angle < 0.05 { e.log(slog.LevelInfo, "ocr hough deskew: angle negligible, skipping rotate", "file", filePath, "angle_deg", angle) return "", nil, 0, false } dst, cleanupFn, _, ok := e.convertStep(ctx, filePath, "hough-deskew", "hough deskew rotate", false, "-rotate", strconv.FormatFloat(angle, 'f', -1, 64)) if !ok { return "", nil, 0, false } e.log(slog.LevelInfo, "ocr hough deskew applied", "file", filePath, "angle_deg", angle, "dst", dst) return dst, cleanupFn, angle, true } // normalizeContrast runs `convert -colorspace Gray -normalize ` to // clean up low-contrast/low-light phone photos of receipts (typical single- // character misreads like "Kittwegei" for "Rittweger", "»erVice" for // "Service") before tesseract sees the image. Runs after deskewImage and // before rotateForOSD in runTesseract, same ordering rationale as deskew: // improving image quality first also improves OSD's own orientation // confidence. -normalize stretches the tonal range to fill black-to-white // per channel, which is enough for these compressed/dim photos — no // -contrast-stretch or -sharpen, tested against dms doc id 5 on // 192.168.1.204 and found to over- or under-correct relative to plain // -normalize. Best-effort like deskewImage: any failure (missing binary, // timeout, non-zero exit) just skips this step and callers proceed with the // original file. Never touches the archived original, same WORM reasoning // as deskewImage. func (e *Extractor) normalizeContrast(ctx context.Context, filePath string) (dstPath string, cleanup func(), ok bool) { dst, cleanupFn, _, ok := e.convertStep(ctx, filePath, "contrast", "contrast normalize", true, "-colorspace", "Gray", "-normalize") return dst, cleanupFn, ok } // binarizeImage runs `convert -auto-threshold Otsu `, applying // Otsu's method to pick a single global black/white threshold from the // image's grayscale histogram. Gated behind e.Binarize (config.OCRConfig. // BinarizeOCR) — unlike deskew/normalizeContrast this is a lossy, one-way // step (every pixel becomes pure black or white), which helps flat text // scans but can destroy information on documents with color stamps, // signatures, or embedded photos, so it must stay easy to disable if it // turns out to hurt in practice (see clampImageSize's despeckle/sharpen/ // unpaper/resize history above — preprocessing changes need field // validation, not just plausibility). Runs last in runTesseract, after // deskewImage/normalizeContrast/rotateForOSD: Otsu needs the deskewed, // contrast-normalized, upright image as input (a skewed or still-rotated // binarized image loses the grayscale information OSD/deskew rely on), and // OSD's own orientation confidence is measured against the grayscale image, // not a pre-binarized one. -auto-threshold Otsu requires ImageMagick 7; // verify the installed version on the target host before enabling by // default (`convert -version` / `magick -version`) — on IM6 hosts this // invocation fails and the best-effort skip below simply leaves OCR running // on the pre-Otsu (grayscale-normalized) image, same degrade-gracefully // behavior as every other step in this pipeline. Best-effort like the other // preprocessing steps: any failure (missing binary, unsupported // -auto-threshold syntax, timeout, non-zero exit) just skips this step and // callers proceed with the pre-Otsu file. Never touches the archived // original, same WORM reasoning as deskewImage/normalizeContrast. func (e *Extractor) binarizeImage(ctx context.Context, filePath string) (dstPath string, cleanup func(), ok bool) { dst, cleanupFn, _, ok := e.convertStep(ctx, filePath, "otsu", "otsu binarize", false, "-auto-threshold", "Otsu") if !ok { return "", nil, false } e.log(slog.LevelInfo, "ocr otsu binarize applied", "file", filePath, "dst", dst) return dst, cleanupFn, true } // runTesseractWithFallback runs `tesseract stdout -l // --psm 1` with a timeout. --psm 1 ("Automatic page segmentation with OSD") // makes tesseract run Orientation and Script Detection on the raw pixel // content before recognition and rotate internally as needed; kept as the // primary attempt here since it still helps on pages runTesseract's own OSD // pass didn't flag (e.g. OSD itself failed on very sparse pages). // osd.traineddata must be installed alongside the language data // (tesseract-ocr-osd package) for --psm 1 to work; it is present on the // current deployment. // // OSD needs a reasonable amount of text on the page to determine // orientation confidently and can fail outright on sparse/mostly-blank // pages ("Too few characters..."). If the --psm 1 run fails, we retry once // with tesseract's default page segmentation (no --psm flag) so pages that // worked fine before this change don't regress into hard OCR failures. func (e *Extractor) runTesseractWithFallback(ctx context.Context, filePath string) (string, error) { text, err := e.runTesseractArgs(ctx, filePath, "--psm", "1") if err == nil { return text, nil } text, fallbackErr := e.runTesseractArgs(ctx, filePath) if fallbackErr != nil { // Report the original (OSD) error, it's usually more informative. return "", err } return text, nil } // minOSDConfidence is the minimum tesseract orientation-confidence value we // trust enough to physically rotate the image ourselves. // // Originally set deliberately very low (0.05, just above zero) on the theory // that a wrong Rotate value at very low confidence is no worse than not // rotating at all. That theory did not hold up: a later comparison of 4 // near-identical test scans (tenant 3, dms doc ids 3/4/5/7) showed OSD // confidence values scattered between 0.06 and 5.57 on essentially the same // input, with rotation applied inconsistently across them — i.e. at that // range the confidence value is closer to noise than signal, and trusting it // produced effectively random 90/180-degree rotations, causing the OCR // output quality to swing wildly between otherwise-similar documents. // Threshold raised to 6.0, just above the observed noise band, so only // confidence readings clearly above that range trigger a physical rotation. const minOSDConfidence = 6.0 var ( osdRotateRegex = regexp.MustCompile(`Rotate:\s*(\d+)`) osdConfidenceRegex = regexp.MustCompile(`Orientation confidence:\s*([\d.]+)`) ) // rotateForOSD runs `tesseract - --psm 0` (orientation/script // detection only, no recognition) against filePath, and if it reports a // non-zero rotation with at least minOSDConfidence confidence, physically // rotates the image via rotateImageFile and returns the path to the // rotated copy plus a cleanup func. ok is false whenever OSD produced // nothing actionable (binary/OSD data missing, too little content to // analyze, rotation already 0, or confidence too low) — callers should // fall back to their normal recognition path on the original file in that // case, this is a best-effort optimization only. func (e *Extractor) rotateForOSD(ctx context.Context, filePath string) (rotatedPath string, cleanup func(), degreesApplied int, ok bool) { cctx, cancel := context.WithTimeout(ctx, e.timeout()) defer cancel() cmd := exec.CommandContext(cctx, e.tesseractPath(), filePath, "-", "--psm", "0") var out bytes.Buffer cmd.Stdout = &out cmd.Stderr = &out // OSD info can land on either stream; we only parse text. if err := cmd.Run(); err != nil { return "", nil, 0, false } report := out.String() rm := osdRotateRegex.FindStringSubmatch(report) cm := osdConfidenceRegex.FindStringSubmatch(report) if rm == nil || cm == nil { return "", nil, 0, false } degrees, err := strconv.Atoi(rm[1]) if err != nil || degrees == 0 { return "", nil, 0, false } confidence, err := strconv.ParseFloat(cm[1], 64) if err != nil || confidence < minOSDConfidence { e.log(slog.LevelInfo, "ocr osd rotation skipped (confidence below threshold)", "file", filePath, "rotate_deg", degrees, "confidence", cm[1], "min_confidence", minOSDConfidence, "rotated", false) return "", nil, 0, false } dstPath, err := rotateImageFile(filePath, degrees, e.tmpDir()) if err != nil { e.log(slog.LevelInfo, "ocr osd rotation failed", "file", filePath, "rotate_deg", degrees, "confidence", confidence, "rotated", false, "err", err) return "", nil, 0, false } e.log(slog.LevelInfo, "ocr osd rotation applied", "file", filePath, "rotate_deg", degrees, "confidence", confidence, "rotated", true) return dstPath, func() { os.Remove(dstPath) }, degrees, true } // rotateImageFile decodes the JPEG/PNG at srcPath, rotates it clockwise by // degrees (must be a multiple of 90 — tesseract's OSD only ever reports // 0/90/180/270), and writes the result to a new file under tmpDir, // preserving the original format. Pure stdlib image/jpeg + image/png, no // CGO and no extra sidecar binary — consistent with the rest of this // package's "shell out to tesseract/poppler only" approach, since rotation // is cheap enough to do in-process. func rotateImageFile(srcPath string, degrees int, tmpDir string) (string, error) { f, err := os.Open(srcPath) if err != nil { return "", err } img, format, err := image.Decode(f) f.Close() if err != nil { return "", err } rotated := img for i := 0; i < (degrees/90)%4; i++ { rotated = rotate90CW(rotated) } ext := ".jpg" if format == "png" { ext = ".png" } dstPath := filepath.Join(tmpDir, "rot-"+randomID()+ext) of, err := os.Create(dstPath) if err != nil { return "", err } defer of.Close() if format == "png" { err = png.Encode(of, rotated) } else { err = jpeg.Encode(of, rotated, &jpeg.Options{Quality: 92}) } if err != nil { os.Remove(dstPath) return "", err } return dstPath, nil } // rotate90CW rotates an image 90 degrees clockwise into a new RGBA image. func rotate90CW(src image.Image) *image.RGBA { b := src.Bounds() w, h := b.Dx(), b.Dy() dst := image.NewRGBA(image.Rect(0, 0, h, w)) for y := 0; y < h; y++ { for x := 0; x < w; x++ { dst.Set(h-1-y, x, src.At(b.Min.X+x, b.Min.Y+y)) } } return dst } func (e *Extractor) runTesseractArgs(ctx context.Context, filePath string, extraArgs ...string) (string, error) { cctx, cancel := context.WithTimeout(ctx, e.timeout()) defer cancel() args := append([]string{filePath, "stdout", "-l", e.languages()}, extraArgs...) cmd := exec.CommandContext(cctx, e.tesseractPath(), args...) var out, stderr bytes.Buffer cmd.Stdout = &out cmd.Stderr = &stderr if err := cmd.Run(); err != nil { return "", fmt.Errorf("ocr: tesseract failed: %w (%s)", err, strings.TrimSpace(stderr.String())) } return out.String(), nil } // wordBoxesWithFallback runs `tesseract stdout -l tsv` // (word-level bounding boxes, tesseract's TSV output mode) and parses the // result into []WordBox. Mirrors runTesseractWithFallback's --psm 1 / default // retry pattern so word-box extraction sees the same page-segmentation // behavior as the text extraction it accompanies — a marginal page that // needed the default-psm fallback for readable text should get the same // fallback for boxes. func (e *Extractor) wordBoxesWithFallback(ctx context.Context, filePath string) ([]WordBox, error) { raw, err := e.runTesseractArgs(ctx, filePath, "--psm", "1", "tsv") if err == nil { words := parseTesseractTSV(raw) if len(words) > 0 { return words, nil } } raw, fallbackErr := e.runTesseractArgs(ctx, filePath, "tsv") if fallbackErr != nil { if err != nil { return nil, err } return nil, fallbackErr } return parseTesseractTSV(raw), nil } // parseTesseractTSV parses tesseract's TSV output mode into []WordBox. // Columns (tab-separated, header row first): level, page_num, block_num, // par_num, line_num, word_num, left, top, width, height, conf, text. Only // level==5 rows (individual words) are kept; rows with empty/whitespace-only // text or conf<0 (tesseract emits -1 for non-word levels, but occasionally // also for boxes with no recognized text) are skipped. Best-effort: a // malformed row is silently skipped rather than aborting the whole page. func parseTesseractTSV(raw string) []WordBox { lines := strings.Split(raw, "\n") if len(lines) < 2 { return nil } var words []WordBox for _, line := range lines[1:] { line = strings.TrimRight(line, "\r") if line == "" { continue } cols := strings.Split(line, "\t") if len(cols) < 12 { continue } level, err := strconv.Atoi(cols[0]) if err != nil || level != 5 { continue } text := cols[11] if strings.TrimSpace(text) == "" { continue } block, _ := strconv.Atoi(cols[2]) par, _ := strconv.Atoi(cols[3]) lineNum, _ := strconv.Atoi(cols[4]) left, err1 := strconv.Atoi(cols[6]) top, err2 := strconv.Atoi(cols[7]) width, err3 := strconv.Atoi(cols[8]) height, err4 := strconv.Atoi(cols[9]) if err1 != nil || err2 != nil || err3 != nil || err4 != nil { continue } conf, _ := strconv.ParseFloat(cols[10], 64) words = append(words, WordBox{ Text: text, Left: left, Top: top, Width: width, Height: height, Confidence: conf, Line: lineNum, Block: block, Par: par, }) } return words } func (e *Extractor) tesseractPath() string { if strings.TrimSpace(e.TesseractPath) == "" { return "tesseract" } return e.TesseractPath } func (e *Extractor) pdftoppmPath() string { if strings.TrimSpace(e.PdftoppmPath) == "" { return "pdftoppm" } return e.PdftoppmPath } func (e *Extractor) convertPath() string { if strings.TrimSpace(e.ConvertPath) == "" { return "convert" } return e.ConvertPath } func (e *Extractor) pythonPath() string { if strings.TrimSpace(e.PythonPath) == "" { return "python3" } return e.PythonPath } func (e *Extractor) houghDeskewScriptPath() string { if strings.TrimSpace(e.HoughDeskewScriptPath) == "" { return "/opt/archivdms/scripts/hough_deskew.py" } return e.HoughDeskewScriptPath } func (e *Extractor) languages() string { if strings.TrimSpace(e.Languages) == "" { return "deu+eng" } return e.Languages } func (e *Extractor) timeout() time.Duration { if e.Timeout <= 0 { return 60 * time.Second } return e.Timeout } func (e *Extractor) tmpDir() string { if strings.TrimSpace(e.TmpDir) == "" { return os.TempDir() } return e.TmpDir }