Files
archivdms/internal/storage/metadata_suggestions_ollama.go
patrick 9a24ea29e1 FDN-01: repository & projektgerüst
Git-Repository für bestehenden archivdms-Code initialisiert, Branch-/Commit-Konvention (feature/<ticket>-<slug>-Branches, Ticket-Prefix in Commit-Nachricht) etabliert.
2026-08-11 21:27:53 +02:00

216 lines
7.9 KiB
Go

package storage
import (
"context"
"encoding/json"
"fmt"
"sort"
"strings"
"time"
"archivdms/internal/llm"
"archivdms/internal/matching"
)
// maxOllamaOCRChars caps how much OCR text is fed into the prompt. The target
// model is small (qwen2.5:1.5b, ~4GB-RAM server) with a limited context window,
// so the first chunk of the document plus the taxonomy lists is all it sees.
const maxOllamaOCRChars = 2000
// ollamaNameMatchFloor is the minimum fuzzy score at which an LLM-returned name
// is accepted as referring to an existing taxonomy entity. The LLM only knows
// names, never IDs, so its answers are mapped back to entities by name; anything
// below this is treated as a hallucinated / non-existent entity and dropped.
const ollamaNameMatchFloor = 0.8
// ollamaSuggestionResponse is the JSON schema the model is asked to fill. It
// deliberately uses plain name lists (not the ID-bearing SuggestionCandidate
// shape) because the LLM has no knowledge of internal IDs — names are mapped
// back to entities afterwards.
type ollamaSuggestionResponse struct {
Title string `json:"title"`
DocTypes []string `json:"doc_types"`
Correspondents []string `json:"correspondents"`
Tags []string `json:"tags"`
}
// GenerateOllamaSuggestions asks an EXTERNAL Ollama server (per-tenant config)
// to propose metadata for a document and persists the result as a
// metadata_suggestions row with provider='ollama', in the SAME SuggestionPayload
// schema the heuristic provider produces (so the API/frontend are unchanged).
//
// The prompt contains the document title, a truncated slice of its OCR text and
// the tenant's existing tags/document_types/correspondents (by name) so the
// model reuses known entities. The model returns names; those are mapped back to
// entity IDs by exact-then-fuzzy name match. Any Ollama error (unreachable,
// timeout, invalid JSON) is returned as-is — NO silent fallback to heuristic
// (GoBD-Nachvollziehbarkeit: the caller reports which provider failed).
func (s *Store) GenerateOllamaSuggestions(ctx context.Context, documentID, tenantID int64, requestedBy *int64, cfg OllamaConfig) (*MetadataSuggestion, error) {
if !cfg.Enabled {
return nil, fmt.Errorf("storage: ollama provider not enabled for tenant")
}
doc, err := s.GetDocument(ctx, documentID, tenantID)
if err != nil {
return nil, err // ErrDocumentNotFound propagates
}
tags, err := s.ListTaxonomyEntities(ctx, "tags", tenantID)
if err != nil {
return nil, err
}
docTypes, err := s.ListTaxonomyEntities(ctx, "document_types", tenantID)
if err != nil {
return nil, err
}
correspondents, err := s.ListTaxonomyEntities(ctx, "correspondents", tenantID)
if err != nil {
return nil, err
}
prompt := buildOllamaPrompt(doc, tags, docTypes, correspondents)
raw, err := llm.GenerateJSON(ctx, cfg.BaseURL, cfg.Model, time.Duration(cfg.TimeoutSeconds)*time.Second, prompt)
if err != nil {
return nil, fmt.Errorf("storage: ollama generate: %w", err)
}
var parsed ollamaSuggestionResponse
if err := json.Unmarshal(raw, &parsed); err != nil {
return nil, fmt.Errorf("storage: ollama response does not match expected schema: %w", err)
}
// Entities already assigned are excluded from the suggestions, matching the
// heuristic provider's behaviour.
assignedTags := map[int64]bool{}
docTags, err := s.ListDocumentTags(ctx, documentID, tenantID)
if err != nil {
return nil, err
}
for _, t := range docTags {
assignedTags[t.ID] = true
}
payload := SuggestionPayload{
DocTypeCandidates: mapNamesToCandidates(parsed.DocTypes, docTypes, func(id int64) bool {
return doc.DocTypeID != nil && *doc.DocTypeID == id
}),
CorrespondentCandidates: mapNamesToCandidates(parsed.Correspondents, correspondents, func(id int64) bool {
return doc.CorrespondentID != nil && *doc.CorrespondentID == id
}),
TagCandidates: mapNamesToCandidates(parsed.Tags, tags, func(id int64) bool { return assignedTags[id] }),
}
if t := strings.TrimSpace(parsed.Title); t != "" && t != doc.Title {
payload.Title = &t
}
rawPayload, err := json.Marshal(payload)
if err != nil {
return nil, fmt.Errorf("storage: marshal ollama suggestion payload: %w", err)
}
row := s.db.QueryRow(ctx, `
INSERT INTO metadata_suggestions (tenant_id, document_id, provider, requested_by, suggestion)
VALUES ($1, $2, 'ollama', $3, $4)
RETURNING `+metadataSuggestionCols,
tenantID, documentID, requestedBy, rawPayload)
m, err := scanMetadataSuggestion(row)
if err != nil {
return nil, fmt.Errorf("storage: insert ollama metadata suggestion: %w", err)
}
return m, nil
}
// buildOllamaPrompt assembles a strict, schema-forcing prompt. Small models
// need the format spelled out explicitly and benefit from being told to only
// pick from the provided lists.
func buildOllamaPrompt(doc *Document, tags, docTypes, correspondents []TaxonomyEntity) string {
ocr := doc.OCRText
if r := []rune(ocr); len(r) > maxOllamaOCRChars {
ocr = string(r[:maxOllamaOCRChars])
}
var b strings.Builder
b.WriteString("Du bist ein Assistent für ein Dokumentenmanagement-System. ")
b.WriteString("Analysiere das folgende Dokument und schlage passende Metadaten vor. ")
b.WriteString("Antworte AUSSCHLIESSLICH mit einem JSON-Objekt in genau diesem Schema, ohne weiteren Text:\n")
b.WriteString(`{"title": string, "doc_types": [string], "correspondents": [string], "tags": [string]}` + "\n\n")
b.WriteString("Regeln:\n")
b.WriteString("- Wähle doc_types, correspondents und tags NUR aus den unten aufgelisteten vorhandenen Werten (exakte Schreibweise).\n")
b.WriteString("- Wenn nichts passt, gib eine leere Liste zurück.\n")
b.WriteString("- title ist ein kurzer, aussagekräftiger Titel für das Dokument.\n\n")
b.WriteString("Vorhandene document_types: ")
b.WriteString(joinEntityNames(docTypes))
b.WriteString("\nVorhandene correspondents: ")
b.WriteString(joinEntityNames(correspondents))
b.WriteString("\nVorhandene tags: ")
b.WriteString(joinEntityNames(tags))
b.WriteString("\n\n")
b.WriteString("Aktueller Titel: ")
b.WriteString(doc.Title)
b.WriteString("\n\nDokumenttext (Auszug):\n")
b.WriteString(ocr)
return b.String()
}
// joinEntityNames renders entity names as a comma-separated list, or "(keine)"
// when the tenant has no entities of that kind, so the prompt is never empty.
func joinEntityNames(entities []TaxonomyEntity) string {
if len(entities) == 0 {
return "(keine)"
}
names := make([]string, 0, len(entities))
for _, e := range entities {
names = append(names, e.Name)
}
return strings.Join(names, ", ")
}
// mapNamesToCandidates resolves LLM-returned names to existing taxonomy
// entities by exact (case-insensitive) then fuzzy name match, dropping names
// that match nothing above ollamaNameMatchFloor, that are already assigned
// (excluded), or that duplicate an already-mapped entity. The Score reflects
// the name-match confidence. Result is always non-nil, sorted by score desc,
// capped at maxSuggestionCandidates.
func mapNamesToCandidates(names []string, entities []TaxonomyEntity, excluded func(id int64) bool) []SuggestionCandidate {
out := make([]SuggestionCandidate, 0)
seen := map[int64]bool{}
for _, raw := range names {
name := strings.TrimSpace(raw)
if name == "" {
continue
}
best := TaxonomyEntity{}
bestScore := 0.0
found := false
for _, e := range entities {
var score float64
if strings.EqualFold(strings.TrimSpace(e.Name), name) {
score = 1.0
} else {
score = matching.FuzzyScore(e.Name, false, name)
}
if score > bestScore {
bestScore = score
best = e
found = true
}
}
if !found || bestScore < ollamaNameMatchFloor {
continue
}
if excluded(best.ID) || seen[best.ID] {
continue
}
seen[best.ID] = true
out = append(out, SuggestionCandidate{ID: best.ID, Name: best.Name, Score: bestScore})
}
sort.SliceStable(out, func(i, j int) bool { return out[i].Score > out[j].Score })
if len(out) > maxSuggestionCandidates {
out = out[:maxSuggestionCandidates]
}
return out
}