feat(api): OCR réel — POST /ocr/jobs + GET /ocr/jobs/:id (tesseract syscall, jobs asynchrones)

- ocr/tesseract.go : Engine → Tesseract subprocess (OCR_LANG fra+eng) ; PDF → calque texte via ledongthuc/pdf (go.mod : nouveau dep)
- repository/ocr_jobs.go : queued→processing→done/failed, scoping device, text/error NULLIFés
- service/ocr.go : Create valide le fichier (GetFile), queue + goroutine de traitement ; physique résolu par glob UPLOAD_DIR/<device>/<id>.* ; fail propre (fichier illisible, erreur moteur)
- handlers/ocr.go : réels (validate fileId, NOT_FOUND si job d'un autre device), fin 501 OCR
- tests end-to-end : cycle queued→done (stub moteur), NOT_FOUND fichier inconnu, scoping device
- smoke réel : PNG 'VAULTDROP' → job done text='VAULTDROP' (tesseract installé)
- docs/AGENTS : §5 + état des routes (tout V1 réel)
This commit is contained in:
m
2026-09-10 19:54:16 +02:00
parent 5d96814853
commit d5886c678b
12 changed files with 452 additions and 10 deletions
+67
View File
@@ -0,0 +1,67 @@
package ocr
import (
"bytes"
"context"
"errors"
"os/exec"
"path/filepath"
"strings"
"github.com/ledongthuc/pdf"
)
// TesseractEngine runs the `tesseract` binary in a subprocess.
// Images are OCR'd directly; PDFs have their text layer extracted first
// (scanned PDFs → empty text, no rendering pipeline in V1).
type TesseractEngine struct{}
func NewTesseract() *TesseractEngine { return &TesseractEngine{} }
// ExtractText implements Engine.
func (t *TesseractEngine) ExtractText(ctx context.Context, filePath, lang string) (string, error) {
if strings.ToLower(filepath.Ext(filePath)) == ".pdf" {
return extractPDFText(filePath)
}
return runTesseract(ctx, filePath, lang)
}
func runTesseract(ctx context.Context, filePath, lang string) (string, error) {
cmd := exec.CommandContext(ctx, "tesseract", filePath, "stdout", "-l", lang)
var stdout, stderr bytes.Buffer
cmd.Stdout = &stdout
cmd.Stderr = &stderr
if err := cmd.Run(); err != nil {
if errors.Is(ctx.Err(), context.Canceled) {
return "", ctx.Err()
}
msg := strings.TrimSpace(stderr.String())
if msg == "" {
msg = err.Error()
}
return "", errors.New("tesseract: " + msg)
}
return strings.TrimSpace(stdout.String()), nil
}
func extractPDFText(filePath string) (string, error) {
f, r, err := pdf.Open(filePath)
if err != nil {
return "", errors.New("pdf: " + err.Error())
}
defer f.Close()
var builder strings.Builder
for i := 1; i <= r.NumPage(); i++ {
p := r.Page(i)
if p.V.IsNull() {
continue
}
plain, err := p.GetPlainText(nil)
if err != nil {
continue
}
builder.WriteString(plain)
builder.WriteString("\n")
}
return strings.TrimSpace(builder.String()), nil
}