feat(api): OCR réel — POST /ocr/jobs + GET /ocr/jobs/:id (tesseract syscall, jobs asynchrones)
- ocr/tesseract.go : Engine → Tesseract subprocess (OCR_LANG fra+eng) ; PDF → calque texte via ledongthuc/pdf (go.mod : nouveau dep) - repository/ocr_jobs.go : queued→processing→done/failed, scoping device, text/error NULLIFés - service/ocr.go : Create valide le fichier (GetFile), queue + goroutine de traitement ; physique résolu par glob UPLOAD_DIR/<device>/<id>.* ; fail propre (fichier illisible, erreur moteur) - handlers/ocr.go : réels (validate fileId, NOT_FOUND si job d'un autre device), fin 501 OCR - tests end-to-end : cycle queued→done (stub moteur), NOT_FOUND fichier inconnu, scoping device - smoke réel : PNG 'VAULTDROP' → job done text='VAULTDROP' (tesseract installé) - docs/AGENTS : §5 + état des routes (tout V1 réel)
This commit is contained in:
@@ -0,0 +1,67 @@
|
||||
package ocr
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
"github.com/ledongthuc/pdf"
|
||||
)
|
||||
|
||||
// TesseractEngine runs the `tesseract` binary in a subprocess.
|
||||
// Images are OCR'd directly; PDFs have their text layer extracted first
|
||||
// (scanned PDFs → empty text, no rendering pipeline in V1).
|
||||
type TesseractEngine struct{}
|
||||
|
||||
func NewTesseract() *TesseractEngine { return &TesseractEngine{} }
|
||||
|
||||
// ExtractText implements Engine.
|
||||
func (t *TesseractEngine) ExtractText(ctx context.Context, filePath, lang string) (string, error) {
|
||||
if strings.ToLower(filepath.Ext(filePath)) == ".pdf" {
|
||||
return extractPDFText(filePath)
|
||||
}
|
||||
return runTesseract(ctx, filePath, lang)
|
||||
}
|
||||
|
||||
func runTesseract(ctx context.Context, filePath, lang string) (string, error) {
|
||||
cmd := exec.CommandContext(ctx, "tesseract", filePath, "stdout", "-l", lang)
|
||||
var stdout, stderr bytes.Buffer
|
||||
cmd.Stdout = &stdout
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
if errors.Is(ctx.Err(), context.Canceled) {
|
||||
return "", ctx.Err()
|
||||
}
|
||||
msg := strings.TrimSpace(stderr.String())
|
||||
if msg == "" {
|
||||
msg = err.Error()
|
||||
}
|
||||
return "", errors.New("tesseract: " + msg)
|
||||
}
|
||||
return strings.TrimSpace(stdout.String()), nil
|
||||
}
|
||||
|
||||
func extractPDFText(filePath string) (string, error) {
|
||||
f, r, err := pdf.Open(filePath)
|
||||
if err != nil {
|
||||
return "", errors.New("pdf: " + err.Error())
|
||||
}
|
||||
defer f.Close()
|
||||
var builder strings.Builder
|
||||
for i := 1; i <= r.NumPage(); i++ {
|
||||
p := r.Page(i)
|
||||
if p.V.IsNull() {
|
||||
continue
|
||||
}
|
||||
plain, err := p.GetPlainText(nil)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
builder.WriteString(plain)
|
||||
builder.WriteString("\n")
|
||||
}
|
||||
return strings.TrimSpace(builder.String()), nil
|
||||
}
|
||||
Reference in New Issue
Block a user