fix: Fehlerzaehlung, Upload-Fehler und ocr.timeout scharf (v0.4.0)
- [ocr].timeout wird als tesseract_timeout (pro Seite) an ocrmypdf durchgereicht; Default 1800 -> 300, 0 = ocrmypdf-Default - Exceptions nach dem OCR zaehlen als Fehler, Datei wird nach error/ gerettet - Fehlgeschlagene Uploads zaehlen als Fehler und loesen Fehler-Mail aus - name_mode wird im Preflight geprueft, nicht erst pro Datei - Fehlende [paths]-Sektion -> ConfigError mit klarer Meldung statt KeyError - Stabilitaets-Timeout zaehlt als Fehler (--once liefert Exit 1) - upload_folder nutzt shutil.copyfile statt read_bytes/write_bytes - OcrConfig.pdfa_level Default "2" -> "" (Ghostscript-Bug, Issue #3) - 35 neue Tests (92 gesamt), pytest.ini - AI_AGENT_BRIEFING.md auf Stand 0.4.0 gebracht Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -1,3 +1,3 @@
|
||||
"""PDF OCR Hotfolder — Scanner-PDFs automatisch durchsuchbar machen."""
|
||||
|
||||
__version__ = "0.3.1"
|
||||
__version__ = "0.4.0"
|
||||
|
||||
@@ -7,7 +7,7 @@ import sys
|
||||
from pathlib import Path
|
||||
|
||||
from . import __version__
|
||||
from .config import load_config
|
||||
from .config import ConfigError, load_config
|
||||
from .service import HotfolderService, PreflightError
|
||||
|
||||
|
||||
@@ -36,7 +36,11 @@ def main() -> int:
|
||||
print(f"Config nicht gefunden: {cfg_path}", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
cfg = load_config(cfg_path)
|
||||
try:
|
||||
cfg = load_config(cfg_path)
|
||||
except ConfigError as e:
|
||||
print(f"FEHLER: {e}", file=sys.stderr)
|
||||
return 2
|
||||
_setup_logging(cfg.log_level)
|
||||
|
||||
service = HotfolderService(cfg)
|
||||
|
||||
@@ -7,6 +7,10 @@ from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
class ConfigError(RuntimeError):
|
||||
"""Konfigurationsdatei ist unvollständig oder fehlerhaft."""
|
||||
|
||||
|
||||
@dataclass
|
||||
class Paths:
|
||||
incoming: Path
|
||||
@@ -21,11 +25,14 @@ class OcrConfig:
|
||||
jobs: int = 4
|
||||
skip_text: bool = True
|
||||
oversample: int = 300
|
||||
pdfa_level: str = "2"
|
||||
# Default bewusst leer: pdfa_level + skip_text zerschießt OCR mit
|
||||
# Ghostscript 10.0.0-10.02.0 (Debian-12-Default), siehe Issue #3
|
||||
pdfa_level: str = ""
|
||||
deskew: bool = True
|
||||
clean: bool = False
|
||||
max_workers: int = 2
|
||||
timeout: int = 1800
|
||||
# Max. Sekunden, die Tesseract pro Seite laufen darf (0 = kein eigenes Limit)
|
||||
timeout: int = 300
|
||||
|
||||
|
||||
@dataclass
|
||||
@@ -107,17 +114,40 @@ def _section(data: dict[str, Any], *keys: str) -> dict[str, Any]:
|
||||
return cur if isinstance(cur, dict) else {}
|
||||
|
||||
|
||||
def _require_path(p: dict[str, Any], key: str, cfg_path: Path) -> Path:
|
||||
"""Holt einen Pflicht-Pfad aus der [paths]-Sektion.
|
||||
|
||||
Wirft ConfigError mit klarer Meldung statt eines nackten KeyError.
|
||||
"""
|
||||
value = p.get(key)
|
||||
if value is None or (isinstance(value, str) and not value.strip()):
|
||||
raise ConfigError(
|
||||
f"{cfg_path}: In der Sektion [paths] fehlt der Eintrag '{key}' "
|
||||
f"(oder er ist leer). Bitte ergänzen, z.B. "
|
||||
f'{key} = "/var/lib/pdf-ocr-hotfolder/{key}" '
|
||||
f"— siehe config.example.toml."
|
||||
)
|
||||
return Path(str(value))
|
||||
|
||||
|
||||
def load_config(path: str | Path) -> Config:
|
||||
path = Path(path)
|
||||
with path.open("rb") as f:
|
||||
data = tomllib.load(f)
|
||||
|
||||
if not isinstance(data.get("paths"), dict):
|
||||
raise ConfigError(
|
||||
f"{path}: Die Sektion [paths] fehlt (oder ist keine Tabelle). "
|
||||
"Sie muss die Einträge incoming, outgoing, working und error "
|
||||
"enthalten — siehe config.example.toml."
|
||||
)
|
||||
|
||||
p = _section(data, "paths")
|
||||
paths = Paths(
|
||||
incoming=Path(p["incoming"]),
|
||||
outgoing=Path(p["outgoing"]),
|
||||
working=Path(p["working"]),
|
||||
error=Path(p["error"]),
|
||||
incoming=_require_path(p, "incoming", path),
|
||||
outgoing=_require_path(p, "outgoing", path),
|
||||
working=_require_path(p, "working", path),
|
||||
error=_require_path(p, "error", path),
|
||||
)
|
||||
|
||||
ocr = OcrConfig(**{k: v for k, v in _section(data, "ocr").items()
|
||||
|
||||
@@ -11,6 +11,9 @@ from .config import OcrConfig, OutputConfig, VeraPdfConfig
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
# Erlaubte Werte für [output].name_mode — wird auch vom Preflight geprüft
|
||||
VALID_NAME_MODES = ("prefix", "suffix", "none")
|
||||
|
||||
|
||||
def build_output_name(src_name: str, mode: str, tag: str) -> str:
|
||||
"""Erzeugt den Ziel-Dateinamen für ein OCR-PDF.
|
||||
@@ -65,6 +68,15 @@ def run_ocr(src: Path, dst: Path, cfg: OcrConfig) -> None:
|
||||
else:
|
||||
kwargs["output_type"] = "pdf"
|
||||
|
||||
# [ocr].timeout = max. Sekunden, die Tesseract pro Seite laufen darf.
|
||||
# ocrmypdf kennt kein Gesamt-Timeout für ein Dokument, nur `tesseract_timeout`
|
||||
# (pro Seite). ACHTUNG: ocrmypdf interpretiert tesseract_timeout=0 als
|
||||
# "OCR komplett überspringen" — deshalb wird 0 bei uns als "kein eigenes
|
||||
# Limit" behandelt und gar nicht erst durchgereicht (dann gilt der
|
||||
# ocrmypdf-Default).
|
||||
if cfg.timeout and cfg.timeout > 0:
|
||||
kwargs["tesseract_timeout"] = float(cfg.timeout)
|
||||
|
||||
log.info("OCR start: %s", src.name)
|
||||
ocrmypdf.ocr(str(src), str(dst), **kwargs)
|
||||
log.info("OCR done: %s", dst.name)
|
||||
|
||||
+119
-27
@@ -15,7 +15,7 @@ from watchdog.events import FileSystemEvent, FileSystemEventHandler
|
||||
from watchdog.observers import Observer
|
||||
|
||||
from .config import Config
|
||||
from .processor import ProcessResult, process_pdf
|
||||
from .processor import VALID_NAME_MODES, ProcessResult, _move_to_error, process_pdf
|
||||
from .uploaders import notify_email, upload_folder, upload_nextcloud, upload_sftp
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
@@ -72,7 +72,8 @@ def detect_ghostscript_version() -> str | None:
|
||||
return result.stdout.strip() or None
|
||||
|
||||
|
||||
def check_output_config(mode: str, archive_dir: str) -> None:
|
||||
def check_output_config(mode: str, archive_dir: str,
|
||||
name_mode: str = "prefix") -> None:
|
||||
"""Validiert die [output]-Section. Wirft PreflightError bei Problemen."""
|
||||
valid_modes = {"delete", "archive"}
|
||||
if mode not in valid_modes:
|
||||
@@ -84,6 +85,13 @@ def check_output_config(mode: str, archive_dir: str) -> None:
|
||||
raise PreflightError(
|
||||
"[output].original_on_success='archive' erfordert [output].archive_dir"
|
||||
)
|
||||
# Früh prüfen: sonst schlägt ein Tippfehler erst pro Datei zu — und zwar
|
||||
# NACH dem Move nach working/, wo die Datei dann liegen bleibt.
|
||||
if name_mode not in VALID_NAME_MODES:
|
||||
raise PreflightError(
|
||||
f"[output].name_mode={name_mode!r} ungültig. "
|
||||
f"Erlaubt: {sorted(VALID_NAME_MODES)}"
|
||||
)
|
||||
|
||||
|
||||
def check_preflight(pdfa_level: str = "") -> None:
|
||||
@@ -188,7 +196,8 @@ class HotfolderService:
|
||||
def run(self) -> None:
|
||||
check_preflight(self.cfg.ocr.pdfa_level)
|
||||
check_output_config(self.cfg.output.original_on_success,
|
||||
self.cfg.output.archive_dir)
|
||||
self.cfg.output.archive_dir,
|
||||
self.cfg.output.name_mode)
|
||||
self.ensure_dirs()
|
||||
self._scan_existing()
|
||||
|
||||
@@ -214,7 +223,8 @@ class HotfolderService:
|
||||
"""
|
||||
check_preflight(self.cfg.ocr.pdfa_level)
|
||||
check_output_config(self.cfg.output.original_on_success,
|
||||
self.cfg.output.archive_dir)
|
||||
self.cfg.output.archive_dir,
|
||||
self.cfg.output.name_mode)
|
||||
self.ensure_dirs()
|
||||
self._scan_existing()
|
||||
self._executor.shutdown(wait=True)
|
||||
@@ -258,39 +268,121 @@ class HotfolderService:
|
||||
|
||||
# ---- Processing ----
|
||||
|
||||
def _count_success(self) -> None:
|
||||
with self._lock:
|
||||
self._success_count += 1
|
||||
|
||||
def _count_error(self) -> None:
|
||||
with self._lock:
|
||||
self._error_count += 1
|
||||
|
||||
def _process(self, path: Path) -> None:
|
||||
if not _wait_until_stable(path):
|
||||
log.warning("Datei nicht stabilisiert, überspringe: %s", path)
|
||||
if not path.exists():
|
||||
# Datei wurde währenddessen entfernt — kein Fehlerfall
|
||||
log.info("Datei vor der Verarbeitung verschwunden: %s", path)
|
||||
return
|
||||
# Bewusst als Fehler zählen: sonst liefert --once trotz liegen
|
||||
# gebliebener Datei Exit 0.
|
||||
log.error(
|
||||
"Datei hat sich nicht stabilisiert (Timeout): %s — bleibt in %s "
|
||||
"liegen und wird beim nächsten Lauf erneut versucht",
|
||||
path, self.cfg.paths.incoming,
|
||||
)
|
||||
self._count_error()
|
||||
return
|
||||
if not path.exists():
|
||||
return
|
||||
|
||||
result: ProcessResult = process_pdf(
|
||||
src=path,
|
||||
working_dir=self.cfg.paths.working,
|
||||
outgoing_dir=self.cfg.paths.outgoing,
|
||||
error_dir=self.cfg.paths.error,
|
||||
ocr_cfg=self.cfg.ocr,
|
||||
vera_cfg=self.cfg.verapdf,
|
||||
output_cfg=self.cfg.output,
|
||||
)
|
||||
try:
|
||||
result: ProcessResult = process_pdf(
|
||||
src=path,
|
||||
working_dir=self.cfg.paths.working,
|
||||
outgoing_dir=self.cfg.paths.outgoing,
|
||||
error_dir=self.cfg.paths.error,
|
||||
ocr_cfg=self.cfg.ocr,
|
||||
vera_cfg=self.cfg.verapdf,
|
||||
output_cfg=self.cfg.output,
|
||||
)
|
||||
except Exception as e: # noqa: BLE001 - kein Fehler darf die Zählung umgehen
|
||||
log.exception("Unerwarteter Fehler bei der Verarbeitung von %s", path.name)
|
||||
self._count_error()
|
||||
self._rescue_to_error(path)
|
||||
self._notify(ProcessResult(
|
||||
path, self.cfg.paths.outgoing / path.name, False,
|
||||
f"unerwarteter Fehler: {e}",
|
||||
))
|
||||
return
|
||||
|
||||
with self._lock:
|
||||
if result.success:
|
||||
self._success_count += 1
|
||||
else:
|
||||
self._error_count += 1
|
||||
if not result.success:
|
||||
self._count_error()
|
||||
self._notify(result)
|
||||
return
|
||||
|
||||
if result.success:
|
||||
self._dispatch_uploads(result.output)
|
||||
failed = self._dispatch_uploads(result.output)
|
||||
if failed:
|
||||
log.error(
|
||||
"Upload fehlgeschlagen (%s) für %s — das OCR selbst war "
|
||||
"erfolgreich, die Datei bleibt daher in %s liegen und wird "
|
||||
"NICHT nach error/ verschoben",
|
||||
", ".join(failed), result.output.name, result.output.parent,
|
||||
)
|
||||
self._count_error()
|
||||
self._notify_upload_failure(result, failed)
|
||||
return
|
||||
|
||||
self._count_success()
|
||||
self._notify(result)
|
||||
|
||||
def _dispatch_uploads(self, pdf: Path) -> None:
|
||||
upload_folder(pdf, self.cfg.folder, self.cfg.paths.outgoing)
|
||||
if self.cfg.nextcloud.enabled:
|
||||
upload_nextcloud(pdf, self.cfg.nextcloud)
|
||||
if self.cfg.sftp.enabled:
|
||||
upload_sftp(pdf, self.cfg.sftp)
|
||||
def _rescue_to_error(self, src: Path) -> None:
|
||||
"""Bringt eine Datei nach einer unerwarteten Exception ins error-Verzeichnis.
|
||||
|
||||
Die Datei kann je nach Abbruchzeitpunkt noch in incoming/ oder schon in
|
||||
working/ liegen. Der erste Treffer wird verschoben (keine Doppel-Moves),
|
||||
Fehler beim Verschieben werden nur geloggt.
|
||||
"""
|
||||
error_dir = self.cfg.paths.error
|
||||
for candidate in (src, self.cfg.paths.working / src.name):
|
||||
try:
|
||||
if not candidate.is_file():
|
||||
continue
|
||||
if candidate.parent.resolve() == error_dir.resolve():
|
||||
return # liegt bereits im error-Verzeichnis
|
||||
except OSError:
|
||||
continue
|
||||
_move_to_error(candidate, error_dir)
|
||||
return
|
||||
log.warning("Datei %s nach Fehler nicht mehr auffindbar — "
|
||||
"kein Verschieben nach error/ möglich", src.name)
|
||||
|
||||
def _dispatch_uploads(self, pdf: Path) -> list[str]:
|
||||
"""Schiebt das fertige PDF an alle Upload-Ziele.
|
||||
|
||||
Die uploader prüfen `cfg.enabled` jeweils selbst und liefern für
|
||||
deaktivierte Ziele True.
|
||||
|
||||
Returns:
|
||||
Namen der fehlgeschlagenen Ziele — leere Liste = alle erfolgreich.
|
||||
"""
|
||||
failed: list[str] = []
|
||||
if not upload_folder(pdf, self.cfg.folder, self.cfg.paths.outgoing):
|
||||
failed.append("folder")
|
||||
if not upload_nextcloud(pdf, self.cfg.nextcloud):
|
||||
failed.append("nextcloud")
|
||||
if not upload_sftp(pdf, self.cfg.sftp):
|
||||
failed.append("sftp")
|
||||
return failed
|
||||
|
||||
def _notify_upload_failure(self, result: ProcessResult, failed: list[str]) -> None:
|
||||
"""Fehler-Mail, wenn das OCR lief, aber mindestens ein Upload scheiterte."""
|
||||
subject = f"[pdf-ocr] FEHLER Upload: {result.source.name}"
|
||||
body = (
|
||||
f"OCR erfolgreich: {result.output}\n\n"
|
||||
f"Fehlgeschlagene Upload-Ziele: {', '.join(failed)}\n\n"
|
||||
f"Das OCR-PDF bleibt in {result.output.parent} liegen und wurde "
|
||||
"NICHT nach error/ verschoben. Details siehe Log.\n"
|
||||
)
|
||||
notify_email(self.cfg.email, subject, body, False)
|
||||
|
||||
def _notify(self, result: ProcessResult) -> None:
|
||||
if result.success:
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import shutil
|
||||
import smtplib
|
||||
import ssl
|
||||
from email.message import EmailMessage
|
||||
@@ -25,7 +26,9 @@ def upload_folder(pdf: Path, cfg: FolderUpload, default_target: Path) -> bool:
|
||||
try:
|
||||
if pdf.resolve() == dest.resolve():
|
||||
return True
|
||||
dest.write_bytes(pdf.read_bytes())
|
||||
# copyfile statt read_bytes/write_bytes: große PDFs nicht komplett
|
||||
# in den Speicher laden
|
||||
shutil.copyfile(pdf, dest)
|
||||
log.info("Folder upload OK: %s", dest)
|
||||
return True
|
||||
except OSError as e:
|
||||
|
||||
Reference in New Issue
Block a user