fix: Fehlerzaehlung, Upload-Fehler und ocr.timeout scharf (v0.4.0)

- [ocr].timeout wird als tesseract_timeout (pro Seite) an ocrmypdf
  durchgereicht; Default 1800 -> 300, 0 = ocrmypdf-Default
- Exceptions nach dem OCR zaehlen als Fehler, Datei wird nach error/ gerettet
- Fehlgeschlagene Uploads zaehlen als Fehler und loesen Fehler-Mail aus
- name_mode wird im Preflight geprueft, nicht erst pro Datei
- Fehlende [paths]-Sektion -> ConfigError mit klarer Meldung statt KeyError
- Stabilitaets-Timeout zaehlt als Fehler (--once liefert Exit 1)
- upload_folder nutzt shutil.copyfile statt read_bytes/write_bytes
- OcrConfig.pdfa_level Default "2" -> "" (Ghostscript-Bug, Issue #3)
- 35 neue Tests (92 gesamt), pytest.ini
- AI_AGENT_BRIEFING.md auf Stand 0.4.0 gebracht

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-09-22 21:01:38 +02:00
parent cbdc9d6664
commit 578472872e
17 changed files with 890 additions and 76 deletions
+1 -1
View File
@@ -1,3 +1,3 @@
"""PDF OCR Hotfolder — Scanner-PDFs automatisch durchsuchbar machen."""
__version__ = "0.3.1"
__version__ = "0.4.0"
+6 -2
View File
@@ -7,7 +7,7 @@ import sys
from pathlib import Path
from . import __version__
from .config import load_config
from .config import ConfigError, load_config
from .service import HotfolderService, PreflightError
@@ -36,7 +36,11 @@ def main() -> int:
print(f"Config nicht gefunden: {cfg_path}", file=sys.stderr)
return 2
cfg = load_config(cfg_path)
try:
cfg = load_config(cfg_path)
except ConfigError as e:
print(f"FEHLER: {e}", file=sys.stderr)
return 2
_setup_logging(cfg.log_level)
service = HotfolderService(cfg)
+36 -6
View File
@@ -7,6 +7,10 @@ from pathlib import Path
from typing import Any
class ConfigError(RuntimeError):
"""Konfigurationsdatei ist unvollständig oder fehlerhaft."""
@dataclass
class Paths:
incoming: Path
@@ -21,11 +25,14 @@ class OcrConfig:
jobs: int = 4
skip_text: bool = True
oversample: int = 300
pdfa_level: str = "2"
# Default bewusst leer: pdfa_level + skip_text zerschießt OCR mit
# Ghostscript 10.0.0-10.02.0 (Debian-12-Default), siehe Issue #3
pdfa_level: str = ""
deskew: bool = True
clean: bool = False
max_workers: int = 2
timeout: int = 1800
# Max. Sekunden, die Tesseract pro Seite laufen darf (0 = kein eigenes Limit)
timeout: int = 300
@dataclass
@@ -107,17 +114,40 @@ def _section(data: dict[str, Any], *keys: str) -> dict[str, Any]:
return cur if isinstance(cur, dict) else {}
def _require_path(p: dict[str, Any], key: str, cfg_path: Path) -> Path:
"""Holt einen Pflicht-Pfad aus der [paths]-Sektion.
Wirft ConfigError mit klarer Meldung statt eines nackten KeyError.
"""
value = p.get(key)
if value is None or (isinstance(value, str) and not value.strip()):
raise ConfigError(
f"{cfg_path}: In der Sektion [paths] fehlt der Eintrag '{key}' "
f"(oder er ist leer). Bitte ergänzen, z.B. "
f'{key} = "/var/lib/pdf-ocr-hotfolder/{key}" '
f"— siehe config.example.toml."
)
return Path(str(value))
def load_config(path: str | Path) -> Config:
path = Path(path)
with path.open("rb") as f:
data = tomllib.load(f)
if not isinstance(data.get("paths"), dict):
raise ConfigError(
f"{path}: Die Sektion [paths] fehlt (oder ist keine Tabelle). "
"Sie muss die Einträge incoming, outgoing, working und error "
"enthalten — siehe config.example.toml."
)
p = _section(data, "paths")
paths = Paths(
incoming=Path(p["incoming"]),
outgoing=Path(p["outgoing"]),
working=Path(p["working"]),
error=Path(p["error"]),
incoming=_require_path(p, "incoming", path),
outgoing=_require_path(p, "outgoing", path),
working=_require_path(p, "working", path),
error=_require_path(p, "error", path),
)
ocr = OcrConfig(**{k: v for k, v in _section(data, "ocr").items()
+12
View File
@@ -11,6 +11,9 @@ from .config import OcrConfig, OutputConfig, VeraPdfConfig
log = logging.getLogger(__name__)
# Erlaubte Werte für [output].name_mode — wird auch vom Preflight geprüft
VALID_NAME_MODES = ("prefix", "suffix", "none")
def build_output_name(src_name: str, mode: str, tag: str) -> str:
"""Erzeugt den Ziel-Dateinamen für ein OCR-PDF.
@@ -65,6 +68,15 @@ def run_ocr(src: Path, dst: Path, cfg: OcrConfig) -> None:
else:
kwargs["output_type"] = "pdf"
# [ocr].timeout = max. Sekunden, die Tesseract pro Seite laufen darf.
# ocrmypdf kennt kein Gesamt-Timeout für ein Dokument, nur `tesseract_timeout`
# (pro Seite). ACHTUNG: ocrmypdf interpretiert tesseract_timeout=0 als
# "OCR komplett überspringen" — deshalb wird 0 bei uns als "kein eigenes
# Limit" behandelt und gar nicht erst durchgereicht (dann gilt der
# ocrmypdf-Default).
if cfg.timeout and cfg.timeout > 0:
kwargs["tesseract_timeout"] = float(cfg.timeout)
log.info("OCR start: %s", src.name)
ocrmypdf.ocr(str(src), str(dst), **kwargs)
log.info("OCR done: %s", dst.name)
+119 -27
View File
@@ -15,7 +15,7 @@ from watchdog.events import FileSystemEvent, FileSystemEventHandler
from watchdog.observers import Observer
from .config import Config
from .processor import ProcessResult, process_pdf
from .processor import VALID_NAME_MODES, ProcessResult, _move_to_error, process_pdf
from .uploaders import notify_email, upload_folder, upload_nextcloud, upload_sftp
log = logging.getLogger(__name__)
@@ -72,7 +72,8 @@ def detect_ghostscript_version() -> str | None:
return result.stdout.strip() or None
def check_output_config(mode: str, archive_dir: str) -> None:
def check_output_config(mode: str, archive_dir: str,
name_mode: str = "prefix") -> None:
"""Validiert die [output]-Section. Wirft PreflightError bei Problemen."""
valid_modes = {"delete", "archive"}
if mode not in valid_modes:
@@ -84,6 +85,13 @@ def check_output_config(mode: str, archive_dir: str) -> None:
raise PreflightError(
"[output].original_on_success='archive' erfordert [output].archive_dir"
)
# Früh prüfen: sonst schlägt ein Tippfehler erst pro Datei zu — und zwar
# NACH dem Move nach working/, wo die Datei dann liegen bleibt.
if name_mode not in VALID_NAME_MODES:
raise PreflightError(
f"[output].name_mode={name_mode!r} ungültig. "
f"Erlaubt: {sorted(VALID_NAME_MODES)}"
)
def check_preflight(pdfa_level: str = "") -> None:
@@ -188,7 +196,8 @@ class HotfolderService:
def run(self) -> None:
check_preflight(self.cfg.ocr.pdfa_level)
check_output_config(self.cfg.output.original_on_success,
self.cfg.output.archive_dir)
self.cfg.output.archive_dir,
self.cfg.output.name_mode)
self.ensure_dirs()
self._scan_existing()
@@ -214,7 +223,8 @@ class HotfolderService:
"""
check_preflight(self.cfg.ocr.pdfa_level)
check_output_config(self.cfg.output.original_on_success,
self.cfg.output.archive_dir)
self.cfg.output.archive_dir,
self.cfg.output.name_mode)
self.ensure_dirs()
self._scan_existing()
self._executor.shutdown(wait=True)
@@ -258,39 +268,121 @@ class HotfolderService:
# ---- Processing ----
def _count_success(self) -> None:
with self._lock:
self._success_count += 1
def _count_error(self) -> None:
with self._lock:
self._error_count += 1
def _process(self, path: Path) -> None:
if not _wait_until_stable(path):
log.warning("Datei nicht stabilisiert, überspringe: %s", path)
if not path.exists():
# Datei wurde währenddessen entfernt — kein Fehlerfall
log.info("Datei vor der Verarbeitung verschwunden: %s", path)
return
# Bewusst als Fehler zählen: sonst liefert --once trotz liegen
# gebliebener Datei Exit 0.
log.error(
"Datei hat sich nicht stabilisiert (Timeout): %s — bleibt in %s "
"liegen und wird beim nächsten Lauf erneut versucht",
path, self.cfg.paths.incoming,
)
self._count_error()
return
if not path.exists():
return
result: ProcessResult = process_pdf(
src=path,
working_dir=self.cfg.paths.working,
outgoing_dir=self.cfg.paths.outgoing,
error_dir=self.cfg.paths.error,
ocr_cfg=self.cfg.ocr,
vera_cfg=self.cfg.verapdf,
output_cfg=self.cfg.output,
)
try:
result: ProcessResult = process_pdf(
src=path,
working_dir=self.cfg.paths.working,
outgoing_dir=self.cfg.paths.outgoing,
error_dir=self.cfg.paths.error,
ocr_cfg=self.cfg.ocr,
vera_cfg=self.cfg.verapdf,
output_cfg=self.cfg.output,
)
except Exception as e: # noqa: BLE001 - kein Fehler darf die Zählung umgehen
log.exception("Unerwarteter Fehler bei der Verarbeitung von %s", path.name)
self._count_error()
self._rescue_to_error(path)
self._notify(ProcessResult(
path, self.cfg.paths.outgoing / path.name, False,
f"unerwarteter Fehler: {e}",
))
return
with self._lock:
if result.success:
self._success_count += 1
else:
self._error_count += 1
if not result.success:
self._count_error()
self._notify(result)
return
if result.success:
self._dispatch_uploads(result.output)
failed = self._dispatch_uploads(result.output)
if failed:
log.error(
"Upload fehlgeschlagen (%s) für %s — das OCR selbst war "
"erfolgreich, die Datei bleibt daher in %s liegen und wird "
"NICHT nach error/ verschoben",
", ".join(failed), result.output.name, result.output.parent,
)
self._count_error()
self._notify_upload_failure(result, failed)
return
self._count_success()
self._notify(result)
def _dispatch_uploads(self, pdf: Path) -> None:
upload_folder(pdf, self.cfg.folder, self.cfg.paths.outgoing)
if self.cfg.nextcloud.enabled:
upload_nextcloud(pdf, self.cfg.nextcloud)
if self.cfg.sftp.enabled:
upload_sftp(pdf, self.cfg.sftp)
def _rescue_to_error(self, src: Path) -> None:
"""Bringt eine Datei nach einer unerwarteten Exception ins error-Verzeichnis.
Die Datei kann je nach Abbruchzeitpunkt noch in incoming/ oder schon in
working/ liegen. Der erste Treffer wird verschoben (keine Doppel-Moves),
Fehler beim Verschieben werden nur geloggt.
"""
error_dir = self.cfg.paths.error
for candidate in (src, self.cfg.paths.working / src.name):
try:
if not candidate.is_file():
continue
if candidate.parent.resolve() == error_dir.resolve():
return # liegt bereits im error-Verzeichnis
except OSError:
continue
_move_to_error(candidate, error_dir)
return
log.warning("Datei %s nach Fehler nicht mehr auffindbar — "
"kein Verschieben nach error/ möglich", src.name)
def _dispatch_uploads(self, pdf: Path) -> list[str]:
"""Schiebt das fertige PDF an alle Upload-Ziele.
Die uploader prüfen `cfg.enabled` jeweils selbst und liefern für
deaktivierte Ziele True.
Returns:
Namen der fehlgeschlagenen Ziele — leere Liste = alle erfolgreich.
"""
failed: list[str] = []
if not upload_folder(pdf, self.cfg.folder, self.cfg.paths.outgoing):
failed.append("folder")
if not upload_nextcloud(pdf, self.cfg.nextcloud):
failed.append("nextcloud")
if not upload_sftp(pdf, self.cfg.sftp):
failed.append("sftp")
return failed
def _notify_upload_failure(self, result: ProcessResult, failed: list[str]) -> None:
"""Fehler-Mail, wenn das OCR lief, aber mindestens ein Upload scheiterte."""
subject = f"[pdf-ocr] FEHLER Upload: {result.source.name}"
body = (
f"OCR erfolgreich: {result.output}\n\n"
f"Fehlgeschlagene Upload-Ziele: {', '.join(failed)}\n\n"
f"Das OCR-PDF bleibt in {result.output.parent} liegen und wurde "
"NICHT nach error/ verschoben. Details siehe Log.\n"
)
notify_email(self.cfg.email, subject, body, False)
def _notify(self, result: ProcessResult) -> None:
if result.success:
+4 -1
View File
@@ -2,6 +2,7 @@
from __future__ import annotations
import logging
import shutil
import smtplib
import ssl
from email.message import EmailMessage
@@ -25,7 +26,9 @@ def upload_folder(pdf: Path, cfg: FolderUpload, default_target: Path) -> bool:
try:
if pdf.resolve() == dest.resolve():
return True
dest.write_bytes(pdf.read_bytes())
# copyfile statt read_bytes/write_bytes: große PDFs nicht komplett
# in den Speicher laden
shutil.copyfile(pdf, dest)
log.info("Folder upload OK: %s", dest)
return True
except OSError as e: