Files
pdf-ocr-hotfolder/pdf_ocr_hotfolder/config.py
T
techadmin 578472872e fix: Fehlerzaehlung, Upload-Fehler und ocr.timeout scharf (v0.4.0)
- [ocr].timeout wird als tesseract_timeout (pro Seite) an ocrmypdf
  durchgereicht; Default 1800 -> 300, 0 = ocrmypdf-Default
- Exceptions nach dem OCR zaehlen als Fehler, Datei wird nach error/ gerettet
- Fehlgeschlagene Uploads zaehlen als Fehler und loesen Fehler-Mail aus
- name_mode wird im Preflight geprueft, nicht erst pro Datei
- Fehlende [paths]-Sektion -> ConfigError mit klarer Meldung statt KeyError
- Stabilitaets-Timeout zaehlt als Fehler (--once liefert Exit 1)
- upload_folder nutzt shutil.copyfile statt read_bytes/write_bytes
- OcrConfig.pdfa_level Default "2" -> "" (Ghostscript-Bug, Issue #3)
- 35 neue Tests (92 gesamt), pytest.ini
- AI_AGENT_BRIEFING.md auf Stand 0.4.0 gebracht

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-22 21:01:38 +02:00

175 lines
5.0 KiB
Python

"""Konfigurations-Loader (TOML)."""
from __future__ import annotations
import tomllib
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any
class ConfigError(RuntimeError):
"""Konfigurationsdatei ist unvollständig oder fehlerhaft."""
@dataclass
class Paths:
incoming: Path
outgoing: Path
working: Path
error: Path
@dataclass
class OcrConfig:
languages: str = "deu+eng"
jobs: int = 4
skip_text: bool = True
oversample: int = 300
# Default bewusst leer: pdfa_level + skip_text zerschießt OCR mit
# Ghostscript 10.0.0-10.02.0 (Debian-12-Default), siehe Issue #3
pdfa_level: str = ""
deskew: bool = True
clean: bool = False
max_workers: int = 2
# Max. Sekunden, die Tesseract pro Seite laufen darf (0 = kein eigenes Limit)
timeout: int = 300
@dataclass
class OutputConfig:
# "prefix" | "suffix" | "none"
name_mode: str = "prefix"
# Tag-String, verbatim eingefügt (Leerstring = kein Tag)
name_tag: str = "OCR_"
# "delete" | "archive"
original_on_success: str = "delete"
# Absoluter Pfad; Pflicht wenn original_on_success == "archive"
archive_dir: str = ""
@dataclass
class VeraPdfConfig:
enabled: bool = False
binary: str = "/opt/verapdf/verapdf"
flavour: str = "1b"
@dataclass
class FolderUpload:
enabled: bool = True
target: str = ""
@dataclass
class NextcloudUpload:
enabled: bool = False
url: str = ""
username: str = ""
password: str = ""
remote_path: str = ""
verify_ssl: bool = True
@dataclass
class SftpUpload:
enabled: bool = False
host: str = ""
port: int = 22
username: str = ""
key_file: str = ""
password: str = ""
remote_path: str = ""
@dataclass
class EmailNotify:
enabled: bool = False
smtp_host: str = ""
smtp_port: int = 587
smtp_user: str = ""
smtp_password: str = ""
use_starttls: bool = True
from_addr: str = ""
to_addrs: list[str] = field(default_factory=list)
on: str = "errors" # always | errors | never
@dataclass
class Config:
paths: Paths
ocr: OcrConfig
output: OutputConfig
verapdf: VeraPdfConfig
folder: FolderUpload
nextcloud: NextcloudUpload
sftp: SftpUpload
email: EmailNotify
log_level: str = "INFO"
def _section(data: dict[str, Any], *keys: str) -> dict[str, Any]:
cur: Any = data
for k in keys:
cur = cur.get(k, {}) if isinstance(cur, dict) else {}
return cur if isinstance(cur, dict) else {}
def _require_path(p: dict[str, Any], key: str, cfg_path: Path) -> Path:
"""Holt einen Pflicht-Pfad aus der [paths]-Sektion.
Wirft ConfigError mit klarer Meldung statt eines nackten KeyError.
"""
value = p.get(key)
if value is None or (isinstance(value, str) and not value.strip()):
raise ConfigError(
f"{cfg_path}: In der Sektion [paths] fehlt der Eintrag '{key}' "
f"(oder er ist leer). Bitte ergänzen, z.B. "
f'{key} = "/var/lib/pdf-ocr-hotfolder/{key}" '
f"— siehe config.example.toml."
)
return Path(str(value))
def load_config(path: str | Path) -> Config:
path = Path(path)
with path.open("rb") as f:
data = tomllib.load(f)
if not isinstance(data.get("paths"), dict):
raise ConfigError(
f"{path}: Die Sektion [paths] fehlt (oder ist keine Tabelle). "
"Sie muss die Einträge incoming, outgoing, working und error "
"enthalten — siehe config.example.toml."
)
p = _section(data, "paths")
paths = Paths(
incoming=_require_path(p, "incoming", path),
outgoing=_require_path(p, "outgoing", path),
working=_require_path(p, "working", path),
error=_require_path(p, "error", path),
)
ocr = OcrConfig(**{k: v for k, v in _section(data, "ocr").items()
if k in OcrConfig.__annotations__})
output = OutputConfig(**{k: v for k, v in _section(data, "output").items()
if k in OutputConfig.__annotations__})
verapdf = VeraPdfConfig(**{k: v for k, v in _section(data, "verapdf").items()
if k in VeraPdfConfig.__annotations__})
folder = FolderUpload(**{k: v for k, v in _section(data, "upload", "folder").items()
if k in FolderUpload.__annotations__})
nextcloud = NextcloudUpload(**{k: v for k, v in _section(data, "upload", "nextcloud").items()
if k in NextcloudUpload.__annotations__})
sftp = SftpUpload(**{k: v for k, v in _section(data, "upload", "sftp").items()
if k in SftpUpload.__annotations__})
email = EmailNotify(**{k: v for k, v in _section(data, "notify", "email").items()
if k in EmailNotify.__annotations__})
log_level = _section(data, "logging").get("level", "INFO")
return Config(
paths=paths, ocr=ocr, output=output, verapdf=verapdf,
folder=folder, nextcloud=nextcloud, sftp=sftp, email=email,
log_level=log_level,
)