578472872e
- [ocr].timeout wird als tesseract_timeout (pro Seite) an ocrmypdf durchgereicht; Default 1800 -> 300, 0 = ocrmypdf-Default - Exceptions nach dem OCR zaehlen als Fehler, Datei wird nach error/ gerettet - Fehlgeschlagene Uploads zaehlen als Fehler und loesen Fehler-Mail aus - name_mode wird im Preflight geprueft, nicht erst pro Datei - Fehlende [paths]-Sektion -> ConfigError mit klarer Meldung statt KeyError - Stabilitaets-Timeout zaehlt als Fehler (--once liefert Exit 1) - upload_folder nutzt shutil.copyfile statt read_bytes/write_bytes - OcrConfig.pdfa_level Default "2" -> "" (Ghostscript-Bug, Issue #3) - 35 neue Tests (92 gesamt), pytest.ini - AI_AGENT_BRIEFING.md auf Stand 0.4.0 gebracht Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
175 lines
5.0 KiB
Python
175 lines
5.0 KiB
Python
"""Konfigurations-Loader (TOML)."""
|
|
from __future__ import annotations
|
|
|
|
import tomllib
|
|
from dataclasses import dataclass, field
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
|
|
class ConfigError(RuntimeError):
|
|
"""Konfigurationsdatei ist unvollständig oder fehlerhaft."""
|
|
|
|
|
|
@dataclass
|
|
class Paths:
|
|
incoming: Path
|
|
outgoing: Path
|
|
working: Path
|
|
error: Path
|
|
|
|
|
|
@dataclass
|
|
class OcrConfig:
|
|
languages: str = "deu+eng"
|
|
jobs: int = 4
|
|
skip_text: bool = True
|
|
oversample: int = 300
|
|
# Default bewusst leer: pdfa_level + skip_text zerschießt OCR mit
|
|
# Ghostscript 10.0.0-10.02.0 (Debian-12-Default), siehe Issue #3
|
|
pdfa_level: str = ""
|
|
deskew: bool = True
|
|
clean: bool = False
|
|
max_workers: int = 2
|
|
# Max. Sekunden, die Tesseract pro Seite laufen darf (0 = kein eigenes Limit)
|
|
timeout: int = 300
|
|
|
|
|
|
@dataclass
|
|
class OutputConfig:
|
|
# "prefix" | "suffix" | "none"
|
|
name_mode: str = "prefix"
|
|
# Tag-String, verbatim eingefügt (Leerstring = kein Tag)
|
|
name_tag: str = "OCR_"
|
|
# "delete" | "archive"
|
|
original_on_success: str = "delete"
|
|
# Absoluter Pfad; Pflicht wenn original_on_success == "archive"
|
|
archive_dir: str = ""
|
|
|
|
|
|
@dataclass
|
|
class VeraPdfConfig:
|
|
enabled: bool = False
|
|
binary: str = "/opt/verapdf/verapdf"
|
|
flavour: str = "1b"
|
|
|
|
|
|
@dataclass
|
|
class FolderUpload:
|
|
enabled: bool = True
|
|
target: str = ""
|
|
|
|
|
|
@dataclass
|
|
class NextcloudUpload:
|
|
enabled: bool = False
|
|
url: str = ""
|
|
username: str = ""
|
|
password: str = ""
|
|
remote_path: str = ""
|
|
verify_ssl: bool = True
|
|
|
|
|
|
@dataclass
|
|
class SftpUpload:
|
|
enabled: bool = False
|
|
host: str = ""
|
|
port: int = 22
|
|
username: str = ""
|
|
key_file: str = ""
|
|
password: str = ""
|
|
remote_path: str = ""
|
|
|
|
|
|
@dataclass
|
|
class EmailNotify:
|
|
enabled: bool = False
|
|
smtp_host: str = ""
|
|
smtp_port: int = 587
|
|
smtp_user: str = ""
|
|
smtp_password: str = ""
|
|
use_starttls: bool = True
|
|
from_addr: str = ""
|
|
to_addrs: list[str] = field(default_factory=list)
|
|
on: str = "errors" # always | errors | never
|
|
|
|
|
|
@dataclass
|
|
class Config:
|
|
paths: Paths
|
|
ocr: OcrConfig
|
|
output: OutputConfig
|
|
verapdf: VeraPdfConfig
|
|
folder: FolderUpload
|
|
nextcloud: NextcloudUpload
|
|
sftp: SftpUpload
|
|
email: EmailNotify
|
|
log_level: str = "INFO"
|
|
|
|
|
|
def _section(data: dict[str, Any], *keys: str) -> dict[str, Any]:
|
|
cur: Any = data
|
|
for k in keys:
|
|
cur = cur.get(k, {}) if isinstance(cur, dict) else {}
|
|
return cur if isinstance(cur, dict) else {}
|
|
|
|
|
|
def _require_path(p: dict[str, Any], key: str, cfg_path: Path) -> Path:
|
|
"""Holt einen Pflicht-Pfad aus der [paths]-Sektion.
|
|
|
|
Wirft ConfigError mit klarer Meldung statt eines nackten KeyError.
|
|
"""
|
|
value = p.get(key)
|
|
if value is None or (isinstance(value, str) and not value.strip()):
|
|
raise ConfigError(
|
|
f"{cfg_path}: In der Sektion [paths] fehlt der Eintrag '{key}' "
|
|
f"(oder er ist leer). Bitte ergänzen, z.B. "
|
|
f'{key} = "/var/lib/pdf-ocr-hotfolder/{key}" '
|
|
f"— siehe config.example.toml."
|
|
)
|
|
return Path(str(value))
|
|
|
|
|
|
def load_config(path: str | Path) -> Config:
|
|
path = Path(path)
|
|
with path.open("rb") as f:
|
|
data = tomllib.load(f)
|
|
|
|
if not isinstance(data.get("paths"), dict):
|
|
raise ConfigError(
|
|
f"{path}: Die Sektion [paths] fehlt (oder ist keine Tabelle). "
|
|
"Sie muss die Einträge incoming, outgoing, working und error "
|
|
"enthalten — siehe config.example.toml."
|
|
)
|
|
|
|
p = _section(data, "paths")
|
|
paths = Paths(
|
|
incoming=_require_path(p, "incoming", path),
|
|
outgoing=_require_path(p, "outgoing", path),
|
|
working=_require_path(p, "working", path),
|
|
error=_require_path(p, "error", path),
|
|
)
|
|
|
|
ocr = OcrConfig(**{k: v for k, v in _section(data, "ocr").items()
|
|
if k in OcrConfig.__annotations__})
|
|
output = OutputConfig(**{k: v for k, v in _section(data, "output").items()
|
|
if k in OutputConfig.__annotations__})
|
|
verapdf = VeraPdfConfig(**{k: v for k, v in _section(data, "verapdf").items()
|
|
if k in VeraPdfConfig.__annotations__})
|
|
folder = FolderUpload(**{k: v for k, v in _section(data, "upload", "folder").items()
|
|
if k in FolderUpload.__annotations__})
|
|
nextcloud = NextcloudUpload(**{k: v for k, v in _section(data, "upload", "nextcloud").items()
|
|
if k in NextcloudUpload.__annotations__})
|
|
sftp = SftpUpload(**{k: v for k, v in _section(data, "upload", "sftp").items()
|
|
if k in SftpUpload.__annotations__})
|
|
email = EmailNotify(**{k: v for k, v in _section(data, "notify", "email").items()
|
|
if k in EmailNotify.__annotations__})
|
|
|
|
log_level = _section(data, "logging").get("level", "INFO")
|
|
|
|
return Config(
|
|
paths=paths, ocr=ocr, output=output, verapdf=verapdf,
|
|
folder=folder, nextcloud=nextcloud, sftp=sftp, email=email,
|
|
log_level=log_level,
|
|
)
|