578472872e
- [ocr].timeout wird als tesseract_timeout (pro Seite) an ocrmypdf durchgereicht; Default 1800 -> 300, 0 = ocrmypdf-Default - Exceptions nach dem OCR zaehlen als Fehler, Datei wird nach error/ gerettet - Fehlgeschlagene Uploads zaehlen als Fehler und loesen Fehler-Mail aus - name_mode wird im Preflight geprueft, nicht erst pro Datei - Fehlende [paths]-Sektion -> ConfigError mit klarer Meldung statt KeyError - Stabilitaets-Timeout zaehlt als Fehler (--once liefert Exit 1) - upload_folder nutzt shutil.copyfile statt read_bytes/write_bytes - OcrConfig.pdfa_level Default "2" -> "" (Ghostscript-Bug, Issue #3) - 35 neue Tests (92 gesamt), pytest.ini - AI_AGENT_BRIEFING.md auf Stand 0.4.0 gebracht Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
82 lines
2.8 KiB
Python
82 lines
2.8 KiB
Python
"""Tests für [ocr].timeout → ocrmypdf `tesseract_timeout`.
|
|
|
|
ocrmypdf wird hier komplett gemockt (per sys.modules), es läuft also nie
|
|
wirklich — die Tests laufen auch ohne installiertes ocrmypdf.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import sys
|
|
import tomllib
|
|
from pathlib import Path
|
|
from types import ModuleType
|
|
|
|
import pytest
|
|
|
|
from pdf_ocr_hotfolder.config import OcrConfig
|
|
from pdf_ocr_hotfolder.processor import run_ocr
|
|
|
|
|
|
@pytest.fixture
|
|
def fake_ocrmypdf(monkeypatch) -> ModuleType:
|
|
"""Schiebt ein Dummy-ocrmypdf in sys.modules und merkt sich die kwargs."""
|
|
mod = ModuleType("ocrmypdf")
|
|
mod.calls = [] # type: ignore[attr-defined]
|
|
|
|
def ocr(src, dst, **kwargs):
|
|
mod.calls.append({"src": src, "dst": dst, "kwargs": kwargs}) # type: ignore[attr-defined]
|
|
Path(dst).write_bytes(b"%PDF-1.4 ocr\n")
|
|
|
|
mod.ocr = ocr # type: ignore[attr-defined]
|
|
monkeypatch.setitem(sys.modules, "ocrmypdf", mod)
|
|
return mod
|
|
|
|
|
|
def _run(fake, tmp_path: Path, cfg: OcrConfig) -> dict:
|
|
src = tmp_path / "in.pdf"
|
|
src.write_bytes(b"%PDF-1.4\n")
|
|
run_ocr(src, tmp_path / "out.pdf", cfg)
|
|
assert len(fake.calls) == 1
|
|
return fake.calls[0]["kwargs"]
|
|
|
|
|
|
def test_timeout_is_passed_as_tesseract_timeout(fake_ocrmypdf, tmp_path: Path) -> None:
|
|
kwargs = _run(fake_ocrmypdf, tmp_path, OcrConfig(timeout=120))
|
|
assert kwargs["tesseract_timeout"] == 120.0
|
|
|
|
|
|
def test_timeout_zero_means_no_limit(fake_ocrmypdf, tmp_path: Path) -> None:
|
|
"""0 = kein Limit → der Key darf NICHT durchgereicht werden.
|
|
|
|
ocrmypdf würde tesseract_timeout=0 als 'OCR überspringen' auslegen.
|
|
"""
|
|
kwargs = _run(fake_ocrmypdf, tmp_path, OcrConfig(timeout=0))
|
|
assert "tesseract_timeout" not in kwargs
|
|
|
|
|
|
def test_negative_timeout_is_ignored(fake_ocrmypdf, tmp_path: Path) -> None:
|
|
kwargs = _run(fake_ocrmypdf, tmp_path, OcrConfig(timeout=-5))
|
|
assert "tesseract_timeout" not in kwargs
|
|
|
|
|
|
def test_default_timeout_is_passed(fake_ocrmypdf, tmp_path: Path) -> None:
|
|
kwargs = _run(fake_ocrmypdf, tmp_path, OcrConfig())
|
|
assert kwargs["tesseract_timeout"] == 300.0
|
|
|
|
|
|
def test_other_kwargs_still_present(fake_ocrmypdf, tmp_path: Path) -> None:
|
|
"""Der neue Key ersetzt nichts Bestehendes."""
|
|
kwargs = _run(fake_ocrmypdf, tmp_path,
|
|
OcrConfig(languages="deu", jobs=2, pdfa_level=""))
|
|
assert kwargs["language"] == "deu"
|
|
assert kwargs["jobs"] == 2
|
|
assert kwargs["output_type"] == "pdf"
|
|
assert kwargs["skip_text"] is True
|
|
|
|
|
|
def test_config_default_matches_example(tmp_path: Path) -> None:
|
|
"""Dataclass-Default und config.example.toml dürfen nicht auseinanderlaufen."""
|
|
cfg_path = Path(__file__).parent.parent / "config.example.toml"
|
|
with cfg_path.open("rb") as f:
|
|
data = tomllib.load(f)
|
|
assert data["ocr"]["timeout"] == OcrConfig().timeout == 300
|