refactor: solidify startup UX and engine-aware preprocessing

- Slim core deps: move ML stack to optional extras (paddle/tables/figures/scientific/full)
- Lazy settings proxy with config search paths (env var, cwd, user dir)
- New commands: init, demo, setup [basic|full], first-run guard on run/watch
- Engine-aware preprocessing: Paddle gets original image (fixes dark mode 0.83->0.95)
- Results table shows Skipped count; lazy run-dir creation
- kg_ocr marked experimental with extra, Docker defaults with OCR_PIPELINE_CONFIG
- 25/25 tests, ruff clean
This commit is contained in:
2026-07-19 22:14:54 +02:00
parent deab02610b
commit a0211155fa
59 changed files with 2426 additions and 629 deletions

View File

@@ -46,8 +46,10 @@ class OCRPipeline:
return []
original_image = self.preprocessor.load_image(image_path)
ocr_image = self.preprocessor.preprocess_image(original_image)
ocr_result = self.engine.process(ocr_image)
# PaddleOCR reads the (size-capped) original; Tesseract the binarized derivative.
ocr_base = self.preprocessor.resize_if_needed(original_image)
preprocessed = self.preprocessor.preprocess_image(original_image)
ocr_result = self.engine.process(ocr_base, preprocessed)
cleaned = clean_text(ocr_result.text)
chunks = chunk_text(cleaned.text)
entities = extract_entities(cleaned.text)
@@ -60,16 +62,42 @@ class OCRPipeline:
output_files: list[Path] = []
for chunk in chunks:
metadata = {
"source_path": str(image_path), "source_hash": image_hash, "timestamp": timestamp,
"ocr_engine": ocr_result.engine, "ocr_confidence_mean": ocr_result.confidence,
"language": ocr_result.language, "entity_extraction_backend": entities.backend, "detected_entities": [entity.text for entity in entities.entities],
"has_figures": bool(figures), "has_tables": bool(tables),
"citations_found": [f"{citation.type}:{citation.identifier}" for citation in citations],
"chunk_index": chunk.chunk_index, "total_chunks": len(chunks),
"source_path": str(image_path),
"source_hash": image_hash,
"timestamp": timestamp,
"ocr_engine": ocr_result.engine,
"ocr_confidence_mean": ocr_result.confidence,
"language": ocr_result.language,
"entity_extraction_backend": entities.backend,
"detected_entities": [entity.text for entity in entities.entities],
"has_figures": bool(figures),
"has_tables": bool(tables),
"citations_found": [
f"{citation.type}:{citation.identifier}" for citation in citations
],
"chunk_index": chunk.chunk_index,
"total_chunks": len(chunks),
}
output_files.append(self.writer.write_chunk(content=chunk.content, metadata=metadata, figures=figures, tables=tables, entities=entities, citations=citations))
output_files.append(
self.writer.write_chunk(
content=chunk.content,
metadata=metadata,
figures=figures,
tables=tables,
entities=entities,
citations=citations,
)
)
self.db.mark_processed(file_hash=image_hash, path=str(image_path), timestamp=time.time(), engine=ocr_result.engine, confidence=ocr_result.confidence, chunks=len(chunks), run_id=self._run_id)
self.db.mark_processed(
file_hash=image_hash,
path=str(image_path),
timestamp=time.time(),
engine=ocr_result.engine,
confidence=ocr_result.confidence,
chunks=len(chunks),
run_id=self._run_id,
)
return output_files
def process_single(self, image_path: Path) -> list[Path]:
@@ -108,7 +136,15 @@ class OCRPipeline:
if consolidated is not None:
result.output_files.append(consolidated)
finally:
self.db.complete_run(self._run_id, {"total": result.total_images, "successful": result.successful, "failed": result.failed, "chunks": result.chunks_created})
self.db.complete_run(
self._run_id,
{
"total": result.total_images,
"successful": result.successful,
"failed": result.failed,
"chunks": result.chunks_created,
},
)
self._run_id = None
return result