refactor: solidify startup UX and engine-aware preprocessing
- Slim core deps: move ML stack to optional extras (paddle/tables/figures/scientific/full) - Lazy settings proxy with config search paths (env var, cwd, user dir) - New commands: init, demo, setup [basic|full], first-run guard on run/watch - Engine-aware preprocessing: Paddle gets original image (fixes dark mode 0.83->0.95) - Results table shows Skipped count; lazy run-dir creation - kg_ocr marked experimental with extra, Docker defaults with OCR_PIPELINE_CONFIG - 25/25 tests, ruff clean
This commit is contained in:
@@ -46,8 +46,10 @@ class OCRPipeline:
|
||||
return []
|
||||
|
||||
original_image = self.preprocessor.load_image(image_path)
|
||||
ocr_image = self.preprocessor.preprocess_image(original_image)
|
||||
ocr_result = self.engine.process(ocr_image)
|
||||
# PaddleOCR reads the (size-capped) original; Tesseract the binarized derivative.
|
||||
ocr_base = self.preprocessor.resize_if_needed(original_image)
|
||||
preprocessed = self.preprocessor.preprocess_image(original_image)
|
||||
ocr_result = self.engine.process(ocr_base, preprocessed)
|
||||
cleaned = clean_text(ocr_result.text)
|
||||
chunks = chunk_text(cleaned.text)
|
||||
entities = extract_entities(cleaned.text)
|
||||
@@ -60,16 +62,42 @@ class OCRPipeline:
|
||||
output_files: list[Path] = []
|
||||
for chunk in chunks:
|
||||
metadata = {
|
||||
"source_path": str(image_path), "source_hash": image_hash, "timestamp": timestamp,
|
||||
"ocr_engine": ocr_result.engine, "ocr_confidence_mean": ocr_result.confidence,
|
||||
"language": ocr_result.language, "entity_extraction_backend": entities.backend, "detected_entities": [entity.text for entity in entities.entities],
|
||||
"has_figures": bool(figures), "has_tables": bool(tables),
|
||||
"citations_found": [f"{citation.type}:{citation.identifier}" for citation in citations],
|
||||
"chunk_index": chunk.chunk_index, "total_chunks": len(chunks),
|
||||
"source_path": str(image_path),
|
||||
"source_hash": image_hash,
|
||||
"timestamp": timestamp,
|
||||
"ocr_engine": ocr_result.engine,
|
||||
"ocr_confidence_mean": ocr_result.confidence,
|
||||
"language": ocr_result.language,
|
||||
"entity_extraction_backend": entities.backend,
|
||||
"detected_entities": [entity.text for entity in entities.entities],
|
||||
"has_figures": bool(figures),
|
||||
"has_tables": bool(tables),
|
||||
"citations_found": [
|
||||
f"{citation.type}:{citation.identifier}" for citation in citations
|
||||
],
|
||||
"chunk_index": chunk.chunk_index,
|
||||
"total_chunks": len(chunks),
|
||||
}
|
||||
output_files.append(self.writer.write_chunk(content=chunk.content, metadata=metadata, figures=figures, tables=tables, entities=entities, citations=citations))
|
||||
output_files.append(
|
||||
self.writer.write_chunk(
|
||||
content=chunk.content,
|
||||
metadata=metadata,
|
||||
figures=figures,
|
||||
tables=tables,
|
||||
entities=entities,
|
||||
citations=citations,
|
||||
)
|
||||
)
|
||||
|
||||
self.db.mark_processed(file_hash=image_hash, path=str(image_path), timestamp=time.time(), engine=ocr_result.engine, confidence=ocr_result.confidence, chunks=len(chunks), run_id=self._run_id)
|
||||
self.db.mark_processed(
|
||||
file_hash=image_hash,
|
||||
path=str(image_path),
|
||||
timestamp=time.time(),
|
||||
engine=ocr_result.engine,
|
||||
confidence=ocr_result.confidence,
|
||||
chunks=len(chunks),
|
||||
run_id=self._run_id,
|
||||
)
|
||||
return output_files
|
||||
|
||||
def process_single(self, image_path: Path) -> list[Path]:
|
||||
@@ -108,7 +136,15 @@ class OCRPipeline:
|
||||
if consolidated is not None:
|
||||
result.output_files.append(consolidated)
|
||||
finally:
|
||||
self.db.complete_run(self._run_id, {"total": result.total_images, "successful": result.successful, "failed": result.failed, "chunks": result.chunks_created})
|
||||
self.db.complete_run(
|
||||
self._run_id,
|
||||
{
|
||||
"total": result.total_images,
|
||||
"successful": result.successful,
|
||||
"failed": result.failed,
|
||||
"chunks": result.chunks_created,
|
||||
},
|
||||
)
|
||||
self._run_id = None
|
||||
return result
|
||||
|
||||
|
||||
Reference in New Issue
Block a user