PaddleOCR with preprocessing, scispaCy NER, figure/table detection, citation extraction, and chunked Markdown output with frontmatter. Includes watch mode and notebook reprocessing.
13 lines
199 B
Python
13 lines
199 B
Python
from __future__ import annotations
|
|
|
|
__version__ = "0.1.0"
|
|
|
|
from .cli import main
|
|
from .pipeline import OCRPipeline, PipelineResult
|
|
|
|
__all__ = [
|
|
"OCRPipeline",
|
|
"PipelineResult",
|
|
"main",
|
|
]
|