kg: graph build, traversal queries, neo4j export - kg_ocr.graph builds a networkx graph (docs, chunks, entities, citations, co-occurrence) from chunk markdown - analyzer for summaries, top entities/citations, anomaly checks - traversal: chunks_for_entity/citation, related_entities, expand_context - export: JSON round-trip, GraphML, batched MERGE into neo4j - new CLI: ocr-pipeline kg build|stats|query|export - lazy kg_ocr imports, networkx/neo4j behind extras - dropped dead watch.py shim, added KgConfig stub - trimmed README, updated TODO
Some checks failed
tests / core (macos-latest, py3.11) (push) Has been cancelled
tests / core (macos-latest, py3.12) (push) Has been cancelled
tests / core (macos-latest, py3.13) (push) Has been cancelled
tests / core (ubuntu-latest, py3.11) (push) Has been cancelled
tests / core (ubuntu-latest, py3.12) (push) Has been cancelled
tests / core (ubuntu-latest, py3.13) (push) Has been cancelled
tests / doctor CLI smoke test (push) Has been cancelled

This commit is contained in:
2026-07-20 19:50:38 +02:00
parent 39655fc35f
commit 7503a441c2
29 changed files with 1343 additions and 160 deletions

View File

@@ -28,10 +28,27 @@ app = typer.Typer(
help="OCR pipeline for screenshots -> RAG-ready Markdown",
add_completion=False,
)
kg_app = typer.Typer(
name="kg",
help="Knowledge graph over OCR output (needs: uv sync --extra kg)",
add_completion=False,
)
app.add_typer(kg_app, name="kg")
console = Console()
logger = get_logger(__name__)
def _load_kg():
"""Import kg_ocr lazily so the base CLI works without the kg extra."""
try:
from kg_ocr import export as kg_export
from kg_ocr import graph as kg_graph
except ImportError as exc:
console.print(f"[red]{exc}[/red]")
raise typer.Exit(code=2) from exc
return kg_graph, kg_export
def _ensure_configured(input_dir: Path | None) -> None:
"""Stop early with guidance instead of scanning a default directory."""
if input_dir is not None or find_config_file() is not None:
@@ -296,6 +313,169 @@ def config():
console.print(settings.model_dump_json(indent=2))
@kg_app.command("build")
def kg_build(
output_dir: Path = typer.Option(
..., "--output-dir", "-d", help="Pipeline output directory with chunk markdown files"
),
save: Path | None = typer.Option(
None, "--save", "-s", help="Graph JSON path (default: <output-dir>/kg_graph.json)"
),
):
"""Build a knowledge graph from pipeline output and save it as JSON."""
kg_graph, kg_export = _load_kg()
if not output_dir.is_dir():
console.print(f"[red]Not a directory:[/red] {output_dir}")
raise typer.Exit(code=2)
graph = kg_graph.build_from_directory(output_dir)
save = save or (output_dir / "kg_graph.json")
kg_export.export_json(graph, save)
counts = kg_graph.summary(graph)
table = Table(title=f"Knowledge graph -> {save}")
table.add_column("Kind", style="cyan")
table.add_column("Count", style="green")
for kind, count in sorted(counts.items()):
table.add_row(kind, str(count))
console.print(table)
@kg_app.command("stats")
def kg_stats(
graph_path: Path = typer.Argument(..., help="Graph JSON written by `kg build`"),
top: int = typer.Option(10, "--top", "-n", help="Rows per top-list"),
):
"""Summarize a graph: counts, top entities/citations, anomalies."""
kg_graph, kg_export = _load_kg()
if not graph_path.is_file():
console.print(f"[red]Graph file not found:[/red] {graph_path}")
raise typer.Exit(code=2)
graph = kg_export.load_json(graph_path)
counts = kg_graph.summary(graph)
table = Table(title="Graph summary")
table.add_column("Kind", style="cyan")
table.add_column("Count", style="green")
for kind, count in sorted(counts.items()):
table.add_row(kind, str(count))
console.print(table)
entities = kg_graph.top_entities(graph, limit=top)
if entities:
table = Table(title=f"Top {top} entities")
table.add_column("Entity", style="cyan")
table.add_column("Mentions", style="green")
for text, mentions in entities:
table.add_row(text, str(mentions))
console.print(table)
citations = kg_graph.top_citations(graph, limit=top)
if citations:
table = Table(title=f"Top {top} citations")
table.add_column("Type", style="cyan")
table.add_column("Identifier")
table.add_column("Citing chunks", style="green")
for ctype, identifier, citing in citations:
table.add_row(ctype, identifier, str(citing))
console.print(table)
anomalies = kg_graph.detect_anomalies(graph)
if anomalies:
console.print(f"\n[yellow]{len(anomalies)} anomalies:[/yellow]")
for anomaly in anomalies[:top]:
console.print(f" [dim]{anomaly.kind}[/dim] {anomaly.node}: {anomaly.detail}")
@kg_app.command("query")
def kg_query(
graph_path: Path = typer.Argument(..., help="Graph JSON written by `kg build`"),
entity: str | None = typer.Option(None, "--entity", "-e", help="Entity to look up"),
citation: str | None = typer.Option(
None, "--citation", "-c", help="Citation identifier (e.g. 10.1038/nature12345)"
),
expand: bool = typer.Option(
False, "--expand", "-x", help="Include chunks from co-occurring entities"
),
limit: int = typer.Option(5, "--limit", "-n", help="Max chunks to show"),
):
"""Retrieve chunks by entity or citation; --expand adds neighbor chunks."""
kg_graph, kg_export = _load_kg()
if not graph_path.is_file():
console.print(f"[red]Graph file not found:[/red] {graph_path}")
raise typer.Exit(code=2)
if not entity and not citation:
console.print("[red]Give --entity or --citation.[/red]")
raise typer.Exit(code=2)
graph = kg_export.load_json(graph_path)
if entity:
chunks = (
kg_graph.expand_context(graph, entity, limit=limit)
if expand
else kg_graph.chunks_for_entity(graph, entity)
)
related = kg_graph.related_entities(graph, entity)
if related:
console.print(
"[dim]Related entities: "
+ ", ".join(f"{text} ({weight})" for text, weight in related[:5])
+ "[/dim]"
)
else:
chunks = kg_graph.chunks_for_citation(graph, citation or "")
if not chunks:
console.print("[yellow]No matching chunks.[/yellow]")
return
for payload in chunks[:limit]:
via = f" via {payload['via_entity']}" if payload.get("via_entity") else ""
console.print(
f"\n[bold]{Path(str(payload['source_path'])).name}[/bold] "
f"chunk {payload['chunk_index']}{via} "
f"[dim]({payload['ocr_engine']} {payload['ocr_confidence_mean']:.2f})[/dim]"
)
excerpt = " ".join(str(payload["text"]).split())[:300]
console.print(f" {excerpt}")
@kg_app.command("export")
def kg_export_cmd(
graph_path: Path = typer.Argument(..., help="Graph JSON written by `kg build`"),
fmt: str = typer.Option("graphml", "--format", "-f", help="graphml | neo4j"),
out: Path | None = typer.Option(None, "--out", "-o", help="Output path for graphml"),
uri: str | None = typer.Option(None, "--uri", help="Neo4j bolt URI (or NEO4J_URI)"),
user: str | None = typer.Option(None, "--user", help="Neo4j user (or NEO4J_USER)"),
password: str | None = typer.Option(
None, "--password", help="Neo4j password (or NEO4J_PASSWORD)"
),
):
"""Export a graph JSON to GraphML or push it into Neo4j."""
_, kg_export = _load_kg()
if not graph_path.is_file():
console.print(f"[red]Graph file not found:[/red] {graph_path}")
raise typer.Exit(code=2)
graph = kg_export.load_json(graph_path)
if fmt == "graphml":
out = out or graph_path.with_suffix(".graphml")
kg_export.export_graphml(graph, out)
console.print(f"[green]GraphML written:[/green] {out}")
elif fmt == "neo4j":
try:
with kg_export.Neo4jExporter(uri=uri, user=user, password=password) as exporter:
pushed = exporter.push(graph)
except ImportError as exc:
console.print(f"[red]{exc}[/red]")
raise typer.Exit(code=2) from exc
console.print(
f"[green]Pushed to Neo4j:[/green] {pushed['nodes']} nodes, {pushed['edges']} edges"
)
else:
console.print(f"[red]Unknown format:[/red] {fmt} (choose graphml or neo4j)")
raise typer.Exit(code=2)
def main():
app()

View File

@@ -1,5 +0,0 @@
"""Compatibility module for the canonical watcher package."""
from .watch.watcher import ScreenshotHandler, Watcher
__all__ = ["ScreenshotHandler", "Watcher"]