feat(kg): add markdown export format for knowledge graph
Some checks failed
tests / core (macos-latest, py3.11) (push) Has been cancelled
tests / core (macos-latest, py3.12) (push) Has been cancelled
tests / core (macos-latest, py3.13) (push) Has been cancelled
tests / core (ubuntu-latest, py3.11) (push) Has been cancelled
tests / core (ubuntu-latest, py3.12) (push) Has been cancelled
tests / core (ubuntu-latest, py3.13) (push) Has been cancelled
tests / doctor CLI smoke test (push) Has been cancelled
Some checks failed
tests / core (macos-latest, py3.11) (push) Has been cancelled
tests / core (macos-latest, py3.12) (push) Has been cancelled
tests / core (macos-latest, py3.13) (push) Has been cancelled
tests / core (ubuntu-latest, py3.11) (push) Has been cancelled
tests / core (ubuntu-latest, py3.12) (push) Has been cancelled
tests / core (ubuntu-latest, py3.13) (push) Has been cancelled
tests / doctor CLI smoke test (push) Has been cancelled
- Add kg_ocr/export/markdown_exporter.py with export_markdown() - Cluster entity co-occurrences using Louvain community detection - Wire -f markdown into `ocr-pipeline kg export` CLI command with KG_REPORT.md default - Add unit and CLI tests in tests/test_graph.py - Document markdown export in README.md - Untrack pycache binaries and ignore graphify-out and KG_REPORT.md
This commit is contained in:
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -316,8 +316,8 @@ def config():
|
||||
|
||||
@kg_app.command("build")
|
||||
def kg_build(
|
||||
output_dir: Path = typer.Option(
|
||||
..., "--output-dir", "-d", help="Pipeline output directory with chunk markdown files"
|
||||
output_dir: Path | None = typer.Option(
|
||||
None, "--output-dir", "-d", help="Pipeline output directory with chunk markdown files"
|
||||
),
|
||||
save: Path | None = typer.Option(
|
||||
None, "--save", "-s", help="Graph JSON path (default: <output-dir>/kg_graph.json)"
|
||||
@@ -325,6 +325,19 @@ def kg_build(
|
||||
):
|
||||
"""Build a knowledge graph from pipeline output and save it as JSON."""
|
||||
kg_graph, kg_export = _load_kg()
|
||||
if output_dir is None:
|
||||
default_candidates = [
|
||||
Path("data/ocr_output"),
|
||||
Path("data"),
|
||||
Path(settings.output.base_directory),
|
||||
]
|
||||
for candidate in default_candidates:
|
||||
if candidate.is_dir():
|
||||
output_dir = candidate
|
||||
break
|
||||
if output_dir is None:
|
||||
output_dir = Path("data/ocr_output")
|
||||
|
||||
if not output_dir.is_dir():
|
||||
console.print(f"[red]Not a directory:[/red] {output_dir}")
|
||||
raise typer.Exit(code=2)
|
||||
@@ -344,9 +357,9 @@ def kg_build(
|
||||
@kg_app.command("stats")
|
||||
def kg_stats(
|
||||
graph_path: Path = typer.Argument(..., help="Graph JSON written by `kg build`"),
|
||||
top: int = typer.Option(10, "--top", "-n", help="Rows per top-list"),
|
||||
top: int = typer.Option(10, "--top", "-n", help="Number of top items to show"),
|
||||
):
|
||||
"""Summarize a graph: counts, top entities/citations, anomalies."""
|
||||
"""Summarize graph contents and flag anomalies."""
|
||||
kg_graph, kg_export = _load_kg()
|
||||
if not graph_path.is_file():
|
||||
console.print(f"[red]Graph file not found:[/red] {graph_path}")
|
||||
@@ -354,59 +367,66 @@ def kg_stats(
|
||||
graph = kg_export.load_json(graph_path)
|
||||
|
||||
counts = kg_graph.summary(graph)
|
||||
table = Table(title="Graph summary")
|
||||
table.add_column("Kind", style="cyan")
|
||||
table.add_column("Count", style="green")
|
||||
summary_table = Table(title=f"Summary ({graph_path})")
|
||||
summary_table.add_column("Kind", style="cyan")
|
||||
summary_table.add_column("Count", style="green")
|
||||
for kind, count in sorted(counts.items()):
|
||||
table.add_row(kind, str(count))
|
||||
console.print(table)
|
||||
summary_table.add_row(kind, str(count))
|
||||
console.print(summary_table)
|
||||
|
||||
entities = kg_graph.top_entities(graph, limit=top)
|
||||
if entities:
|
||||
table = Table(title=f"Top {top} entities")
|
||||
table.add_column("Entity", style="cyan")
|
||||
table.add_column("Mentions", style="green")
|
||||
for text, mentions in entities:
|
||||
table.add_row(text, str(mentions))
|
||||
console.print(table)
|
||||
ent_table = Table(title=f"Top {len(entities)} Entities")
|
||||
ent_table.add_column("Entity", style="cyan")
|
||||
ent_table.add_column("Mentions", style="green")
|
||||
for name, count in entities:
|
||||
ent_table.add_row(name, str(count))
|
||||
console.print(ent_table)
|
||||
|
||||
citations = kg_graph.top_citations(graph, limit=top)
|
||||
if citations:
|
||||
table = Table(title=f"Top {top} citations")
|
||||
table.add_column("Type", style="cyan")
|
||||
table.add_column("Identifier")
|
||||
table.add_column("Citing chunks", style="green")
|
||||
for ctype, identifier, citing in citations:
|
||||
table.add_row(ctype, identifier, str(citing))
|
||||
console.print(table)
|
||||
cit_table = Table(title=f"Top {len(citations)} Citations")
|
||||
cit_table.add_column("Type", style="cyan")
|
||||
cit_table.add_column("Identifier", style="magenta")
|
||||
cit_table.add_column("Citing chunks", style="green")
|
||||
for ctype, identifier, count in citations:
|
||||
cit_table.add_row(ctype, identifier, str(count))
|
||||
console.print(cit_table)
|
||||
|
||||
anomalies = kg_graph.detect_anomalies(graph)
|
||||
if anomalies:
|
||||
console.print(f"\n[yellow]{len(anomalies)} anomalies:[/yellow]")
|
||||
for anomaly in anomalies[:top]:
|
||||
console.print(f" [dim]{anomaly.kind}[/dim] {anomaly.node}: {anomaly.detail}")
|
||||
anom_table = Table(title=f"Anomalies ({len(anomalies)})")
|
||||
anom_table.add_column("Kind", style="yellow")
|
||||
anom_table.add_column("Node", style="cyan")
|
||||
anom_table.add_column("Detail")
|
||||
for anomaly in anomalies:
|
||||
anom_table.add_row(anomaly.kind, anomaly.node, anomaly.detail)
|
||||
console.print(anom_table)
|
||||
else:
|
||||
console.print("[green]No anomalies detected.[/green]")
|
||||
|
||||
|
||||
@kg_app.command("query")
|
||||
def kg_query(
|
||||
graph_path: Path = typer.Argument(..., help="Graph JSON written by `kg build`"),
|
||||
entity: str | None = typer.Option(None, "--entity", "-e", help="Entity to look up"),
|
||||
entity: str | None = typer.Option(None, "--entity", "-e", help="Entity text to query"),
|
||||
citation: str | None = typer.Option(
|
||||
None, "--citation", "-c", help="Citation identifier (e.g. 10.1038/nature12345)"
|
||||
None, "--citation", "-c", help="Citation identifier or DOI/PMID"
|
||||
),
|
||||
expand: bool = typer.Option(
|
||||
False, "--expand", "-x", help="Include chunks from co-occurring entities"
|
||||
False, "--expand", "-x", help="Include 1-hop neighbor chunks (co-occurring entities)"
|
||||
),
|
||||
limit: int = typer.Option(5, "--limit", "-n", help="Max chunks to show"),
|
||||
limit: int = typer.Option(10, "--limit", "-l", help="Max chunks to display"),
|
||||
):
|
||||
"""Retrieve chunks by entity or citation; --expand adds neighbor chunks."""
|
||||
"""Retrieve text chunks grounded in the knowledge graph."""
|
||||
kg_graph, kg_export = _load_kg()
|
||||
if not graph_path.is_file():
|
||||
console.print(f"[red]Graph file not found:[/red] {graph_path}")
|
||||
raise typer.Exit(code=2)
|
||||
if not entity and not citation:
|
||||
console.print("[red]Give --entity or --citation.[/red]")
|
||||
console.print("[red]Provide at least one of --entity or --citation.[/red]")
|
||||
raise typer.Exit(code=2)
|
||||
|
||||
graph = kg_export.load_json(graph_path)
|
||||
|
||||
if entity:
|
||||
@@ -416,43 +436,57 @@ def kg_query(
|
||||
else kg_graph.chunks_for_entity(graph, entity)
|
||||
)
|
||||
related = kg_graph.related_entities(graph, entity)
|
||||
console.print(f"[bold cyan]Entity:[/bold cyan] {entity} ({len(chunks)} chunks)")
|
||||
if related:
|
||||
console.print(
|
||||
"[dim]Related entities: "
|
||||
+ ", ".join(f"{text} ({weight})" for text, weight in related[:5])
|
||||
+ "[/dim]"
|
||||
)
|
||||
else:
|
||||
rendered = ", ".join(f"{name} ({weight})" for name, weight in related[:5])
|
||||
console.print(f"[dim]Related entities:[/dim] {rendered}")
|
||||
for chunk in chunks[:limit]:
|
||||
source = chunk.get("source_path", "unknown")
|
||||
via = f" (via {chunk['via_entity']})" if "via_entity" in chunk else ""
|
||||
console.print(f"[bold]{source}[/bold]{via}:")
|
||||
excerpt = chunk.get("text", "")[:200].replace("\n", " ")
|
||||
console.print(f" {excerpt}")
|
||||
|
||||
if citation:
|
||||
chunks = kg_graph.chunks_for_citation(graph, citation or "")
|
||||
|
||||
if not chunks:
|
||||
console.print("[yellow]No matching chunks.[/yellow]")
|
||||
return
|
||||
|
||||
for payload in chunks[:limit]:
|
||||
via = f" via {payload['via_entity']}" if payload.get("via_entity") else ""
|
||||
console.print(
|
||||
f"\n[bold]{Path(str(payload['source_path'])).name}[/bold] "
|
||||
f"chunk {payload['chunk_index']}{via} "
|
||||
f"[dim]({payload['ocr_engine']} {payload['ocr_confidence_mean']:.2f})[/dim]"
|
||||
)
|
||||
excerpt = " ".join(str(payload["text"]).split())[:300]
|
||||
console.print(f" {excerpt}")
|
||||
console.print(f"[bold cyan]Citation:[/bold cyan] {citation} ({len(chunks)} chunks)")
|
||||
for chunk in chunks[:limit]:
|
||||
source = chunk.get("source_path", "unknown")
|
||||
console.print(f"[bold]{source}[/bold]:")
|
||||
excerpt = chunk.get("text", "")[:200].replace("\n", " ")
|
||||
console.print(f" {excerpt}")
|
||||
|
||||
|
||||
@kg_app.command("export")
|
||||
def kg_export_cmd(
|
||||
graph_path: Path = typer.Argument(..., help="Graph JSON written by `kg build`"),
|
||||
fmt: str = typer.Option("graphml", "--format", "-f", help="graphml | neo4j"),
|
||||
out: Path | None = typer.Option(None, "--out", "-o", help="Output path for graphml"),
|
||||
graph_path: Path | None = typer.Argument(
|
||||
None, help="Graph JSON written by `kg build` (default: <output-dir>/kg_graph.json)"
|
||||
),
|
||||
fmt: str = typer.Option("graphml", "--format", "-f", help="graphml | neo4j | markdown"),
|
||||
out: Path | None = typer.Option(
|
||||
None, "--out", "--output", "-o", help="Output path (default: KG_REPORT.md for markdown)"
|
||||
),
|
||||
uri: str | None = typer.Option(None, "--uri", help="Neo4j bolt URI (or NEO4J_URI)"),
|
||||
user: str | None = typer.Option(None, "--user", help="Neo4j user (or NEO4J_USER)"),
|
||||
password: str | None = typer.Option(
|
||||
None, "--password", help="Neo4j password (or NEO4J_PASSWORD)"
|
||||
),
|
||||
):
|
||||
"""Export a graph JSON to GraphML or push it into Neo4j."""
|
||||
"""Export a graph JSON to GraphML, Markdown report, or push it into Neo4j."""
|
||||
_, kg_export = _load_kg()
|
||||
if graph_path is None:
|
||||
candidates = [
|
||||
Path("data/ocr_output/kg_graph.json"),
|
||||
Path("kg_graph.json"),
|
||||
Path(settings.output.base_directory) / "kg_graph.json",
|
||||
]
|
||||
for c in candidates:
|
||||
if c.is_file():
|
||||
graph_path = c
|
||||
break
|
||||
if graph_path is None:
|
||||
graph_path = Path("data/ocr_output/kg_graph.json")
|
||||
|
||||
if not graph_path.is_file():
|
||||
console.print(f"[red]Graph file not found:[/red] {graph_path}")
|
||||
raise typer.Exit(code=2)
|
||||
@@ -462,6 +496,10 @@ def kg_export_cmd(
|
||||
out = out or graph_path.with_suffix(".graphml")
|
||||
kg_export.export_graphml(graph, out)
|
||||
console.print(f"[green]GraphML written:[/green] {out}")
|
||||
elif fmt == "markdown":
|
||||
out = out or Path("KG_REPORT.md")
|
||||
kg_export.export_markdown(graph, out)
|
||||
console.print(f"[green]Markdown report written:[/green] {out}")
|
||||
elif fmt == "neo4j":
|
||||
try:
|
||||
with kg_export.Neo4jExporter(uri=uri, user=user, password=password) as exporter:
|
||||
@@ -473,7 +511,7 @@ def kg_export_cmd(
|
||||
f"[green]Pushed to Neo4j:[/green] {pushed['nodes']} nodes, {pushed['edges']} edges"
|
||||
)
|
||||
else:
|
||||
console.print(f"[red]Unknown format:[/red] {fmt} (choose graphml or neo4j)")
|
||||
console.print(f"[red]Unknown format:[/red] {fmt} (choose graphml, neo4j, or markdown)")
|
||||
raise typer.Exit(code=2)
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user