feat(kg): add markdown export format for knowledge graph
Some checks failed
tests / core (macos-latest, py3.11) (push) Has been cancelled
tests / core (macos-latest, py3.12) (push) Has been cancelled
tests / core (macos-latest, py3.13) (push) Has been cancelled
tests / core (ubuntu-latest, py3.11) (push) Has been cancelled
tests / core (ubuntu-latest, py3.12) (push) Has been cancelled
tests / core (ubuntu-latest, py3.13) (push) Has been cancelled
tests / doctor CLI smoke test (push) Has been cancelled
Some checks failed
tests / core (macos-latest, py3.11) (push) Has been cancelled
tests / core (macos-latest, py3.12) (push) Has been cancelled
tests / core (macos-latest, py3.13) (push) Has been cancelled
tests / core (ubuntu-latest, py3.11) (push) Has been cancelled
tests / core (ubuntu-latest, py3.12) (push) Has been cancelled
tests / core (ubuntu-latest, py3.13) (push) Has been cancelled
tests / doctor CLI smoke test (push) Has been cancelled
- Add kg_ocr/export/markdown_exporter.py with export_markdown() - Cluster entity co-occurrences using Louvain community detection - Wire -f markdown into `ocr-pipeline kg export` CLI command with KG_REPORT.md default - Add unit and CLI tests in tests/test_graph.py - Document markdown export in README.md - Untrack pycache binaries and ignore graphify-out and KG_REPORT.md
This commit is contained in:
4
.gitignore
vendored
4
.gitignore
vendored
@@ -80,4 +80,6 @@ Thumbs.db
|
|||||||
# Project specific
|
# Project specific
|
||||||
data/
|
data/
|
||||||
notebooks/
|
notebooks/
|
||||||
.env
|
.env
|
||||||
|
graphify-out/
|
||||||
|
KG_REPORT.md
|
||||||
|
|||||||
@@ -79,6 +79,7 @@ uv run ocr-pipeline kg query data/ocr_output/kg_graph.json --entity BRCA1 --expa
|
|||||||
uv run ocr-pipeline kg query data/ocr_output/kg_graph.json --citation 10.1038/nature12345
|
uv run ocr-pipeline kg query data/ocr_output/kg_graph.json --citation 10.1038/nature12345
|
||||||
|
|
||||||
# export
|
# export
|
||||||
|
uv run ocr-pipeline kg export data/ocr_output/kg_graph.json -f markdown
|
||||||
uv run ocr-pipeline kg export data/ocr_output/kg_graph.json -f graphml
|
uv run ocr-pipeline kg export data/ocr_output/kg_graph.json -f graphml
|
||||||
uv run ocr-pipeline kg export data/ocr_output/kg_graph.json -f neo4j
|
uv run ocr-pipeline kg export data/ocr_output/kg_graph.json -f neo4j
|
||||||
```
|
```
|
||||||
@@ -239,4 +240,3 @@ Gene/protein names go through scispaCy's `en_core_sci_lg`. Chemical formulas
|
|||||||
and units (µM, ng/mL, kb/Mb/Gb, °C, ×g) get normalized, scientific notation
|
and units (µM, ng/mL, kb/Mb/Gb, °C, ×g) get normalized, scientific notation
|
||||||
gets cleaned up (`1.5×10⁻³` → `1.5×10^-3`), and gel/blot figures get their
|
gets cleaned up (`1.5×10⁻³` → `1.5×10^-3`), and gel/blot figures get their
|
||||||
captions pulled out separately.
|
captions pulled out separately.
|
||||||
|
|
||||||
|
|||||||
@@ -25,6 +25,7 @@ _LAZY: dict[str, tuple[str, str]] = {
|
|||||||
"expand_context": (".graph", "expand_context"),
|
"expand_context": (".graph", "expand_context"),
|
||||||
"export_json": (".export", "export_json"),
|
"export_json": (".export", "export_json"),
|
||||||
"export_graphml": (".export", "export_graphml"),
|
"export_graphml": (".export", "export_graphml"),
|
||||||
|
"export_markdown": (".export", "export_markdown"),
|
||||||
"load_json": (".export", "load_json"),
|
"load_json": (".export", "load_json"),
|
||||||
"Neo4jExporter": (".export", "Neo4jExporter"),
|
"Neo4jExporter": (".export", "Neo4jExporter"),
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,3 +1,4 @@
|
|||||||
|
from .markdown_exporter import export_markdown
|
||||||
from .neo4j_exporter import Neo4jExporter, export_graphml, export_json, load_json
|
from .neo4j_exporter import Neo4jExporter, export_graphml, export_json, load_json
|
||||||
|
|
||||||
__all__ = ["Neo4jExporter", "export_graphml", "export_json", "load_json"]
|
__all__ = ["Neo4jExporter", "export_graphml", "export_json", "export_markdown", "load_json"]
|
||||||
|
|||||||
270
kg_ocr/export/markdown_exporter.py
Normal file
270
kg_ocr/export/markdown_exporter.py
Normal file
@@ -0,0 +1,270 @@
|
|||||||
|
"""Export a knowledge graph to a Markdown report."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from kg_ocr.graph.analyzer import (
|
||||||
|
Anomaly,
|
||||||
|
detect_anomalies,
|
||||||
|
summary,
|
||||||
|
top_citations,
|
||||||
|
top_cooccurrences,
|
||||||
|
top_entities,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _load_graph(graph_input: Any) -> Any:
|
||||||
|
"""Load a NetworkX graph from a file path or return it directly."""
|
||||||
|
import networkx as nx
|
||||||
|
|
||||||
|
if isinstance(graph_input, (str, Path)):
|
||||||
|
path = Path(graph_input)
|
||||||
|
data = json.loads(path.read_text(encoding="utf-8"))
|
||||||
|
try:
|
||||||
|
return nx.node_link_graph(data, edges="edges")
|
||||||
|
except TypeError:
|
||||||
|
return nx.node_link_graph(data)
|
||||||
|
return graph_input
|
||||||
|
|
||||||
|
|
||||||
|
def _cluster_entity_cooccurrences(graph: Any) -> list[list[dict[str, Any]]]:
|
||||||
|
"""Find concept clusters over the entity co-occurrence subgraph."""
|
||||||
|
import networkx as nx
|
||||||
|
|
||||||
|
cooccur_graph = nx.Graph()
|
||||||
|
for node, data in graph.nodes(data=True):
|
||||||
|
if data.get("kind") == "entity":
|
||||||
|
cooccur_graph.add_node(
|
||||||
|
node,
|
||||||
|
text=data.get("text", node.replace("entity:", "")),
|
||||||
|
mentions=int(data.get("mentions", 0)),
|
||||||
|
)
|
||||||
|
|
||||||
|
for u, v, data in graph.edges(data=True):
|
||||||
|
if data.get("kind") == "CO_OCCURS":
|
||||||
|
cooccur_graph.add_edge(u, v, weight=data.get("weight", 1))
|
||||||
|
|
||||||
|
if cooccur_graph.number_of_nodes() == 0:
|
||||||
|
return []
|
||||||
|
|
||||||
|
# Community detection
|
||||||
|
communities: list[set[str]] = []
|
||||||
|
if cooccur_graph.number_of_edges() > 0:
|
||||||
|
try:
|
||||||
|
import networkx.algorithms.community as nx_comm # type: ignore[import-untyped]
|
||||||
|
|
||||||
|
communities = list(
|
||||||
|
nx_comm.louvain_communities(cooccur_graph, weight="weight", seed=42)
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
try:
|
||||||
|
import networkx.algorithms.community as nx_comm # type: ignore[import-untyped]
|
||||||
|
|
||||||
|
communities = list(
|
||||||
|
nx_comm.greedy_modularity_communities(cooccur_graph, weight="weight")
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
try:
|
||||||
|
import community as community_louvain # type: ignore[import-not-found]
|
||||||
|
|
||||||
|
partition = community_louvain.best_partition(
|
||||||
|
cooccur_graph, weight="weight", random_state=42
|
||||||
|
)
|
||||||
|
comm_dict: dict[int, set[str]] = {}
|
||||||
|
for n, cid in partition.items():
|
||||||
|
comm_dict.setdefault(cid, set()).add(n)
|
||||||
|
communities = list(comm_dict.values())
|
||||||
|
except Exception:
|
||||||
|
communities = list(nx.connected_components(cooccur_graph))
|
||||||
|
else:
|
||||||
|
# No edges: each entity is an isolated node
|
||||||
|
communities = [{n} for n in cooccur_graph.nodes()]
|
||||||
|
|
||||||
|
# Format and sort clusters
|
||||||
|
clusters: list[list[dict[str, Any]]] = []
|
||||||
|
for comm in communities:
|
||||||
|
cluster_nodes = []
|
||||||
|
for node in comm:
|
||||||
|
cluster_nodes.append(
|
||||||
|
{
|
||||||
|
"id": node,
|
||||||
|
"text": cooccur_graph.nodes[node].get("text", node),
|
||||||
|
"mentions": cooccur_graph.nodes[node].get("mentions", 0),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
# Sort entities inside cluster by mentions descending
|
||||||
|
cluster_nodes.sort(key=lambda item: item["mentions"], reverse=True)
|
||||||
|
clusters.append(cluster_nodes)
|
||||||
|
|
||||||
|
# Sort clusters by size (number of entities) descending, then top entity mentions
|
||||||
|
clusters.sort(
|
||||||
|
key=lambda c: (len(c), c[0]["mentions"] if c else 0),
|
||||||
|
reverse=True,
|
||||||
|
)
|
||||||
|
return clusters
|
||||||
|
|
||||||
|
|
||||||
|
def export_markdown(graph_path: str | Path, output_path: str | Path) -> Path:
|
||||||
|
"""Render a knowledge graph into a Markdown summary report.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
graph_path : str | Path | nx.Graph
|
||||||
|
Path to `kg_graph.json` or an in-memory NetworkX graph.
|
||||||
|
output_path : str | Path
|
||||||
|
Target path for the exported Markdown report.
|
||||||
|
|
||||||
|
Returns
|
||||||
|
-------
|
||||||
|
Path
|
||||||
|
The output Path written to.
|
||||||
|
"""
|
||||||
|
graph = _load_graph(graph_path)
|
||||||
|
out = Path(output_path)
|
||||||
|
out.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
counts = summary(graph)
|
||||||
|
doc_count = counts.get("nodes:document", 0)
|
||||||
|
chunk_count = counts.get("nodes:chunk", 0)
|
||||||
|
entity_count = counts.get("nodes:entity", 0)
|
||||||
|
citation_count = counts.get("nodes:citation", 0)
|
||||||
|
contains_edges = counts.get("edges:CONTAINS", 0)
|
||||||
|
mentions_edges = counts.get("edges:MENTIONS", 0)
|
||||||
|
cites_edges = counts.get("edges:CITES", 0)
|
||||||
|
cooccurs_edges = counts.get("edges:CO_OCCURS", 0)
|
||||||
|
|
||||||
|
entities = top_entities(graph, limit=20)
|
||||||
|
cooccurrences = top_cooccurrences(graph, limit=20)
|
||||||
|
citations = top_citations(graph, limit=20)
|
||||||
|
clusters = _cluster_entity_cooccurrences(graph)
|
||||||
|
|
||||||
|
# Anomalies detection
|
||||||
|
anomalies: list[Anomaly] = list(detect_anomalies(graph))
|
||||||
|
# Detect isolated citations (citations with 0 citing chunks)
|
||||||
|
existing_anomaly_nodes = {a.node for a in anomalies}
|
||||||
|
for node, data in graph.nodes(data=True):
|
||||||
|
if data.get("kind") == "citation" and node not in existing_anomaly_nodes:
|
||||||
|
citing_count = sum(
|
||||||
|
1 for _, _, edge in graph.in_edges(node, data=True) if edge.get("kind") == "CITES"
|
||||||
|
)
|
||||||
|
if citing_count == 0:
|
||||||
|
ident = data.get("identifier", node)
|
||||||
|
anomalies.append(
|
||||||
|
Anomaly(
|
||||||
|
"isolated_citation",
|
||||||
|
node,
|
||||||
|
f"citation {ident} has no referencing chunks",
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
lines: list[str] = [
|
||||||
|
"# Knowledge Graph Report",
|
||||||
|
"",
|
||||||
|
"## Summary",
|
||||||
|
f"- **Documents**: {doc_count:,}",
|
||||||
|
f"- **Chunks**: {chunk_count:,}",
|
||||||
|
f"- **Entities**: {entity_count:,}",
|
||||||
|
f"- **Citations**: {citation_count:,}",
|
||||||
|
f"- **Relationships**: {contains_edges + mentions_edges + cites_edges + cooccurs_edges:,} "
|
||||||
|
f"(CONTAINS: {contains_edges:,}, MENTIONS: {mentions_edges:,}, CITES: {cites_edges:,}, CO_OCCURS: {cooccurs_edges:,})",
|
||||||
|
"",
|
||||||
|
]
|
||||||
|
|
||||||
|
# Section: Top Entities
|
||||||
|
lines.append("## Top Entities")
|
||||||
|
lines.append("")
|
||||||
|
if entities:
|
||||||
|
lines.append("| Entity | Mentions |")
|
||||||
|
lines.append("|:-------|:---------|")
|
||||||
|
for text, mention_count in entities:
|
||||||
|
clean_text = text.replace("|", "\\|")
|
||||||
|
lines.append(f"| {clean_text} | {mention_count:,} |")
|
||||||
|
else:
|
||||||
|
lines.append("No entities found.")
|
||||||
|
lines.append("")
|
||||||
|
|
||||||
|
# Section: Top Co-occurrences
|
||||||
|
lines.append("## Top Co-occurrences")
|
||||||
|
lines.append("")
|
||||||
|
if cooccurrences:
|
||||||
|
lines.append("| Entity A | Entity B | Co-occurrence Weight |")
|
||||||
|
lines.append("|:---------|:---------|:---------------------|")
|
||||||
|
for left, right, weight in cooccurrences:
|
||||||
|
clean_left = left.replace("|", "\\|")
|
||||||
|
clean_right = right.replace("|", "\\|")
|
||||||
|
lines.append(f"| {clean_left} | {clean_right} | {weight:,} |")
|
||||||
|
else:
|
||||||
|
lines.append("No entity co-occurrences found.")
|
||||||
|
lines.append("")
|
||||||
|
|
||||||
|
# Section: Top Citations
|
||||||
|
lines.append("## Top Citations")
|
||||||
|
lines.append("")
|
||||||
|
if citations:
|
||||||
|
lines.append("| Type | Identifier | Citing Chunks |")
|
||||||
|
lines.append("|:-----|:-----------|:--------------|")
|
||||||
|
for ctype, identifier, count in citations:
|
||||||
|
lines.append(f"| {ctype.upper() or 'UNKNOWN'} | `{identifier}` | {count:,} |")
|
||||||
|
else:
|
||||||
|
lines.append("No citations found.")
|
||||||
|
lines.append("")
|
||||||
|
|
||||||
|
# Section: Concept Clusters
|
||||||
|
lines.append("## Concept Clusters")
|
||||||
|
lines.append("")
|
||||||
|
if clusters:
|
||||||
|
multi_entity_clusters = [c for c in clusters if len(c) > 1]
|
||||||
|
single_entity_clusters = [c for c in clusters if len(c) == 1]
|
||||||
|
|
||||||
|
if multi_entity_clusters:
|
||||||
|
for idx, cluster in enumerate(multi_entity_clusters, start=1):
|
||||||
|
entity_names = [e["text"] for e in cluster[:3]]
|
||||||
|
cluster_label = ", ".join(entity_names)
|
||||||
|
if len(cluster) > 3:
|
||||||
|
cluster_label += f" (+{len(cluster) - 3} more)"
|
||||||
|
lines.append(f"### Cluster {idx}: {cluster_label}")
|
||||||
|
lines.append(f"**Size**: {len(cluster)} entities")
|
||||||
|
lines.append("")
|
||||||
|
lines.append("| Entity | Mentions |")
|
||||||
|
lines.append("|:-------|:---------|")
|
||||||
|
for item in cluster:
|
||||||
|
clean_name = item["text"].replace("|", "\\|")
|
||||||
|
lines.append(f"| {clean_name} | {item['mentions']:,} |")
|
||||||
|
lines.append("")
|
||||||
|
|
||||||
|
if single_entity_clusters:
|
||||||
|
lines.append("### Isolated / Unclustered Entities")
|
||||||
|
lines.append(
|
||||||
|
f"*{len(single_entity_clusters)} entities with no co-occurrences across documents:*"
|
||||||
|
)
|
||||||
|
lines.append("")
|
||||||
|
items_preview = [
|
||||||
|
f"{c[0]['text']} ({c[0]['mentions']})" for c in single_entity_clusters[:30]
|
||||||
|
]
|
||||||
|
lines.append(", ".join(items_preview))
|
||||||
|
if len(single_entity_clusters) > 30:
|
||||||
|
lines.append(f"*(and {len(single_entity_clusters) - 30} more)*")
|
||||||
|
lines.append("")
|
||||||
|
else:
|
||||||
|
lines.append("No concept clusters detected.")
|
||||||
|
lines.append("")
|
||||||
|
|
||||||
|
# Section: Anomalies
|
||||||
|
lines.append("## Anomalies")
|
||||||
|
lines.append("")
|
||||||
|
if anomalies:
|
||||||
|
lines.append("| Anomaly Type | Node | Detail |")
|
||||||
|
lines.append("|:-------------|:-----|:-------|")
|
||||||
|
for a in anomalies:
|
||||||
|
clean_node = a.node.replace("|", "\\|")
|
||||||
|
clean_detail = a.detail.replace("|", "\\|")
|
||||||
|
lines.append(f"| `{a.kind}` | `{clean_node}` | {clean_detail} |")
|
||||||
|
else:
|
||||||
|
lines.append("No anomalies detected.")
|
||||||
|
lines.append("")
|
||||||
|
|
||||||
|
out.write_text("\n".join(lines), encoding="utf-8")
|
||||||
|
return out
|
||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -316,8 +316,8 @@ def config():
|
|||||||
|
|
||||||
@kg_app.command("build")
|
@kg_app.command("build")
|
||||||
def kg_build(
|
def kg_build(
|
||||||
output_dir: Path = typer.Option(
|
output_dir: Path | None = typer.Option(
|
||||||
..., "--output-dir", "-d", help="Pipeline output directory with chunk markdown files"
|
None, "--output-dir", "-d", help="Pipeline output directory with chunk markdown files"
|
||||||
),
|
),
|
||||||
save: Path | None = typer.Option(
|
save: Path | None = typer.Option(
|
||||||
None, "--save", "-s", help="Graph JSON path (default: <output-dir>/kg_graph.json)"
|
None, "--save", "-s", help="Graph JSON path (default: <output-dir>/kg_graph.json)"
|
||||||
@@ -325,6 +325,19 @@ def kg_build(
|
|||||||
):
|
):
|
||||||
"""Build a knowledge graph from pipeline output and save it as JSON."""
|
"""Build a knowledge graph from pipeline output and save it as JSON."""
|
||||||
kg_graph, kg_export = _load_kg()
|
kg_graph, kg_export = _load_kg()
|
||||||
|
if output_dir is None:
|
||||||
|
default_candidates = [
|
||||||
|
Path("data/ocr_output"),
|
||||||
|
Path("data"),
|
||||||
|
Path(settings.output.base_directory),
|
||||||
|
]
|
||||||
|
for candidate in default_candidates:
|
||||||
|
if candidate.is_dir():
|
||||||
|
output_dir = candidate
|
||||||
|
break
|
||||||
|
if output_dir is None:
|
||||||
|
output_dir = Path("data/ocr_output")
|
||||||
|
|
||||||
if not output_dir.is_dir():
|
if not output_dir.is_dir():
|
||||||
console.print(f"[red]Not a directory:[/red] {output_dir}")
|
console.print(f"[red]Not a directory:[/red] {output_dir}")
|
||||||
raise typer.Exit(code=2)
|
raise typer.Exit(code=2)
|
||||||
@@ -344,9 +357,9 @@ def kg_build(
|
|||||||
@kg_app.command("stats")
|
@kg_app.command("stats")
|
||||||
def kg_stats(
|
def kg_stats(
|
||||||
graph_path: Path = typer.Argument(..., help="Graph JSON written by `kg build`"),
|
graph_path: Path = typer.Argument(..., help="Graph JSON written by `kg build`"),
|
||||||
top: int = typer.Option(10, "--top", "-n", help="Rows per top-list"),
|
top: int = typer.Option(10, "--top", "-n", help="Number of top items to show"),
|
||||||
):
|
):
|
||||||
"""Summarize a graph: counts, top entities/citations, anomalies."""
|
"""Summarize graph contents and flag anomalies."""
|
||||||
kg_graph, kg_export = _load_kg()
|
kg_graph, kg_export = _load_kg()
|
||||||
if not graph_path.is_file():
|
if not graph_path.is_file():
|
||||||
console.print(f"[red]Graph file not found:[/red] {graph_path}")
|
console.print(f"[red]Graph file not found:[/red] {graph_path}")
|
||||||
@@ -354,59 +367,66 @@ def kg_stats(
|
|||||||
graph = kg_export.load_json(graph_path)
|
graph = kg_export.load_json(graph_path)
|
||||||
|
|
||||||
counts = kg_graph.summary(graph)
|
counts = kg_graph.summary(graph)
|
||||||
table = Table(title="Graph summary")
|
summary_table = Table(title=f"Summary ({graph_path})")
|
||||||
table.add_column("Kind", style="cyan")
|
summary_table.add_column("Kind", style="cyan")
|
||||||
table.add_column("Count", style="green")
|
summary_table.add_column("Count", style="green")
|
||||||
for kind, count in sorted(counts.items()):
|
for kind, count in sorted(counts.items()):
|
||||||
table.add_row(kind, str(count))
|
summary_table.add_row(kind, str(count))
|
||||||
console.print(table)
|
console.print(summary_table)
|
||||||
|
|
||||||
entities = kg_graph.top_entities(graph, limit=top)
|
entities = kg_graph.top_entities(graph, limit=top)
|
||||||
if entities:
|
if entities:
|
||||||
table = Table(title=f"Top {top} entities")
|
ent_table = Table(title=f"Top {len(entities)} Entities")
|
||||||
table.add_column("Entity", style="cyan")
|
ent_table.add_column("Entity", style="cyan")
|
||||||
table.add_column("Mentions", style="green")
|
ent_table.add_column("Mentions", style="green")
|
||||||
for text, mentions in entities:
|
for name, count in entities:
|
||||||
table.add_row(text, str(mentions))
|
ent_table.add_row(name, str(count))
|
||||||
console.print(table)
|
console.print(ent_table)
|
||||||
|
|
||||||
citations = kg_graph.top_citations(graph, limit=top)
|
citations = kg_graph.top_citations(graph, limit=top)
|
||||||
if citations:
|
if citations:
|
||||||
table = Table(title=f"Top {top} citations")
|
cit_table = Table(title=f"Top {len(citations)} Citations")
|
||||||
table.add_column("Type", style="cyan")
|
cit_table.add_column("Type", style="cyan")
|
||||||
table.add_column("Identifier")
|
cit_table.add_column("Identifier", style="magenta")
|
||||||
table.add_column("Citing chunks", style="green")
|
cit_table.add_column("Citing chunks", style="green")
|
||||||
for ctype, identifier, citing in citations:
|
for ctype, identifier, count in citations:
|
||||||
table.add_row(ctype, identifier, str(citing))
|
cit_table.add_row(ctype, identifier, str(count))
|
||||||
console.print(table)
|
console.print(cit_table)
|
||||||
|
|
||||||
anomalies = kg_graph.detect_anomalies(graph)
|
anomalies = kg_graph.detect_anomalies(graph)
|
||||||
if anomalies:
|
if anomalies:
|
||||||
console.print(f"\n[yellow]{len(anomalies)} anomalies:[/yellow]")
|
anom_table = Table(title=f"Anomalies ({len(anomalies)})")
|
||||||
for anomaly in anomalies[:top]:
|
anom_table.add_column("Kind", style="yellow")
|
||||||
console.print(f" [dim]{anomaly.kind}[/dim] {anomaly.node}: {anomaly.detail}")
|
anom_table.add_column("Node", style="cyan")
|
||||||
|
anom_table.add_column("Detail")
|
||||||
|
for anomaly in anomalies:
|
||||||
|
anom_table.add_row(anomaly.kind, anomaly.node, anomaly.detail)
|
||||||
|
console.print(anom_table)
|
||||||
|
else:
|
||||||
|
console.print("[green]No anomalies detected.[/green]")
|
||||||
|
|
||||||
|
|
||||||
@kg_app.command("query")
|
@kg_app.command("query")
|
||||||
def kg_query(
|
def kg_query(
|
||||||
graph_path: Path = typer.Argument(..., help="Graph JSON written by `kg build`"),
|
graph_path: Path = typer.Argument(..., help="Graph JSON written by `kg build`"),
|
||||||
entity: str | None = typer.Option(None, "--entity", "-e", help="Entity to look up"),
|
entity: str | None = typer.Option(None, "--entity", "-e", help="Entity text to query"),
|
||||||
citation: str | None = typer.Option(
|
citation: str | None = typer.Option(
|
||||||
None, "--citation", "-c", help="Citation identifier (e.g. 10.1038/nature12345)"
|
None, "--citation", "-c", help="Citation identifier or DOI/PMID"
|
||||||
),
|
),
|
||||||
expand: bool = typer.Option(
|
expand: bool = typer.Option(
|
||||||
False, "--expand", "-x", help="Include chunks from co-occurring entities"
|
False, "--expand", "-x", help="Include 1-hop neighbor chunks (co-occurring entities)"
|
||||||
),
|
),
|
||||||
limit: int = typer.Option(5, "--limit", "-n", help="Max chunks to show"),
|
limit: int = typer.Option(10, "--limit", "-l", help="Max chunks to display"),
|
||||||
):
|
):
|
||||||
"""Retrieve chunks by entity or citation; --expand adds neighbor chunks."""
|
"""Retrieve text chunks grounded in the knowledge graph."""
|
||||||
kg_graph, kg_export = _load_kg()
|
kg_graph, kg_export = _load_kg()
|
||||||
if not graph_path.is_file():
|
if not graph_path.is_file():
|
||||||
console.print(f"[red]Graph file not found:[/red] {graph_path}")
|
console.print(f"[red]Graph file not found:[/red] {graph_path}")
|
||||||
raise typer.Exit(code=2)
|
raise typer.Exit(code=2)
|
||||||
if not entity and not citation:
|
if not entity and not citation:
|
||||||
console.print("[red]Give --entity or --citation.[/red]")
|
console.print("[red]Provide at least one of --entity or --citation.[/red]")
|
||||||
raise typer.Exit(code=2)
|
raise typer.Exit(code=2)
|
||||||
|
|
||||||
graph = kg_export.load_json(graph_path)
|
graph = kg_export.load_json(graph_path)
|
||||||
|
|
||||||
if entity:
|
if entity:
|
||||||
@@ -416,43 +436,57 @@ def kg_query(
|
|||||||
else kg_graph.chunks_for_entity(graph, entity)
|
else kg_graph.chunks_for_entity(graph, entity)
|
||||||
)
|
)
|
||||||
related = kg_graph.related_entities(graph, entity)
|
related = kg_graph.related_entities(graph, entity)
|
||||||
|
console.print(f"[bold cyan]Entity:[/bold cyan] {entity} ({len(chunks)} chunks)")
|
||||||
if related:
|
if related:
|
||||||
console.print(
|
rendered = ", ".join(f"{name} ({weight})" for name, weight in related[:5])
|
||||||
"[dim]Related entities: "
|
console.print(f"[dim]Related entities:[/dim] {rendered}")
|
||||||
+ ", ".join(f"{text} ({weight})" for text, weight in related[:5])
|
for chunk in chunks[:limit]:
|
||||||
+ "[/dim]"
|
source = chunk.get("source_path", "unknown")
|
||||||
)
|
via = f" (via {chunk['via_entity']})" if "via_entity" in chunk else ""
|
||||||
else:
|
console.print(f"[bold]{source}[/bold]{via}:")
|
||||||
|
excerpt = chunk.get("text", "")[:200].replace("\n", " ")
|
||||||
|
console.print(f" {excerpt}")
|
||||||
|
|
||||||
|
if citation:
|
||||||
chunks = kg_graph.chunks_for_citation(graph, citation or "")
|
chunks = kg_graph.chunks_for_citation(graph, citation or "")
|
||||||
|
console.print(f"[bold cyan]Citation:[/bold cyan] {citation} ({len(chunks)} chunks)")
|
||||||
if not chunks:
|
for chunk in chunks[:limit]:
|
||||||
console.print("[yellow]No matching chunks.[/yellow]")
|
source = chunk.get("source_path", "unknown")
|
||||||
return
|
console.print(f"[bold]{source}[/bold]:")
|
||||||
|
excerpt = chunk.get("text", "")[:200].replace("\n", " ")
|
||||||
for payload in chunks[:limit]:
|
console.print(f" {excerpt}")
|
||||||
via = f" via {payload['via_entity']}" if payload.get("via_entity") else ""
|
|
||||||
console.print(
|
|
||||||
f"\n[bold]{Path(str(payload['source_path'])).name}[/bold] "
|
|
||||||
f"chunk {payload['chunk_index']}{via} "
|
|
||||||
f"[dim]({payload['ocr_engine']} {payload['ocr_confidence_mean']:.2f})[/dim]"
|
|
||||||
)
|
|
||||||
excerpt = " ".join(str(payload["text"]).split())[:300]
|
|
||||||
console.print(f" {excerpt}")
|
|
||||||
|
|
||||||
|
|
||||||
@kg_app.command("export")
|
@kg_app.command("export")
|
||||||
def kg_export_cmd(
|
def kg_export_cmd(
|
||||||
graph_path: Path = typer.Argument(..., help="Graph JSON written by `kg build`"),
|
graph_path: Path | None = typer.Argument(
|
||||||
fmt: str = typer.Option("graphml", "--format", "-f", help="graphml | neo4j"),
|
None, help="Graph JSON written by `kg build` (default: <output-dir>/kg_graph.json)"
|
||||||
out: Path | None = typer.Option(None, "--out", "-o", help="Output path for graphml"),
|
),
|
||||||
|
fmt: str = typer.Option("graphml", "--format", "-f", help="graphml | neo4j | markdown"),
|
||||||
|
out: Path | None = typer.Option(
|
||||||
|
None, "--out", "--output", "-o", help="Output path (default: KG_REPORT.md for markdown)"
|
||||||
|
),
|
||||||
uri: str | None = typer.Option(None, "--uri", help="Neo4j bolt URI (or NEO4J_URI)"),
|
uri: str | None = typer.Option(None, "--uri", help="Neo4j bolt URI (or NEO4J_URI)"),
|
||||||
user: str | None = typer.Option(None, "--user", help="Neo4j user (or NEO4J_USER)"),
|
user: str | None = typer.Option(None, "--user", help="Neo4j user (or NEO4J_USER)"),
|
||||||
password: str | None = typer.Option(
|
password: str | None = typer.Option(
|
||||||
None, "--password", help="Neo4j password (or NEO4J_PASSWORD)"
|
None, "--password", help="Neo4j password (or NEO4J_PASSWORD)"
|
||||||
),
|
),
|
||||||
):
|
):
|
||||||
"""Export a graph JSON to GraphML or push it into Neo4j."""
|
"""Export a graph JSON to GraphML, Markdown report, or push it into Neo4j."""
|
||||||
_, kg_export = _load_kg()
|
_, kg_export = _load_kg()
|
||||||
|
if graph_path is None:
|
||||||
|
candidates = [
|
||||||
|
Path("data/ocr_output/kg_graph.json"),
|
||||||
|
Path("kg_graph.json"),
|
||||||
|
Path(settings.output.base_directory) / "kg_graph.json",
|
||||||
|
]
|
||||||
|
for c in candidates:
|
||||||
|
if c.is_file():
|
||||||
|
graph_path = c
|
||||||
|
break
|
||||||
|
if graph_path is None:
|
||||||
|
graph_path = Path("data/ocr_output/kg_graph.json")
|
||||||
|
|
||||||
if not graph_path.is_file():
|
if not graph_path.is_file():
|
||||||
console.print(f"[red]Graph file not found:[/red] {graph_path}")
|
console.print(f"[red]Graph file not found:[/red] {graph_path}")
|
||||||
raise typer.Exit(code=2)
|
raise typer.Exit(code=2)
|
||||||
@@ -462,6 +496,10 @@ def kg_export_cmd(
|
|||||||
out = out or graph_path.with_suffix(".graphml")
|
out = out or graph_path.with_suffix(".graphml")
|
||||||
kg_export.export_graphml(graph, out)
|
kg_export.export_graphml(graph, out)
|
||||||
console.print(f"[green]GraphML written:[/green] {out}")
|
console.print(f"[green]GraphML written:[/green] {out}")
|
||||||
|
elif fmt == "markdown":
|
||||||
|
out = out or Path("KG_REPORT.md")
|
||||||
|
kg_export.export_markdown(graph, out)
|
||||||
|
console.print(f"[green]Markdown report written:[/green] {out}")
|
||||||
elif fmt == "neo4j":
|
elif fmt == "neo4j":
|
||||||
try:
|
try:
|
||||||
with kg_export.Neo4jExporter(uri=uri, user=user, password=password) as exporter:
|
with kg_export.Neo4jExporter(uri=uri, user=user, password=password) as exporter:
|
||||||
@@ -473,7 +511,7 @@ def kg_export_cmd(
|
|||||||
f"[green]Pushed to Neo4j:[/green] {pushed['nodes']} nodes, {pushed['edges']} edges"
|
f"[green]Pushed to Neo4j:[/green] {pushed['nodes']} nodes, {pushed['edges']} edges"
|
||||||
)
|
)
|
||||||
else:
|
else:
|
||||||
console.print(f"[red]Unknown format:[/red] {fmt} (choose graphml or neo4j)")
|
console.print(f"[red]Unknown format:[/red] {fmt} (choose graphml, neo4j, or markdown)")
|
||||||
raise typer.Exit(code=2)
|
raise typer.Exit(code=2)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ import pytest
|
|||||||
|
|
||||||
nx = pytest.importorskip("networkx", reason="kg extra not installed")
|
nx = pytest.importorskip("networkx", reason="kg extra not installed")
|
||||||
|
|
||||||
from kg_ocr.export import export_graphml, export_json, load_json # noqa: E402
|
from kg_ocr.export import export_graphml, export_json, export_markdown, load_json # noqa: E402
|
||||||
from kg_ocr.graph import ( # noqa: E402
|
from kg_ocr.graph import ( # noqa: E402
|
||||||
build_from_directory,
|
build_from_directory,
|
||||||
build_graph,
|
build_graph,
|
||||||
@@ -202,6 +202,50 @@ def test_kg_cli_build_and_stats(chunk_dir: Path, tmp_path: Path) -> None:
|
|||||||
assert result.exit_code == 0, result.output
|
assert result.exit_code == 0, result.output
|
||||||
assert out_graphml.is_file()
|
assert out_graphml.is_file()
|
||||||
|
|
||||||
|
out_md = tmp_path / "g_report.md"
|
||||||
|
result = runner.invoke(app, ["kg", "export", str(save), "-f", "markdown", "-o", str(out_md)])
|
||||||
|
assert result.exit_code == 0, result.output
|
||||||
|
assert out_md.is_file()
|
||||||
|
content = out_md.read_text(encoding="utf-8")
|
||||||
|
assert "## Top Entities" in content
|
||||||
|
assert "## Top Co-occurrences" in content
|
||||||
|
assert "## Top Citations" in content
|
||||||
|
assert "## Concept Clusters" in content
|
||||||
|
assert "## Anomalies" in content
|
||||||
|
assert "BRCA1" in content
|
||||||
|
assert "10.1038/nature12345" in content
|
||||||
|
|
||||||
|
|
||||||
|
def test_export_markdown(chunk_dir: Path, tmp_path: Path) -> None:
|
||||||
|
graph = build_from_directory(chunk_dir)
|
||||||
|
save_json = tmp_path / "graph.json"
|
||||||
|
export_json(graph, save_json)
|
||||||
|
|
||||||
|
out_md = tmp_path / "report.md"
|
||||||
|
# Test with string paths as per signature requirement
|
||||||
|
written = export_markdown(str(save_json), str(out_md))
|
||||||
|
assert written == out_md
|
||||||
|
assert out_md.is_file()
|
||||||
|
|
||||||
|
text = out_md.read_text(encoding="utf-8")
|
||||||
|
assert "# Knowledge Graph Report" in text
|
||||||
|
assert "## Summary" in text
|
||||||
|
assert "## Top Entities" in text
|
||||||
|
assert "## Top Co-occurrences" in text
|
||||||
|
assert "## Top Citations" in text
|
||||||
|
assert "## Concept Clusters" in text
|
||||||
|
assert "## Anomalies" in text
|
||||||
|
assert "BRCA1" in text
|
||||||
|
assert "PARP" in text
|
||||||
|
assert "10.1038/nature12345" in text
|
||||||
|
assert "low_confidence" in text
|
||||||
|
|
||||||
|
# Also test passing in-memory graph directly
|
||||||
|
out_direct = tmp_path / "report_direct.md"
|
||||||
|
export_markdown(graph, out_direct)
|
||||||
|
assert out_direct.is_file()
|
||||||
|
assert "## Top Entities" in out_direct.read_text(encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
def test_chunks_for_entity(chunk_dir: Path) -> None:
|
def test_chunks_for_entity(chunk_dir: Path) -> None:
|
||||||
from kg_ocr.graph import chunks_for_entity
|
from kg_ocr.graph import chunks_for_entity
|
||||||
|
|||||||
Reference in New Issue
Block a user