Some checks failed
tests / core (macos-latest, py3.11) (push) Has been cancelled
tests / core (macos-latest, py3.12) (push) Has been cancelled
tests / core (macos-latest, py3.13) (push) Has been cancelled
tests / core (ubuntu-latest, py3.11) (push) Has been cancelled
tests / core (ubuntu-latest, py3.12) (push) Has been cancelled
tests / core (ubuntu-latest, py3.13) (push) Has been cancelled
tests / doctor CLI smoke test (push) Has been cancelled
- Add kg_ocr/export/markdown_exporter.py with export_markdown() - Cluster entity co-occurrences using Louvain community detection - Wire -f markdown into `ocr-pipeline kg export` CLI command with KG_REPORT.md default - Add unit and CLI tests in tests/test_graph.py - Document markdown export in README.md - Untrack pycache binaries and ignore graphify-out and KG_REPORT.md
271 lines
9.8 KiB
Python
271 lines
9.8 KiB
Python
"""Export a knowledge graph to a Markdown report."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
from kg_ocr.graph.analyzer import (
|
|
Anomaly,
|
|
detect_anomalies,
|
|
summary,
|
|
top_citations,
|
|
top_cooccurrences,
|
|
top_entities,
|
|
)
|
|
|
|
|
|
def _load_graph(graph_input: Any) -> Any:
|
|
"""Load a NetworkX graph from a file path or return it directly."""
|
|
import networkx as nx
|
|
|
|
if isinstance(graph_input, (str, Path)):
|
|
path = Path(graph_input)
|
|
data = json.loads(path.read_text(encoding="utf-8"))
|
|
try:
|
|
return nx.node_link_graph(data, edges="edges")
|
|
except TypeError:
|
|
return nx.node_link_graph(data)
|
|
return graph_input
|
|
|
|
|
|
def _cluster_entity_cooccurrences(graph: Any) -> list[list[dict[str, Any]]]:
|
|
"""Find concept clusters over the entity co-occurrence subgraph."""
|
|
import networkx as nx
|
|
|
|
cooccur_graph = nx.Graph()
|
|
for node, data in graph.nodes(data=True):
|
|
if data.get("kind") == "entity":
|
|
cooccur_graph.add_node(
|
|
node,
|
|
text=data.get("text", node.replace("entity:", "")),
|
|
mentions=int(data.get("mentions", 0)),
|
|
)
|
|
|
|
for u, v, data in graph.edges(data=True):
|
|
if data.get("kind") == "CO_OCCURS":
|
|
cooccur_graph.add_edge(u, v, weight=data.get("weight", 1))
|
|
|
|
if cooccur_graph.number_of_nodes() == 0:
|
|
return []
|
|
|
|
# Community detection
|
|
communities: list[set[str]] = []
|
|
if cooccur_graph.number_of_edges() > 0:
|
|
try:
|
|
import networkx.algorithms.community as nx_comm # type: ignore[import-untyped]
|
|
|
|
communities = list(
|
|
nx_comm.louvain_communities(cooccur_graph, weight="weight", seed=42)
|
|
)
|
|
except Exception:
|
|
try:
|
|
import networkx.algorithms.community as nx_comm # type: ignore[import-untyped]
|
|
|
|
communities = list(
|
|
nx_comm.greedy_modularity_communities(cooccur_graph, weight="weight")
|
|
)
|
|
except Exception:
|
|
try:
|
|
import community as community_louvain # type: ignore[import-not-found]
|
|
|
|
partition = community_louvain.best_partition(
|
|
cooccur_graph, weight="weight", random_state=42
|
|
)
|
|
comm_dict: dict[int, set[str]] = {}
|
|
for n, cid in partition.items():
|
|
comm_dict.setdefault(cid, set()).add(n)
|
|
communities = list(comm_dict.values())
|
|
except Exception:
|
|
communities = list(nx.connected_components(cooccur_graph))
|
|
else:
|
|
# No edges: each entity is an isolated node
|
|
communities = [{n} for n in cooccur_graph.nodes()]
|
|
|
|
# Format and sort clusters
|
|
clusters: list[list[dict[str, Any]]] = []
|
|
for comm in communities:
|
|
cluster_nodes = []
|
|
for node in comm:
|
|
cluster_nodes.append(
|
|
{
|
|
"id": node,
|
|
"text": cooccur_graph.nodes[node].get("text", node),
|
|
"mentions": cooccur_graph.nodes[node].get("mentions", 0),
|
|
}
|
|
)
|
|
# Sort entities inside cluster by mentions descending
|
|
cluster_nodes.sort(key=lambda item: item["mentions"], reverse=True)
|
|
clusters.append(cluster_nodes)
|
|
|
|
# Sort clusters by size (number of entities) descending, then top entity mentions
|
|
clusters.sort(
|
|
key=lambda c: (len(c), c[0]["mentions"] if c else 0),
|
|
reverse=True,
|
|
)
|
|
return clusters
|
|
|
|
|
|
def export_markdown(graph_path: str | Path, output_path: str | Path) -> Path:
|
|
"""Render a knowledge graph into a Markdown summary report.
|
|
|
|
Parameters
|
|
----------
|
|
graph_path : str | Path | nx.Graph
|
|
Path to `kg_graph.json` or an in-memory NetworkX graph.
|
|
output_path : str | Path
|
|
Target path for the exported Markdown report.
|
|
|
|
Returns
|
|
-------
|
|
Path
|
|
The output Path written to.
|
|
"""
|
|
graph = _load_graph(graph_path)
|
|
out = Path(output_path)
|
|
out.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
counts = summary(graph)
|
|
doc_count = counts.get("nodes:document", 0)
|
|
chunk_count = counts.get("nodes:chunk", 0)
|
|
entity_count = counts.get("nodes:entity", 0)
|
|
citation_count = counts.get("nodes:citation", 0)
|
|
contains_edges = counts.get("edges:CONTAINS", 0)
|
|
mentions_edges = counts.get("edges:MENTIONS", 0)
|
|
cites_edges = counts.get("edges:CITES", 0)
|
|
cooccurs_edges = counts.get("edges:CO_OCCURS", 0)
|
|
|
|
entities = top_entities(graph, limit=20)
|
|
cooccurrences = top_cooccurrences(graph, limit=20)
|
|
citations = top_citations(graph, limit=20)
|
|
clusters = _cluster_entity_cooccurrences(graph)
|
|
|
|
# Anomalies detection
|
|
anomalies: list[Anomaly] = list(detect_anomalies(graph))
|
|
# Detect isolated citations (citations with 0 citing chunks)
|
|
existing_anomaly_nodes = {a.node for a in anomalies}
|
|
for node, data in graph.nodes(data=True):
|
|
if data.get("kind") == "citation" and node not in existing_anomaly_nodes:
|
|
citing_count = sum(
|
|
1 for _, _, edge in graph.in_edges(node, data=True) if edge.get("kind") == "CITES"
|
|
)
|
|
if citing_count == 0:
|
|
ident = data.get("identifier", node)
|
|
anomalies.append(
|
|
Anomaly(
|
|
"isolated_citation",
|
|
node,
|
|
f"citation {ident} has no referencing chunks",
|
|
)
|
|
)
|
|
|
|
lines: list[str] = [
|
|
"# Knowledge Graph Report",
|
|
"",
|
|
"## Summary",
|
|
f"- **Documents**: {doc_count:,}",
|
|
f"- **Chunks**: {chunk_count:,}",
|
|
f"- **Entities**: {entity_count:,}",
|
|
f"- **Citations**: {citation_count:,}",
|
|
f"- **Relationships**: {contains_edges + mentions_edges + cites_edges + cooccurs_edges:,} "
|
|
f"(CONTAINS: {contains_edges:,}, MENTIONS: {mentions_edges:,}, CITES: {cites_edges:,}, CO_OCCURS: {cooccurs_edges:,})",
|
|
"",
|
|
]
|
|
|
|
# Section: Top Entities
|
|
lines.append("## Top Entities")
|
|
lines.append("")
|
|
if entities:
|
|
lines.append("| Entity | Mentions |")
|
|
lines.append("|:-------|:---------|")
|
|
for text, mention_count in entities:
|
|
clean_text = text.replace("|", "\\|")
|
|
lines.append(f"| {clean_text} | {mention_count:,} |")
|
|
else:
|
|
lines.append("No entities found.")
|
|
lines.append("")
|
|
|
|
# Section: Top Co-occurrences
|
|
lines.append("## Top Co-occurrences")
|
|
lines.append("")
|
|
if cooccurrences:
|
|
lines.append("| Entity A | Entity B | Co-occurrence Weight |")
|
|
lines.append("|:---------|:---------|:---------------------|")
|
|
for left, right, weight in cooccurrences:
|
|
clean_left = left.replace("|", "\\|")
|
|
clean_right = right.replace("|", "\\|")
|
|
lines.append(f"| {clean_left} | {clean_right} | {weight:,} |")
|
|
else:
|
|
lines.append("No entity co-occurrences found.")
|
|
lines.append("")
|
|
|
|
# Section: Top Citations
|
|
lines.append("## Top Citations")
|
|
lines.append("")
|
|
if citations:
|
|
lines.append("| Type | Identifier | Citing Chunks |")
|
|
lines.append("|:-----|:-----------|:--------------|")
|
|
for ctype, identifier, count in citations:
|
|
lines.append(f"| {ctype.upper() or 'UNKNOWN'} | `{identifier}` | {count:,} |")
|
|
else:
|
|
lines.append("No citations found.")
|
|
lines.append("")
|
|
|
|
# Section: Concept Clusters
|
|
lines.append("## Concept Clusters")
|
|
lines.append("")
|
|
if clusters:
|
|
multi_entity_clusters = [c for c in clusters if len(c) > 1]
|
|
single_entity_clusters = [c for c in clusters if len(c) == 1]
|
|
|
|
if multi_entity_clusters:
|
|
for idx, cluster in enumerate(multi_entity_clusters, start=1):
|
|
entity_names = [e["text"] for e in cluster[:3]]
|
|
cluster_label = ", ".join(entity_names)
|
|
if len(cluster) > 3:
|
|
cluster_label += f" (+{len(cluster) - 3} more)"
|
|
lines.append(f"### Cluster {idx}: {cluster_label}")
|
|
lines.append(f"**Size**: {len(cluster)} entities")
|
|
lines.append("")
|
|
lines.append("| Entity | Mentions |")
|
|
lines.append("|:-------|:---------|")
|
|
for item in cluster:
|
|
clean_name = item["text"].replace("|", "\\|")
|
|
lines.append(f"| {clean_name} | {item['mentions']:,} |")
|
|
lines.append("")
|
|
|
|
if single_entity_clusters:
|
|
lines.append("### Isolated / Unclustered Entities")
|
|
lines.append(
|
|
f"*{len(single_entity_clusters)} entities with no co-occurrences across documents:*"
|
|
)
|
|
lines.append("")
|
|
items_preview = [
|
|
f"{c[0]['text']} ({c[0]['mentions']})" for c in single_entity_clusters[:30]
|
|
]
|
|
lines.append(", ".join(items_preview))
|
|
if len(single_entity_clusters) > 30:
|
|
lines.append(f"*(and {len(single_entity_clusters) - 30} more)*")
|
|
lines.append("")
|
|
else:
|
|
lines.append("No concept clusters detected.")
|
|
lines.append("")
|
|
|
|
# Section: Anomalies
|
|
lines.append("## Anomalies")
|
|
lines.append("")
|
|
if anomalies:
|
|
lines.append("| Anomaly Type | Node | Detail |")
|
|
lines.append("|:-------------|:-----|:-------|")
|
|
for a in anomalies:
|
|
clean_node = a.node.replace("|", "\\|")
|
|
clean_detail = a.detail.replace("|", "\\|")
|
|
lines.append(f"| `{a.kind}` | `{clean_node}` | {clean_detail} |")
|
|
else:
|
|
lines.append("No anomalies detected.")
|
|
lines.append("")
|
|
|
|
out.write_text("\n".join(lines), encoding="utf-8")
|
|
return out
|