"""Export a knowledge graph to a Markdown report.""" from __future__ import annotations import json from pathlib import Path from typing import Any from kg_ocr.graph.analyzer import ( Anomaly, detect_anomalies, summary, top_citations, top_cooccurrences, top_entities, ) def _load_graph(graph_input: Any) -> Any: """Load a NetworkX graph from a file path or return it directly.""" import networkx as nx if isinstance(graph_input, (str, Path)): path = Path(graph_input) data = json.loads(path.read_text(encoding="utf-8")) try: return nx.node_link_graph(data, edges="edges") except TypeError: return nx.node_link_graph(data) return graph_input def _cluster_entity_cooccurrences(graph: Any) -> list[list[dict[str, Any]]]: """Find concept clusters over the entity co-occurrence subgraph.""" import networkx as nx cooccur_graph = nx.Graph() for node, data in graph.nodes(data=True): if data.get("kind") == "entity": cooccur_graph.add_node( node, text=data.get("text", node.replace("entity:", "")), mentions=int(data.get("mentions", 0)), ) for u, v, data in graph.edges(data=True): if data.get("kind") == "CO_OCCURS": cooccur_graph.add_edge(u, v, weight=data.get("weight", 1)) if cooccur_graph.number_of_nodes() == 0: return [] # Community detection communities: list[set[str]] = [] if cooccur_graph.number_of_edges() > 0: try: import networkx.algorithms.community as nx_comm # type: ignore[import-untyped] communities = list( nx_comm.louvain_communities(cooccur_graph, weight="weight", seed=42) ) except Exception: try: import networkx.algorithms.community as nx_comm # type: ignore[import-untyped] communities = list( nx_comm.greedy_modularity_communities(cooccur_graph, weight="weight") ) except Exception: try: import community as community_louvain # type: ignore[import-not-found] partition = community_louvain.best_partition( cooccur_graph, weight="weight", random_state=42 ) comm_dict: dict[int, set[str]] = {} for n, cid in partition.items(): comm_dict.setdefault(cid, set()).add(n) communities = list(comm_dict.values()) except Exception: communities = list(nx.connected_components(cooccur_graph)) else: # No edges: each entity is an isolated node communities = [{n} for n in cooccur_graph.nodes()] # Format and sort clusters clusters: list[list[dict[str, Any]]] = [] for comm in communities: cluster_nodes = [] for node in comm: cluster_nodes.append( { "id": node, "text": cooccur_graph.nodes[node].get("text", node), "mentions": cooccur_graph.nodes[node].get("mentions", 0), } ) # Sort entities inside cluster by mentions descending cluster_nodes.sort(key=lambda item: item["mentions"], reverse=True) clusters.append(cluster_nodes) # Sort clusters by size (number of entities) descending, then top entity mentions clusters.sort( key=lambda c: (len(c), c[0]["mentions"] if c else 0), reverse=True, ) return clusters def export_markdown(graph_path: str | Path, output_path: str | Path) -> Path: """Render a knowledge graph into a Markdown summary report. Parameters ---------- graph_path : str | Path | nx.Graph Path to `kg_graph.json` or an in-memory NetworkX graph. output_path : str | Path Target path for the exported Markdown report. Returns ------- Path The output Path written to. """ graph = _load_graph(graph_path) out = Path(output_path) out.parent.mkdir(parents=True, exist_ok=True) counts = summary(graph) doc_count = counts.get("nodes:document", 0) chunk_count = counts.get("nodes:chunk", 0) entity_count = counts.get("nodes:entity", 0) citation_count = counts.get("nodes:citation", 0) contains_edges = counts.get("edges:CONTAINS", 0) mentions_edges = counts.get("edges:MENTIONS", 0) cites_edges = counts.get("edges:CITES", 0) cooccurs_edges = counts.get("edges:CO_OCCURS", 0) entities = top_entities(graph, limit=20) cooccurrences = top_cooccurrences(graph, limit=20) citations = top_citations(graph, limit=20) clusters = _cluster_entity_cooccurrences(graph) # Anomalies detection anomalies: list[Anomaly] = list(detect_anomalies(graph)) # Detect isolated citations (citations with 0 citing chunks) existing_anomaly_nodes = {a.node for a in anomalies} for node, data in graph.nodes(data=True): if data.get("kind") == "citation" and node not in existing_anomaly_nodes: citing_count = sum( 1 for _, _, edge in graph.in_edges(node, data=True) if edge.get("kind") == "CITES" ) if citing_count == 0: ident = data.get("identifier", node) anomalies.append( Anomaly( "isolated_citation", node, f"citation {ident} has no referencing chunks", ) ) lines: list[str] = [ "# Knowledge Graph Report", "", "## Summary", f"- **Documents**: {doc_count:,}", f"- **Chunks**: {chunk_count:,}", f"- **Entities**: {entity_count:,}", f"- **Citations**: {citation_count:,}", f"- **Relationships**: {contains_edges + mentions_edges + cites_edges + cooccurs_edges:,} " f"(CONTAINS: {contains_edges:,}, MENTIONS: {mentions_edges:,}, CITES: {cites_edges:,}, CO_OCCURS: {cooccurs_edges:,})", "", ] # Section: Top Entities lines.append("## Top Entities") lines.append("") if entities: lines.append("| Entity | Mentions |") lines.append("|:-------|:---------|") for text, mention_count in entities: clean_text = text.replace("|", "\\|") lines.append(f"| {clean_text} | {mention_count:,} |") else: lines.append("No entities found.") lines.append("") # Section: Top Co-occurrences lines.append("## Top Co-occurrences") lines.append("") if cooccurrences: lines.append("| Entity A | Entity B | Co-occurrence Weight |") lines.append("|:---------|:---------|:---------------------|") for left, right, weight in cooccurrences: clean_left = left.replace("|", "\\|") clean_right = right.replace("|", "\\|") lines.append(f"| {clean_left} | {clean_right} | {weight:,} |") else: lines.append("No entity co-occurrences found.") lines.append("") # Section: Top Citations lines.append("## Top Citations") lines.append("") if citations: lines.append("| Type | Identifier | Citing Chunks |") lines.append("|:-----|:-----------|:--------------|") for ctype, identifier, count in citations: lines.append(f"| {ctype.upper() or 'UNKNOWN'} | `{identifier}` | {count:,} |") else: lines.append("No citations found.") lines.append("") # Section: Concept Clusters lines.append("## Concept Clusters") lines.append("") if clusters: multi_entity_clusters = [c for c in clusters if len(c) > 1] single_entity_clusters = [c for c in clusters if len(c) == 1] if multi_entity_clusters: for idx, cluster in enumerate(multi_entity_clusters, start=1): entity_names = [e["text"] for e in cluster[:3]] cluster_label = ", ".join(entity_names) if len(cluster) > 3: cluster_label += f" (+{len(cluster) - 3} more)" lines.append(f"### Cluster {idx}: {cluster_label}") lines.append(f"**Size**: {len(cluster)} entities") lines.append("") lines.append("| Entity | Mentions |") lines.append("|:-------|:---------|") for item in cluster: clean_name = item["text"].replace("|", "\\|") lines.append(f"| {clean_name} | {item['mentions']:,} |") lines.append("") if single_entity_clusters: lines.append("### Isolated / Unclustered Entities") lines.append( f"*{len(single_entity_clusters)} entities with no co-occurrences across documents:*" ) lines.append("") items_preview = [ f"{c[0]['text']} ({c[0]['mentions']})" for c in single_entity_clusters[:30] ] lines.append(", ".join(items_preview)) if len(single_entity_clusters) > 30: lines.append(f"*(and {len(single_entity_clusters) - 30} more)*") lines.append("") else: lines.append("No concept clusters detected.") lines.append("") # Section: Anomalies lines.append("## Anomalies") lines.append("") if anomalies: lines.append("| Anomaly Type | Node | Detail |") lines.append("|:-------------|:-----|:-------|") for a in anomalies: clean_node = a.node.replace("|", "\\|") clean_detail = a.detail.replace("|", "\\|") lines.append(f"| `{a.kind}` | `{clean_node}` | {clean_detail} |") else: lines.append("No anomalies detected.") lines.append("") out.write_text("\n".join(lines), encoding="utf-8") return out