diff --git a/.gitignore b/.gitignore index 26987af..8ba922b 100644 --- a/.gitignore +++ b/.gitignore @@ -80,4 +80,6 @@ Thumbs.db # Project specific data/ notebooks/ -.env \ No newline at end of file +.env +graphify-out/ +KG_REPORT.md diff --git a/README.md b/README.md index da44084..82ada60 100644 --- a/README.md +++ b/README.md @@ -79,6 +79,7 @@ uv run ocr-pipeline kg query data/ocr_output/kg_graph.json --entity BRCA1 --expa uv run ocr-pipeline kg query data/ocr_output/kg_graph.json --citation 10.1038/nature12345 # export +uv run ocr-pipeline kg export data/ocr_output/kg_graph.json -f markdown uv run ocr-pipeline kg export data/ocr_output/kg_graph.json -f graphml uv run ocr-pipeline kg export data/ocr_output/kg_graph.json -f neo4j ``` @@ -239,4 +240,3 @@ Gene/protein names go through scispaCy's `en_core_sci_lg`. Chemical formulas and units (µM, ng/mL, kb/Mb/Gb, °C, ×g) get normalized, scientific notation gets cleaned up (`1.5×10⁻³` → `1.5×10^-3`), and gel/blot figures get their captions pulled out separately. - diff --git a/kg_ocr/__init__.py b/kg_ocr/__init__.py index 28d03dc..f07e811 100644 --- a/kg_ocr/__init__.py +++ b/kg_ocr/__init__.py @@ -25,6 +25,7 @@ _LAZY: dict[str, tuple[str, str]] = { "expand_context": (".graph", "expand_context"), "export_json": (".export", "export_json"), "export_graphml": (".export", "export_graphml"), + "export_markdown": (".export", "export_markdown"), "load_json": (".export", "load_json"), "Neo4jExporter": (".export", "Neo4jExporter"), } diff --git a/kg_ocr/export/__init__.py b/kg_ocr/export/__init__.py index 4f0f980..0d87d1e 100644 --- a/kg_ocr/export/__init__.py +++ b/kg_ocr/export/__init__.py @@ -1,3 +1,4 @@ +from .markdown_exporter import export_markdown from .neo4j_exporter import Neo4jExporter, export_graphml, export_json, load_json -__all__ = ["Neo4jExporter", "export_graphml", "export_json", "load_json"] +__all__ = ["Neo4jExporter", "export_graphml", "export_json", "export_markdown", "load_json"] diff --git a/kg_ocr/export/markdown_exporter.py b/kg_ocr/export/markdown_exporter.py new file mode 100644 index 0000000..6b31338 --- /dev/null +++ b/kg_ocr/export/markdown_exporter.py @@ -0,0 +1,270 @@ +"""Export a knowledge graph to a Markdown report.""" + +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any + +from kg_ocr.graph.analyzer import ( + Anomaly, + detect_anomalies, + summary, + top_citations, + top_cooccurrences, + top_entities, +) + + +def _load_graph(graph_input: Any) -> Any: + """Load a NetworkX graph from a file path or return it directly.""" + import networkx as nx + + if isinstance(graph_input, (str, Path)): + path = Path(graph_input) + data = json.loads(path.read_text(encoding="utf-8")) + try: + return nx.node_link_graph(data, edges="edges") + except TypeError: + return nx.node_link_graph(data) + return graph_input + + +def _cluster_entity_cooccurrences(graph: Any) -> list[list[dict[str, Any]]]: + """Find concept clusters over the entity co-occurrence subgraph.""" + import networkx as nx + + cooccur_graph = nx.Graph() + for node, data in graph.nodes(data=True): + if data.get("kind") == "entity": + cooccur_graph.add_node( + node, + text=data.get("text", node.replace("entity:", "")), + mentions=int(data.get("mentions", 0)), + ) + + for u, v, data in graph.edges(data=True): + if data.get("kind") == "CO_OCCURS": + cooccur_graph.add_edge(u, v, weight=data.get("weight", 1)) + + if cooccur_graph.number_of_nodes() == 0: + return [] + + # Community detection + communities: list[set[str]] = [] + if cooccur_graph.number_of_edges() > 0: + try: + import networkx.algorithms.community as nx_comm # type: ignore[import-untyped] + + communities = list( + nx_comm.louvain_communities(cooccur_graph, weight="weight", seed=42) + ) + except Exception: + try: + import networkx.algorithms.community as nx_comm # type: ignore[import-untyped] + + communities = list( + nx_comm.greedy_modularity_communities(cooccur_graph, weight="weight") + ) + except Exception: + try: + import community as community_louvain # type: ignore[import-not-found] + + partition = community_louvain.best_partition( + cooccur_graph, weight="weight", random_state=42 + ) + comm_dict: dict[int, set[str]] = {} + for n, cid in partition.items(): + comm_dict.setdefault(cid, set()).add(n) + communities = list(comm_dict.values()) + except Exception: + communities = list(nx.connected_components(cooccur_graph)) + else: + # No edges: each entity is an isolated node + communities = [{n} for n in cooccur_graph.nodes()] + + # Format and sort clusters + clusters: list[list[dict[str, Any]]] = [] + for comm in communities: + cluster_nodes = [] + for node in comm: + cluster_nodes.append( + { + "id": node, + "text": cooccur_graph.nodes[node].get("text", node), + "mentions": cooccur_graph.nodes[node].get("mentions", 0), + } + ) + # Sort entities inside cluster by mentions descending + cluster_nodes.sort(key=lambda item: item["mentions"], reverse=True) + clusters.append(cluster_nodes) + + # Sort clusters by size (number of entities) descending, then top entity mentions + clusters.sort( + key=lambda c: (len(c), c[0]["mentions"] if c else 0), + reverse=True, + ) + return clusters + + +def export_markdown(graph_path: str | Path, output_path: str | Path) -> Path: + """Render a knowledge graph into a Markdown summary report. + + Parameters + ---------- + graph_path : str | Path | nx.Graph + Path to `kg_graph.json` or an in-memory NetworkX graph. + output_path : str | Path + Target path for the exported Markdown report. + + Returns + ------- + Path + The output Path written to. + """ + graph = _load_graph(graph_path) + out = Path(output_path) + out.parent.mkdir(parents=True, exist_ok=True) + + counts = summary(graph) + doc_count = counts.get("nodes:document", 0) + chunk_count = counts.get("nodes:chunk", 0) + entity_count = counts.get("nodes:entity", 0) + citation_count = counts.get("nodes:citation", 0) + contains_edges = counts.get("edges:CONTAINS", 0) + mentions_edges = counts.get("edges:MENTIONS", 0) + cites_edges = counts.get("edges:CITES", 0) + cooccurs_edges = counts.get("edges:CO_OCCURS", 0) + + entities = top_entities(graph, limit=20) + cooccurrences = top_cooccurrences(graph, limit=20) + citations = top_citations(graph, limit=20) + clusters = _cluster_entity_cooccurrences(graph) + + # Anomalies detection + anomalies: list[Anomaly] = list(detect_anomalies(graph)) + # Detect isolated citations (citations with 0 citing chunks) + existing_anomaly_nodes = {a.node for a in anomalies} + for node, data in graph.nodes(data=True): + if data.get("kind") == "citation" and node not in existing_anomaly_nodes: + citing_count = sum( + 1 for _, _, edge in graph.in_edges(node, data=True) if edge.get("kind") == "CITES" + ) + if citing_count == 0: + ident = data.get("identifier", node) + anomalies.append( + Anomaly( + "isolated_citation", + node, + f"citation {ident} has no referencing chunks", + ) + ) + + lines: list[str] = [ + "# Knowledge Graph Report", + "", + "## Summary", + f"- **Documents**: {doc_count:,}", + f"- **Chunks**: {chunk_count:,}", + f"- **Entities**: {entity_count:,}", + f"- **Citations**: {citation_count:,}", + f"- **Relationships**: {contains_edges + mentions_edges + cites_edges + cooccurs_edges:,} " + f"(CONTAINS: {contains_edges:,}, MENTIONS: {mentions_edges:,}, CITES: {cites_edges:,}, CO_OCCURS: {cooccurs_edges:,})", + "", + ] + + # Section: Top Entities + lines.append("## Top Entities") + lines.append("") + if entities: + lines.append("| Entity | Mentions |") + lines.append("|:-------|:---------|") + for text, mention_count in entities: + clean_text = text.replace("|", "\\|") + lines.append(f"| {clean_text} | {mention_count:,} |") + else: + lines.append("No entities found.") + lines.append("") + + # Section: Top Co-occurrences + lines.append("## Top Co-occurrences") + lines.append("") + if cooccurrences: + lines.append("| Entity A | Entity B | Co-occurrence Weight |") + lines.append("|:---------|:---------|:---------------------|") + for left, right, weight in cooccurrences: + clean_left = left.replace("|", "\\|") + clean_right = right.replace("|", "\\|") + lines.append(f"| {clean_left} | {clean_right} | {weight:,} |") + else: + lines.append("No entity co-occurrences found.") + lines.append("") + + # Section: Top Citations + lines.append("## Top Citations") + lines.append("") + if citations: + lines.append("| Type | Identifier | Citing Chunks |") + lines.append("|:-----|:-----------|:--------------|") + for ctype, identifier, count in citations: + lines.append(f"| {ctype.upper() or 'UNKNOWN'} | `{identifier}` | {count:,} |") + else: + lines.append("No citations found.") + lines.append("") + + # Section: Concept Clusters + lines.append("## Concept Clusters") + lines.append("") + if clusters: + multi_entity_clusters = [c for c in clusters if len(c) > 1] + single_entity_clusters = [c for c in clusters if len(c) == 1] + + if multi_entity_clusters: + for idx, cluster in enumerate(multi_entity_clusters, start=1): + entity_names = [e["text"] for e in cluster[:3]] + cluster_label = ", ".join(entity_names) + if len(cluster) > 3: + cluster_label += f" (+{len(cluster) - 3} more)" + lines.append(f"### Cluster {idx}: {cluster_label}") + lines.append(f"**Size**: {len(cluster)} entities") + lines.append("") + lines.append("| Entity | Mentions |") + lines.append("|:-------|:---------|") + for item in cluster: + clean_name = item["text"].replace("|", "\\|") + lines.append(f"| {clean_name} | {item['mentions']:,} |") + lines.append("") + + if single_entity_clusters: + lines.append("### Isolated / Unclustered Entities") + lines.append( + f"*{len(single_entity_clusters)} entities with no co-occurrences across documents:*" + ) + lines.append("") + items_preview = [ + f"{c[0]['text']} ({c[0]['mentions']})" for c in single_entity_clusters[:30] + ] + lines.append(", ".join(items_preview)) + if len(single_entity_clusters) > 30: + lines.append(f"*(and {len(single_entity_clusters) - 30} more)*") + lines.append("") + else: + lines.append("No concept clusters detected.") + lines.append("") + + # Section: Anomalies + lines.append("## Anomalies") + lines.append("") + if anomalies: + lines.append("| Anomaly Type | Node | Detail |") + lines.append("|:-------------|:-----|:-------|") + for a in anomalies: + clean_node = a.node.replace("|", "\\|") + clean_detail = a.detail.replace("|", "\\|") + lines.append(f"| `{a.kind}` | `{clean_node}` | {clean_detail} |") + else: + lines.append("No anomalies detected.") + lines.append("") + + out.write_text("\n".join(lines), encoding="utf-8") + return out diff --git a/src/ocr_pipeline/__pycache__/__init__.cpython-313.pyc b/src/ocr_pipeline/__pycache__/__init__.cpython-313.pyc deleted file mode 100644 index e5757a4..0000000 Binary files a/src/ocr_pipeline/__pycache__/__init__.cpython-313.pyc and /dev/null differ diff --git a/src/ocr_pipeline/__pycache__/cli.cpython-313.pyc b/src/ocr_pipeline/__pycache__/cli.cpython-313.pyc deleted file mode 100644 index 8902c6c..0000000 Binary files a/src/ocr_pipeline/__pycache__/cli.cpython-313.pyc and /dev/null differ diff --git a/src/ocr_pipeline/__pycache__/config.cpython-313.pyc b/src/ocr_pipeline/__pycache__/config.cpython-313.pyc deleted file mode 100644 index 9a618a4..0000000 Binary files a/src/ocr_pipeline/__pycache__/config.cpython-313.pyc and /dev/null differ diff --git a/src/ocr_pipeline/__pycache__/pipeline.cpython-313.pyc b/src/ocr_pipeline/__pycache__/pipeline.cpython-313.pyc deleted file mode 100644 index 8f390ac..0000000 Binary files a/src/ocr_pipeline/__pycache__/pipeline.cpython-313.pyc and /dev/null differ diff --git a/src/ocr_pipeline/__pycache__/watch.cpython-313.pyc b/src/ocr_pipeline/__pycache__/watch.cpython-313.pyc deleted file mode 100644 index 2322f16..0000000 Binary files a/src/ocr_pipeline/__pycache__/watch.cpython-313.pyc and /dev/null differ diff --git a/src/ocr_pipeline/cli.py b/src/ocr_pipeline/cli.py index 18e82f8..e0f3f53 100644 --- a/src/ocr_pipeline/cli.py +++ b/src/ocr_pipeline/cli.py @@ -316,8 +316,8 @@ def config(): @kg_app.command("build") def kg_build( - output_dir: Path = typer.Option( - ..., "--output-dir", "-d", help="Pipeline output directory with chunk markdown files" + output_dir: Path | None = typer.Option( + None, "--output-dir", "-d", help="Pipeline output directory with chunk markdown files" ), save: Path | None = typer.Option( None, "--save", "-s", help="Graph JSON path (default: /kg_graph.json)" @@ -325,6 +325,19 @@ def kg_build( ): """Build a knowledge graph from pipeline output and save it as JSON.""" kg_graph, kg_export = _load_kg() + if output_dir is None: + default_candidates = [ + Path("data/ocr_output"), + Path("data"), + Path(settings.output.base_directory), + ] + for candidate in default_candidates: + if candidate.is_dir(): + output_dir = candidate + break + if output_dir is None: + output_dir = Path("data/ocr_output") + if not output_dir.is_dir(): console.print(f"[red]Not a directory:[/red] {output_dir}") raise typer.Exit(code=2) @@ -344,9 +357,9 @@ def kg_build( @kg_app.command("stats") def kg_stats( graph_path: Path = typer.Argument(..., help="Graph JSON written by `kg build`"), - top: int = typer.Option(10, "--top", "-n", help="Rows per top-list"), + top: int = typer.Option(10, "--top", "-n", help="Number of top items to show"), ): - """Summarize a graph: counts, top entities/citations, anomalies.""" + """Summarize graph contents and flag anomalies.""" kg_graph, kg_export = _load_kg() if not graph_path.is_file(): console.print(f"[red]Graph file not found:[/red] {graph_path}") @@ -354,59 +367,66 @@ def kg_stats( graph = kg_export.load_json(graph_path) counts = kg_graph.summary(graph) - table = Table(title="Graph summary") - table.add_column("Kind", style="cyan") - table.add_column("Count", style="green") + summary_table = Table(title=f"Summary ({graph_path})") + summary_table.add_column("Kind", style="cyan") + summary_table.add_column("Count", style="green") for kind, count in sorted(counts.items()): - table.add_row(kind, str(count)) - console.print(table) + summary_table.add_row(kind, str(count)) + console.print(summary_table) entities = kg_graph.top_entities(graph, limit=top) if entities: - table = Table(title=f"Top {top} entities") - table.add_column("Entity", style="cyan") - table.add_column("Mentions", style="green") - for text, mentions in entities: - table.add_row(text, str(mentions)) - console.print(table) + ent_table = Table(title=f"Top {len(entities)} Entities") + ent_table.add_column("Entity", style="cyan") + ent_table.add_column("Mentions", style="green") + for name, count in entities: + ent_table.add_row(name, str(count)) + console.print(ent_table) citations = kg_graph.top_citations(graph, limit=top) if citations: - table = Table(title=f"Top {top} citations") - table.add_column("Type", style="cyan") - table.add_column("Identifier") - table.add_column("Citing chunks", style="green") - for ctype, identifier, citing in citations: - table.add_row(ctype, identifier, str(citing)) - console.print(table) + cit_table = Table(title=f"Top {len(citations)} Citations") + cit_table.add_column("Type", style="cyan") + cit_table.add_column("Identifier", style="magenta") + cit_table.add_column("Citing chunks", style="green") + for ctype, identifier, count in citations: + cit_table.add_row(ctype, identifier, str(count)) + console.print(cit_table) anomalies = kg_graph.detect_anomalies(graph) if anomalies: - console.print(f"\n[yellow]{len(anomalies)} anomalies:[/yellow]") - for anomaly in anomalies[:top]: - console.print(f" [dim]{anomaly.kind}[/dim] {anomaly.node}: {anomaly.detail}") + anom_table = Table(title=f"Anomalies ({len(anomalies)})") + anom_table.add_column("Kind", style="yellow") + anom_table.add_column("Node", style="cyan") + anom_table.add_column("Detail") + for anomaly in anomalies: + anom_table.add_row(anomaly.kind, anomaly.node, anomaly.detail) + console.print(anom_table) + else: + console.print("[green]No anomalies detected.[/green]") @kg_app.command("query") def kg_query( graph_path: Path = typer.Argument(..., help="Graph JSON written by `kg build`"), - entity: str | None = typer.Option(None, "--entity", "-e", help="Entity to look up"), + entity: str | None = typer.Option(None, "--entity", "-e", help="Entity text to query"), citation: str | None = typer.Option( - None, "--citation", "-c", help="Citation identifier (e.g. 10.1038/nature12345)" + None, "--citation", "-c", help="Citation identifier or DOI/PMID" ), expand: bool = typer.Option( - False, "--expand", "-x", help="Include chunks from co-occurring entities" + False, "--expand", "-x", help="Include 1-hop neighbor chunks (co-occurring entities)" ), - limit: int = typer.Option(5, "--limit", "-n", help="Max chunks to show"), + limit: int = typer.Option(10, "--limit", "-l", help="Max chunks to display"), ): - """Retrieve chunks by entity or citation; --expand adds neighbor chunks.""" + """Retrieve text chunks grounded in the knowledge graph.""" kg_graph, kg_export = _load_kg() if not graph_path.is_file(): console.print(f"[red]Graph file not found:[/red] {graph_path}") raise typer.Exit(code=2) if not entity and not citation: - console.print("[red]Give --entity or --citation.[/red]") + console.print("[red]Provide at least one of --entity or --citation.[/red]") raise typer.Exit(code=2) + graph = kg_export.load_json(graph_path) if entity: @@ -416,43 +436,57 @@ def kg_query( else kg_graph.chunks_for_entity(graph, entity) ) related = kg_graph.related_entities(graph, entity) + console.print(f"[bold cyan]Entity:[/bold cyan] {entity} ({len(chunks)} chunks)") if related: - console.print( - "[dim]Related entities: " - + ", ".join(f"{text} ({weight})" for text, weight in related[:5]) - + "[/dim]" - ) - else: + rendered = ", ".join(f"{name} ({weight})" for name, weight in related[:5]) + console.print(f"[dim]Related entities:[/dim] {rendered}") + for chunk in chunks[:limit]: + source = chunk.get("source_path", "unknown") + via = f" (via {chunk['via_entity']})" if "via_entity" in chunk else "" + console.print(f"[bold]{source}[/bold]{via}:") + excerpt = chunk.get("text", "")[:200].replace("\n", " ") + console.print(f" {excerpt}") + + if citation: chunks = kg_graph.chunks_for_citation(graph, citation or "") - - if not chunks: - console.print("[yellow]No matching chunks.[/yellow]") - return - - for payload in chunks[:limit]: - via = f" via {payload['via_entity']}" if payload.get("via_entity") else "" - console.print( - f"\n[bold]{Path(str(payload['source_path'])).name}[/bold] " - f"chunk {payload['chunk_index']}{via} " - f"[dim]({payload['ocr_engine']} {payload['ocr_confidence_mean']:.2f})[/dim]" - ) - excerpt = " ".join(str(payload["text"]).split())[:300] - console.print(f" {excerpt}") + console.print(f"[bold cyan]Citation:[/bold cyan] {citation} ({len(chunks)} chunks)") + for chunk in chunks[:limit]: + source = chunk.get("source_path", "unknown") + console.print(f"[bold]{source}[/bold]:") + excerpt = chunk.get("text", "")[:200].replace("\n", " ") + console.print(f" {excerpt}") @kg_app.command("export") def kg_export_cmd( - graph_path: Path = typer.Argument(..., help="Graph JSON written by `kg build`"), - fmt: str = typer.Option("graphml", "--format", "-f", help="graphml | neo4j"), - out: Path | None = typer.Option(None, "--out", "-o", help="Output path for graphml"), + graph_path: Path | None = typer.Argument( + None, help="Graph JSON written by `kg build` (default: /kg_graph.json)" + ), + fmt: str = typer.Option("graphml", "--format", "-f", help="graphml | neo4j | markdown"), + out: Path | None = typer.Option( + None, "--out", "--output", "-o", help="Output path (default: KG_REPORT.md for markdown)" + ), uri: str | None = typer.Option(None, "--uri", help="Neo4j bolt URI (or NEO4J_URI)"), user: str | None = typer.Option(None, "--user", help="Neo4j user (or NEO4J_USER)"), password: str | None = typer.Option( None, "--password", help="Neo4j password (or NEO4J_PASSWORD)" ), ): - """Export a graph JSON to GraphML or push it into Neo4j.""" + """Export a graph JSON to GraphML, Markdown report, or push it into Neo4j.""" _, kg_export = _load_kg() + if graph_path is None: + candidates = [ + Path("data/ocr_output/kg_graph.json"), + Path("kg_graph.json"), + Path(settings.output.base_directory) / "kg_graph.json", + ] + for c in candidates: + if c.is_file(): + graph_path = c + break + if graph_path is None: + graph_path = Path("data/ocr_output/kg_graph.json") + if not graph_path.is_file(): console.print(f"[red]Graph file not found:[/red] {graph_path}") raise typer.Exit(code=2) @@ -462,6 +496,10 @@ def kg_export_cmd( out = out or graph_path.with_suffix(".graphml") kg_export.export_graphml(graph, out) console.print(f"[green]GraphML written:[/green] {out}") + elif fmt == "markdown": + out = out or Path("KG_REPORT.md") + kg_export.export_markdown(graph, out) + console.print(f"[green]Markdown report written:[/green] {out}") elif fmt == "neo4j": try: with kg_export.Neo4jExporter(uri=uri, user=user, password=password) as exporter: @@ -473,7 +511,7 @@ def kg_export_cmd( f"[green]Pushed to Neo4j:[/green] {pushed['nodes']} nodes, {pushed['edges']} edges" ) else: - console.print(f"[red]Unknown format:[/red] {fmt} (choose graphml or neo4j)") + console.print(f"[red]Unknown format:[/red] {fmt} (choose graphml, neo4j, or markdown)") raise typer.Exit(code=2) diff --git a/tests/test_graph.py b/tests/test_graph.py index 0164b4a..51e0436 100644 --- a/tests/test_graph.py +++ b/tests/test_graph.py @@ -6,7 +6,7 @@ import pytest nx = pytest.importorskip("networkx", reason="kg extra not installed") -from kg_ocr.export import export_graphml, export_json, load_json # noqa: E402 +from kg_ocr.export import export_graphml, export_json, export_markdown, load_json # noqa: E402 from kg_ocr.graph import ( # noqa: E402 build_from_directory, build_graph, @@ -202,6 +202,50 @@ def test_kg_cli_build_and_stats(chunk_dir: Path, tmp_path: Path) -> None: assert result.exit_code == 0, result.output assert out_graphml.is_file() + out_md = tmp_path / "g_report.md" + result = runner.invoke(app, ["kg", "export", str(save), "-f", "markdown", "-o", str(out_md)]) + assert result.exit_code == 0, result.output + assert out_md.is_file() + content = out_md.read_text(encoding="utf-8") + assert "## Top Entities" in content + assert "## Top Co-occurrences" in content + assert "## Top Citations" in content + assert "## Concept Clusters" in content + assert "## Anomalies" in content + assert "BRCA1" in content + assert "10.1038/nature12345" in content + + +def test_export_markdown(chunk_dir: Path, tmp_path: Path) -> None: + graph = build_from_directory(chunk_dir) + save_json = tmp_path / "graph.json" + export_json(graph, save_json) + + out_md = tmp_path / "report.md" + # Test with string paths as per signature requirement + written = export_markdown(str(save_json), str(out_md)) + assert written == out_md + assert out_md.is_file() + + text = out_md.read_text(encoding="utf-8") + assert "# Knowledge Graph Report" in text + assert "## Summary" in text + assert "## Top Entities" in text + assert "## Top Co-occurrences" in text + assert "## Top Citations" in text + assert "## Concept Clusters" in text + assert "## Anomalies" in text + assert "BRCA1" in text + assert "PARP" in text + assert "10.1038/nature12345" in text + assert "low_confidence" in text + + # Also test passing in-memory graph directly + out_direct = tmp_path / "report_direct.md" + export_markdown(graph, out_direct) + assert out_direct.is_file() + assert "## Top Entities" in out_direct.read_text(encoding="utf-8") + def test_chunks_for_entity(chunk_dir: Path) -> None: from kg_ocr.graph import chunks_for_entity