Some checks failed
tests / core (macos-latest, py3.11) (push) Has been cancelled
tests / core (macos-latest, py3.12) (push) Has been cancelled
tests / core (macos-latest, py3.13) (push) Has been cancelled
tests / core (ubuntu-latest, py3.11) (push) Has been cancelled
tests / core (ubuntu-latest, py3.12) (push) Has been cancelled
tests / core (ubuntu-latest, py3.13) (push) Has been cancelled
tests / doctor CLI smoke test (push) Has been cancelled
118 lines
4.2 KiB
Python
118 lines
4.2 KiB
Python
"""Analyze a knowledge graph built from OCR pipeline output."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from dataclasses import dataclass
|
|
from typing import Any
|
|
|
|
LOW_CONFIDENCE_THRESHOLD = 0.4
|
|
HUB_MENTIONS_THRESHOLD = 20
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class Anomaly:
|
|
kind: str
|
|
node: str
|
|
detail: str
|
|
|
|
|
|
def _nodes_of_kind(graph: Any, kind: str) -> list[tuple[str, dict]]:
|
|
return [(n, d) for n, d in graph.nodes(data=True) if d.get("kind") == kind]
|
|
|
|
|
|
def summary(graph: Any) -> dict[str, int]:
|
|
"""Node and edge counts grouped by kind."""
|
|
counts: dict[str, int] = {}
|
|
for _, data in graph.nodes(data=True):
|
|
kind = data.get("kind", "unknown")
|
|
counts[f"nodes:{kind}"] = counts.get(f"nodes:{kind}", 0) + 1
|
|
for _, _, data in graph.edges(data=True):
|
|
kind = data.get("kind", "unknown")
|
|
counts[f"edges:{kind}"] = counts.get(f"edges:{kind}", 0) + 1
|
|
return counts
|
|
|
|
|
|
def top_entities(graph: Any, limit: int = 10) -> list[tuple[str, int]]:
|
|
"""Most-mentioned entities as (text, mentions), descending."""
|
|
entities = [
|
|
(data.get("text", ""), int(data.get("mentions", 0)))
|
|
for _, data in _nodes_of_kind(graph, "entity")
|
|
]
|
|
entities.sort(key=lambda item: item[1], reverse=True)
|
|
return entities[:limit]
|
|
|
|
|
|
def top_cooccurrences(graph: Any, limit: int = 10) -> list[tuple[str, str, int]]:
|
|
"""Strongest entity co-occurrence pairs as (left, right, weight)."""
|
|
pairs = []
|
|
for left, right, data in graph.edges(data=True):
|
|
if data.get("kind") != "CO_OCCURS":
|
|
continue
|
|
left_text = graph.nodes[left].get("text", left)
|
|
right_text = graph.nodes[right].get("text", right)
|
|
pairs.append((left_text, right_text, int(data.get("weight", 1))))
|
|
pairs.sort(key=lambda item: item[2], reverse=True)
|
|
return pairs[:limit]
|
|
|
|
|
|
def top_citations(graph: Any, limit: int = 10) -> list[tuple[str, str, int]]:
|
|
"""Most-cited identifiers as (type, identifier, citing chunks)."""
|
|
rows = []
|
|
for node, data in _nodes_of_kind(graph, "citation"):
|
|
citing = sum(
|
|
1 for _, _, edge in graph.in_edges(node, data=True) if edge.get("kind") == "CITES"
|
|
)
|
|
rows.append((data.get("citation_type", ""), data.get("identifier", node), citing))
|
|
rows.sort(key=lambda item: item[2], reverse=True)
|
|
return rows[:limit]
|
|
|
|
|
|
def detect_anomalies(
|
|
graph: Any,
|
|
low_confidence: float = LOW_CONFIDENCE_THRESHOLD,
|
|
hub_mentions: int = HUB_MENTIONS_THRESHOLD,
|
|
) -> list[Anomaly]:
|
|
"""Flag suspicious graph regions worth re-checking.
|
|
|
|
- documents with very low OCR confidence (likely OCR failure)
|
|
- chunks with empty extracted text
|
|
- documents with zero entities and zero citations (nothing extracted)
|
|
- entity hubs (mentioned everywhere; often a junk pattern)
|
|
"""
|
|
anomalies: list[Anomaly] = []
|
|
|
|
documents = _nodes_of_kind(graph, "document")
|
|
for node, data in documents:
|
|
confidence = float(data.get("ocr_confidence_mean") or 0.0)
|
|
if confidence < low_confidence:
|
|
anomalies.append(
|
|
Anomaly(
|
|
"low_confidence",
|
|
node,
|
|
f"ocr_confidence_mean={confidence:.2f} < {low_confidence}",
|
|
)
|
|
)
|
|
|
|
meaningful_out = 0
|
|
for _, target, edge in graph.out_edges(node, data=True):
|
|
if edge.get("kind") != "CONTAINS":
|
|
continue
|
|
for _, _, chunk_edge in graph.out_edges(target, data=True):
|
|
if chunk_edge.get("kind") in ("MENTIONS", "CITES"):
|
|
meaningful_out += 1
|
|
if meaningful_out == 0:
|
|
anomalies.append(Anomaly("empty_document", node, "no entities or citations extracted"))
|
|
|
|
for node, data in _nodes_of_kind(graph, "chunk"):
|
|
if not (data.get("text") or "").strip():
|
|
anomalies.append(Anomaly("empty_chunk", node, "extracted text is empty"))
|
|
|
|
for node, data in _nodes_of_kind(graph, "entity"):
|
|
mentions = int(data.get("mentions", 0))
|
|
if mentions >= hub_mentions:
|
|
anomalies.append(
|
|
Anomaly("entity_hub", node, f"mentioned {mentions} times; possible junk pattern")
|
|
)
|
|
|
|
return anomalies
|