Files
Celes Renata a72f336ad1 feat: Intelligence Pipeline v3 — full implementation
Multi-stage evidence-grounded inference architecture replacing the
monolithic 9B model extraction pipeline. CPU-first specialist services
handle routine extraction while the 9B vLLM model is preserved for
semantic adjudication of ambiguous cases.

Key components:
- Capability-aware inference gateway (OpenAI-compatible + Ollama)
- Endpoint registry with DB migrations and REST API
- Sentence-aware document segmenter (property tests)
- Deterministic financial parsing with offset integrity
- Symbol resolution with ambiguity detection
- Specialist service (GLiNER2, dynamic batching, K8s deployment)
- Company-specific sentiment (FinBERT, calibration)
- Retrieval-based novelty and duplicate detection
- Confidence calibration pipeline
- Deterministic routing engine (property tests)
- 9B adjudication layer with VRAM gating
- Stock-specific impact model (features, labels, baseline, trained)
- Pipeline orchestrator (state machine, queues, leases, feature flags)
- Bounded parallelism (async workers, semaphore, load shedding)
- Observability (tracing, metrics, alerts)
- Compatibility adapter (v3→v2 golden mapping tests)
- Shadow/canary promotion framework
- Active learning and fine-tuning pipeline

Test results: 1,161 tests pass, ruff lint clean.
All 282 spec tasks completed.
2026-07-13 02:14:59 +00:00

315 lines
9.1 KiB
Python

"""Evidence offset validity, support rate, coverage, and orphan metrics.
Implements evaluation metrics for evidence grounding quality:
- Offset validity rate: proportion of spans where text matches source at offsets
- Support rate: proportion of extracted items with at least one valid evidence span
- Coverage score: average proportion of required fields supported by evidence
- Orphan rate: proportion of evidence spans not referenced by any extracted item
- Per-field support: breakdown of support rate by field type
- Unsupported claim rate: proportion of extracted items with no valid evidence
Validates: Requirements 16.3, 16.4
"""
from __future__ import annotations
from enum import Enum
from pydantic import BaseModel, Field
# ---------------------------------------------------------------------------
# Domain Models
# ---------------------------------------------------------------------------
class FieldType(str, Enum):
"""Types of extracted fields that can be evidence-supported."""
entity = "entity"
event = "event"
fact = "fact"
sentiment = "sentiment"
class EvidenceSpan(BaseModel):
"""An evidence span with text and character offsets into source."""
span_id: str
text: str
start_char: int
end_char: int
document_id: str = ""
class ExtractionResult(BaseModel):
"""An extracted item referencing evidence spans by ID."""
item_id: str
field_type: FieldType
evidence_ids: list[str] = Field(default_factory=list)
required_fields: list[str] = Field(default_factory=list)
supported_fields: list[str] = Field(default_factory=list)
# ---------------------------------------------------------------------------
# Result Models
# ---------------------------------------------------------------------------
class EvidenceMetricsResult(BaseModel):
"""Complete evidence evaluation result."""
validity_rate: float = Field(ge=0.0, le=1.0)
support_rate: float = Field(ge=0.0, le=1.0)
coverage_score: float = Field(ge=0.0, le=1.0)
orphan_rate: float = Field(ge=0.0, le=1.0)
per_field_support: dict[str, float]
unsupported_claim_rate: float = Field(ge=0.0, le=1.0)
total_spans: int = Field(ge=0)
valid_spans: int = Field(ge=0)
total_items: int = Field(ge=0)
supported_items: int = Field(ge=0)
orphan_spans: int = Field(ge=0)
# ---------------------------------------------------------------------------
# Core Metric Computation
# ---------------------------------------------------------------------------
def compute_offset_validity(
spans: list[EvidenceSpan],
source_text: str,
) -> tuple[float, int, int]:
"""Compute the proportion of spans whose text matches source at offsets.
Args:
spans: Evidence spans with text and character offsets.
source_text: The original source document text.
Returns:
Tuple of (validity_rate, valid_count, total_count).
"""
if not spans:
return (1.0, 0, 0)
valid = 0
for span in spans:
start = span.start_char
end = span.end_char
# Basic bounds check
if start < 0 or end < 0 or start > end:
continue
if end > len(source_text):
continue
source_slice = source_text[start:end]
if source_slice == span.text:
valid += 1
total = len(spans)
rate = valid / total
return (rate, valid, total)
def compute_support_rate(
items: list[ExtractionResult],
valid_span_ids: set[str],
) -> tuple[float, int, int]:
"""Compute proportion of items with at least one valid evidence span.
Args:
items: Extraction results referencing evidence span IDs.
valid_span_ids: Set of span IDs that passed offset validity.
Returns:
Tuple of (support_rate, supported_count, total_count).
"""
if not items:
return (1.0, 0, 0)
supported = 0
for item in items:
if any(eid in valid_span_ids for eid in item.evidence_ids):
supported += 1
total = len(items)
rate = supported / total
return (rate, supported, total)
def compute_coverage_score(
items: list[ExtractionResult],
) -> float:
"""Compute average proportion of required fields supported by evidence.
For each item, coverage = len(supported_fields ∩ required_fields) / len(required_fields).
Items with no required fields are treated as fully covered.
Args:
items: Extraction results with required and supported field lists.
Returns:
Average coverage score across all items.
"""
if not items:
return 1.0
total_coverage = 0.0
for item in items:
if not item.required_fields:
total_coverage += 1.0
continue
required = set(item.required_fields)
supported = set(item.supported_fields)
covered = required & supported
total_coverage += len(covered) / len(required)
return total_coverage / len(items)
def compute_orphan_rate(
spans: list[EvidenceSpan],
items: list[ExtractionResult],
) -> tuple[float, int]:
"""Compute proportion of evidence spans not referenced by any item.
Args:
spans: All evidence spans.
items: Extraction results referencing evidence span IDs.
Returns:
Tuple of (orphan_rate, orphan_count).
"""
if not spans:
return (0.0, 0)
referenced_ids: set[str] = set()
for item in items:
referenced_ids.update(item.evidence_ids)
orphan_count = sum(1 for span in spans if span.span_id not in referenced_ids)
rate = orphan_count / len(spans)
return (rate, orphan_count)
def compute_per_field_support(
items: list[ExtractionResult],
valid_span_ids: set[str],
) -> dict[str, float]:
"""Compute support rate broken down by field type.
Args:
items: Extraction results with field types and evidence IDs.
valid_span_ids: Set of span IDs that passed offset validity.
Returns:
Dict mapping field type name to support rate.
"""
by_type: dict[str, list[ExtractionResult]] = {}
for item in items:
key = item.field_type.value
by_type.setdefault(key, []).append(item)
result: dict[str, float] = {}
for field_type, type_items in sorted(by_type.items()):
rate, _, _ = compute_support_rate(type_items, valid_span_ids)
result[field_type] = rate
return result
def compute_unsupported_claim_rate(
items: list[ExtractionResult],
valid_span_ids: set[str],
) -> float:
"""Compute proportion of items with no valid evidence at all.
An item is unsupported if it has no evidence_ids OR none of its
evidence_ids are in the valid set.
Args:
items: Extraction results referencing evidence span IDs.
valid_span_ids: Set of span IDs that passed offset validity.
Returns:
Unsupported claim rate (0.0 to 1.0).
"""
if not items:
return 0.0
unsupported = 0
for item in items:
if not item.evidence_ids:
unsupported += 1
elif not any(eid in valid_span_ids for eid in item.evidence_ids):
unsupported += 1
return unsupported / len(items)
# ---------------------------------------------------------------------------
# Public API
# ---------------------------------------------------------------------------
def evaluate_evidence(
spans: list[EvidenceSpan],
source_text: str,
items: list[ExtractionResult],
) -> EvidenceMetricsResult:
"""Run full evidence evaluation producing a complete metrics report.
Args:
spans: All evidence spans with text and offsets.
source_text: The original source document text.
items: Extraction results referencing evidence spans.
Returns:
EvidenceMetricsResult with all computed metrics.
"""
# Step 1: Offset validity
validity_rate, valid_count, total_spans = compute_offset_validity(spans, source_text)
# Step 2: Build valid span ID set
valid_span_ids: set[str] = set()
for span in spans:
start = span.start_char
end = span.end_char
if start < 0 or end < 0 or start > end:
continue
if end > len(source_text):
continue
if source_text[start:end] == span.text:
valid_span_ids.add(span.span_id)
# Step 3: Support rate
support_rate, supported_count, total_items = compute_support_rate(items, valid_span_ids)
# Step 4: Coverage score
coverage_score = compute_coverage_score(items)
# Step 5: Orphan rate
orphan_rate, orphan_count = compute_orphan_rate(spans, items)
# Step 6: Per-field support
per_field_support = compute_per_field_support(items, valid_span_ids)
# Step 7: Unsupported claim rate
unsupported_claim_rate = compute_unsupported_claim_rate(items, valid_span_ids)
return EvidenceMetricsResult(
validity_rate=validity_rate,
support_rate=support_rate,
coverage_score=coverage_score,
orphan_rate=orphan_rate,
per_field_support=per_field_support,
unsupported_claim_rate=unsupported_claim_rate,
total_spans=total_spans,
valid_spans=valid_count,
total_items=total_items,
supported_items=supported_count,
orphan_spans=orphan_count,
)