"""Core evidence verifier for Intelligence Pipeline v3. Validates offsets, entity-evidence associations, and numeric consistency for extracted candidates against source text and evidence spans. """ from __future__ import annotations import re from dataclasses import dataclass, field from datetime import datetime, timezone from services.intelligence_pipeline_v3.verification.models import ( AssociationVerification, NumericVerification, OffsetVerification, RejectedCandidate, RejectionReason, VerificationReport, ) @dataclass class EvidenceSpan: """Minimal evidence span representation for verification.""" id: str start_char: int end_char: int text: str @dataclass class Entity: """Minimal entity representation for verification.""" id: str literal_text: str evidence_ids: list[str] @dataclass class NumericFact: """Minimal numeric fact representation for verification.""" id: str literal_value: str normalized_value: float | None evidence_ids: list[str] @dataclass class Candidate: """Generic candidate for full verification.""" candidate_type: str candidate_id: str candidate_data: dict = field(default_factory=dict) evidence_ids: list[str] = field(default_factory=list) literal_text: str | None = None normalized_value: float | None = None # Regex for extracting numbers from text (integers, decimals, negative) _NUMBER_RE = re.compile(r"-?\d[\d,]*\.?\d*") def _extract_numbers(text: str) -> list[float]: """Extract all numeric values from a string.""" results: list[float] = [] for match in _NUMBER_RE.finditer(text): try: # Remove commas before parsing cleaned = match.group().replace(",", "") results.append(float(cleaned)) except ValueError: continue return results class EvidenceVerifier: """Verifies evidence spans, entity associations, and numeric consistency. Implements requirement 8 (Evidence Verification and Grounding): - Validates span offsets match source text - Checks entity text appears in linked evidence spans - Verifies numeric values can be found in evidence text - Produces a full verification report with rejected candidates """ def __init__(self, numeric_tolerance: float = 0.01) -> None: """Initialize the verifier. Args: numeric_tolerance: Relative tolerance for numeric comparison (default 1%). """ self._numeric_tolerance = numeric_tolerance self._rejected: list[RejectedCandidate] = [] @property def rejected_candidates(self) -> list[RejectedCandidate]: """All rejected candidates accumulated during verification.""" return list(self._rejected) def reset(self) -> None: """Clear accumulated rejected candidates.""" self._rejected = [] def verify_offsets( self, spans: list[EvidenceSpan], source_text: str ) -> list[OffsetVerification]: """Check that each span's text matches the source at declared offsets. Args: spans: Evidence spans with start_char, end_char, and text. source_text: The full original document text. Returns: A list of OffsetVerification results, one per span. """ results: list[OffsetVerification] = [] for span in spans: # Bounds check if span.start_char < 0 or span.end_char > len(source_text): results.append( OffsetVerification( span_id=span.id, valid=False, reason=( f"Offset out of bounds: start={span.start_char}, " f"end={span.end_char}, source_length={len(source_text)}" ), ) ) self._reject_span(span, RejectionReason.INVALID_OFFSET) continue if span.end_char <= span.start_char: results.append( OffsetVerification( span_id=span.id, valid=False, reason=f"Invalid range: end_char ({span.end_char}) <= start_char ({span.start_char})", ) ) self._reject_span(span, RejectionReason.INVALID_OFFSET) continue # Extract text at offsets and compare actual_text = source_text[span.start_char : span.end_char] if actual_text == span.text: results.append(OffsetVerification(span_id=span.id, valid=True)) else: results.append( OffsetVerification( span_id=span.id, valid=False, reason=( f"Text mismatch at [{span.start_char}:{span.end_char}]: " f"expected {span.text!r}, found {actual_text!r}" ), ) ) self._reject_span(span, RejectionReason.TEXT_MISMATCH) return results def verify_entity_association( self, entity: Entity, evidence_spans: list[EvidenceSpan] ) -> AssociationVerification: """Check that entity text appears in at least one linked evidence span. The check is case-insensitive to handle variation in capitalization. Args: entity: Entity with literal_text and evidence_ids. evidence_spans: All available evidence spans. Returns: AssociationVerification indicating whether the entity is supported. """ # Build a map of span_id -> span for quick lookup span_map = {s.id: s for s in evidence_spans} entity_lower = entity.literal_text.lower() for eid in entity.evidence_ids: span = span_map.get(eid) if span and entity_lower in span.text.lower(): return AssociationVerification(entity_id=entity.id, valid=True) reason = ( f"Entity text {entity.literal_text!r} not found in any linked evidence span " f"(checked {len(entity.evidence_ids)} spans)" ) self._rejected.append( RejectedCandidate( candidate_type="entity", candidate_data={ "entity_id": entity.id, "literal_text": entity.literal_text, "evidence_ids": entity.evidence_ids, }, rejection_reason=RejectionReason.ENTITY_NOT_IN_EVIDENCE, stage="entity_association_verification", timestamp=datetime.now(tz=timezone.utc), ) ) return AssociationVerification( entity_id=entity.id, valid=False, reason=reason ) def verify_numeric_consistency( self, fact: NumericFact, evidence_spans: list[EvidenceSpan] ) -> NumericVerification: """Check that a numeric fact's value can be found in its evidence text. Looks for the literal value or the normalized value within the text of linked evidence spans. Uses tolerance-aware comparison for normalized values. Args: fact: Numeric fact with literal_value, normalized_value, and evidence_ids. evidence_spans: All available evidence spans. Returns: NumericVerification indicating whether the value was found. """ span_map = {s.id: s for s in evidence_spans} expected_str = fact.literal_value # First: check if literal value string appears in any linked span for eid in fact.evidence_ids: span = span_map.get(eid) if span and fact.literal_value in span.text: return NumericVerification( fact_id=fact.id, valid=True, found_value=fact.literal_value, expected_value=expected_str, ) # Second: if we have a normalized value, look for numeric matches if fact.normalized_value is not None: for eid in fact.evidence_ids: span = span_map.get(eid) if not span: continue numbers = _extract_numbers(span.text) for num in numbers: if self._values_match(num, fact.normalized_value): return NumericVerification( fact_id=fact.id, valid=True, found_value=str(num), expected_value=expected_str, ) # Rejection reason = ( f"Value {expected_str!r} (normalized={fact.normalized_value}) " f"not found in evidence spans" ) self._rejected.append( RejectedCandidate( candidate_type="fact", candidate_data={ "fact_id": fact.id, "literal_value": fact.literal_value, "normalized_value": fact.normalized_value, "evidence_ids": fact.evidence_ids, }, rejection_reason=RejectionReason.NUMERIC_INCONSISTENCY, stage="numeric_consistency_verification", timestamp=datetime.now(tz=timezone.utc), ) ) return NumericVerification( fact_id=fact.id, valid=False, found_value=None, expected_value=expected_str, reason=reason, ) def verify_all( self, candidates: list[Candidate], evidence_spans: list[EvidenceSpan], source_text: str, ) -> VerificationReport: """Run full verification across all candidates. Performs offset verification on spans, then entity association and numeric consistency for each candidate as appropriate. Args: candidates: All extraction candidates to verify. evidence_spans: All evidence spans in the document. source_text: Full original document text. Returns: A VerificationReport summarizing the verification results. """ self.reset() # Step 1: Verify all span offsets offset_results = self.verify_offsets(evidence_spans, source_text) valid_span_ids = {r.span_id for r in offset_results if r.valid} # Step 2: Filter spans to only valid ones for downstream checks valid_spans = [s for s in evidence_spans if s.id in valid_span_ids] total = len(candidates) verified = 0 for candidate in candidates: # Filter candidate's evidence_ids to only valid spans linked_valid_spans = [ s for s in valid_spans if s.id in set(candidate.evidence_ids) ] if not linked_valid_spans and candidate.evidence_ids: # All linked spans were invalid self._rejected.append( RejectedCandidate( candidate_type=candidate.candidate_type, candidate_data=candidate.candidate_data, rejection_reason=RejectionReason.INVALID_OFFSET, stage="full_verification", timestamp=datetime.now(tz=timezone.utc), ) ) continue # Entity association check if candidate.literal_text is not None: entity = Entity( id=candidate.candidate_id, literal_text=candidate.literal_text, evidence_ids=candidate.evidence_ids, ) assoc = self.verify_entity_association(entity, valid_spans) if not assoc.valid: continue # Numeric consistency check if candidate.normalized_value is not None: fact = NumericFact( id=candidate.candidate_id, literal_value=candidate.literal_text or str(candidate.normalized_value), normalized_value=candidate.normalized_value, evidence_ids=candidate.evidence_ids, ) num_check = self.verify_numeric_consistency(fact, valid_spans) if not num_check.valid: continue verified += 1 rejected = total - verified coverage_rate = verified / total if total > 0 else 1.0 # Build rejection breakdown breakdown: dict[str, int] = {} for rc in self._rejected: key = rc.rejection_reason.value breakdown[key] = breakdown.get(key, 0) + 1 return VerificationReport( total_candidates=total, verified=verified, rejected=rejected, coverage_rate=coverage_rate, rejection_breakdown=breakdown, ) def _values_match(self, found: float, expected: float) -> bool: """Compare two numeric values with tolerance.""" if expected == 0: return abs(found) < self._numeric_tolerance return abs(found - expected) / abs(expected) <= self._numeric_tolerance def _reject_span(self, span: EvidenceSpan, reason: RejectionReason) -> None: """Record a span rejection.""" self._rejected.append( RejectedCandidate( candidate_type="evidence_span", candidate_data={ "span_id": span.id, "start_char": span.start_char, "end_char": span.end_char, "text": span.text[:100], # Truncate for storage }, rejection_reason=reason, stage="offset_verification", timestamp=datetime.now(tz=timezone.utc), ) )