Files
Celes Renata a72f336ad1 feat: Intelligence Pipeline v3 — full implementation
Multi-stage evidence-grounded inference architecture replacing the
monolithic 9B model extraction pipeline. CPU-first specialist services
handle routine extraction while the 9B vLLM model is preserved for
semantic adjudication of ambiguous cases.

Key components:
- Capability-aware inference gateway (OpenAI-compatible + Ollama)
- Endpoint registry with DB migrations and REST API
- Sentence-aware document segmenter (property tests)
- Deterministic financial parsing with offset integrity
- Symbol resolution with ambiguity detection
- Specialist service (GLiNER2, dynamic batching, K8s deployment)
- Company-specific sentiment (FinBERT, calibration)
- Retrieval-based novelty and duplicate detection
- Confidence calibration pipeline
- Deterministic routing engine (property tests)
- 9B adjudication layer with VRAM gating
- Stock-specific impact model (features, labels, baseline, trained)
- Pipeline orchestrator (state machine, queues, leases, feature flags)
- Bounded parallelism (async workers, semaphore, load shedding)
- Observability (tracing, metrics, alerts)
- Compatibility adapter (v3→v2 golden mapping tests)
- Shadow/canary promotion framework
- Active learning and fine-tuning pipeline

Test results: 1,161 tests pass, ruff lint clean.
All 282 spec tasks completed.
2026-07-13 02:14:59 +00:00

186 lines
5.9 KiB
Python

"""Symbol resolver: maps textual mentions to canonical company identities.
Uses an in-memory alias index built from company registry data. Returns
ranked candidates with confidence scores and ambiguity margins. Does NOT
invent tickers for unresolved mentions.
"""
from __future__ import annotations
from services.intelligence_pipeline_v3.resolution.alias_index import (
AliasIndex,
IndexEntry,
build_alias_index,
)
from services.intelligence_pipeline_v3.resolution.models import (
MatchType,
MentionType,
ResolutionCandidate,
ResolutionResult,
UnresolvedMention,
UnresolvedReason,
)
# Ambiguity threshold: if the gap between top-2 candidates is below this,
# the result is marked as ambiguous.
_AMBIGUITY_THRESHOLD = 0.15
class SymbolResolver:
"""Resolve document mentions to canonical company identifiers.
Usage:
resolver = SymbolResolver()
resolver.load_registry(companies)
result = resolver.resolve("Apple")
"""
def __init__(self, ambiguity_threshold: float = _AMBIGUITY_THRESHOLD) -> None:
self._index = AliasIndex()
self._ambiguity_threshold = ambiguity_threshold
@property
def index(self) -> AliasIndex:
"""Access the underlying alias index."""
return self._index
def load_registry(self, companies: list[dict]) -> None:
"""Load company data into the alias index.
Each company dict should contain at minimum:
- id: str (company UUID)
- ticker: str
- legal_name: str
Optional fields:
- aliases: list[dict] with keys "alias" and optionally "alias_type"
OR list[tuple[str, str]] of (alias_text, alias_type)
Uses `build_alias_index` to construct the full index from registry data.
"""
self._index = build_alias_index(companies)
def resolve(
self,
mention: str,
context: str = "",
mention_type: MentionType = MentionType.explicit,
) -> ResolutionResult:
"""Resolve a textual mention to ranked company candidates.
Args:
mention: The text to resolve (ticker, name, alias, etc.)
context: Optional surrounding text for future disambiguation use.
mention_type: Whether this is an explicit mention or inferred exposure.
Returns:
ResolutionResult with ranked candidates, ambiguity margin, and metadata.
If no match is found, candidates list is empty.
"""
entries = self._index.lookup(mention)
if not entries:
return ResolutionResult(
candidates=[],
ambiguity_margin=1.0,
is_ambiguous=False,
mention_type=mention_type,
)
# Deduplicate by company_id, keeping the best match type per company.
best_per_company: dict[str, IndexEntry] = {}
for entry in entries:
existing = best_per_company.get(entry.company_id)
if existing is None or _match_priority(entry.match_type) > _match_priority(existing.match_type):
best_per_company[entry.company_id] = entry
# Score candidates.
candidates: list[ResolutionCandidate] = []
for entry in best_per_company.values():
confidence = _score_match(entry.match_type, mention, entry)
candidates.append(
ResolutionCandidate(
company_id=entry.company_id,
ticker=entry.ticker,
name=entry.name,
confidence=confidence,
match_type=MatchType(entry.match_type),
)
)
# Sort by confidence descending.
candidates.sort(key=lambda c: c.confidence, reverse=True)
# Calculate ambiguity margin.
if len(candidates) >= 2:
ambiguity_margin = candidates[0].confidence - candidates[1].confidence
else:
ambiguity_margin = 1.0
is_ambiguous = ambiguity_margin < self._ambiguity_threshold
return ResolutionResult(
candidates=candidates,
ambiguity_margin=ambiguity_margin,
is_ambiguous=is_ambiguous,
mention_type=mention_type,
)
def resolve_or_unresolved(
self,
mention: str,
start_char: int,
end_char: int,
context: str = "",
mention_type: MentionType = MentionType.explicit,
) -> ResolutionResult | UnresolvedMention:
"""Resolve a mention, returning UnresolvedMention if no match found.
This ensures unresolved mentions are preserved with their literal text
and position rather than having a ticker invented.
"""
result = self.resolve(mention, context=context, mention_type=mention_type)
if not result.candidates:
return UnresolvedMention(
literal_text=mention,
start_char=start_char,
end_char=end_char,
reason=UnresolvedReason.not_in_registry,
)
if result.is_ambiguous:
return UnresolvedMention(
literal_text=mention,
start_char=start_char,
end_char=end_char,
reason=UnresolvedReason.ambiguous,
)
return result
def _match_priority(match_type: str) -> int:
"""Higher priority = stronger match type."""
priorities = {
"exact_ticker": 3,
"exact_name": 2,
"alias": 1,
"fuzzy": 0,
}
return priorities.get(match_type, 0)
def _score_match(match_type: str, mention: str, entry: IndexEntry) -> float:
"""Score a match based on type and exact-match quality.
Exact ticker matches get highest confidence, then exact name, then alias.
"""
base_scores = {
"exact_ticker": 0.95,
"exact_name": 0.90,
"alias": 0.80,
"fuzzy": 0.50,
}
return base_scores.get(match_type, 0.50)