Files
stonks-oracle/services/intelligence_pipeline_v3/compatibility/adapter.py
T
Celes Renata a72f336ad1 feat: Intelligence Pipeline v3 — full implementation
Multi-stage evidence-grounded inference architecture replacing the
monolithic 9B model extraction pipeline. CPU-first specialist services
handle routine extraction while the 9B vLLM model is preserved for
semantic adjudication of ambiguous cases.

Key components:
- Capability-aware inference gateway (OpenAI-compatible + Ollama)
- Endpoint registry with DB migrations and REST API
- Sentence-aware document segmenter (property tests)
- Deterministic financial parsing with offset integrity
- Symbol resolution with ambiguity detection
- Specialist service (GLiNER2, dynamic batching, K8s deployment)
- Company-specific sentiment (FinBERT, calibration)
- Retrieval-based novelty and duplicate detection
- Confidence calibration pipeline
- Deterministic routing engine (property tests)
- 9B adjudication layer with VRAM gating
- Stock-specific impact model (features, labels, baseline, trained)
- Pipeline orchestrator (state machine, queues, leases, feature flags)
- Bounded parallelism (async workers, semaphore, load shedding)
- Observability (tracing, metrics, alerts)
- Compatibility adapter (v3→v2 golden mapping tests)
- Shadow/canary promotion framework
- Active learning and fine-tuning pipeline

Test results: 1,161 tests pass, ruff lint clean.
All 282 spec tasks completed.
2026-07-13 02:14:59 +00:00

209 lines
7.0 KiB
Python

"""Compatibility adapter — maps approved v3 records to current v2 data classes.
The adapter creates current-format records without discarding v3 provenance.
It marks model_provider='hybrid' and stores complete stage lineage separately.
Design reference: Section K (Compatibility Adapter) in design.md.
"""
from __future__ import annotations
import uuid
from services.intelligence_pipeline_v3.compatibility.config import (
AdapterMode,
is_adapter_enabled,
)
from services.intelligence_pipeline_v3.compatibility.models import (
AdapterLineage,
V2ImpactRecord,
V2IntelligenceRecord,
V3CompanySignal,
V3HorizonProbabilities,
V3IntelligenceRecord,
V3SentimentDistribution,
)
ADAPTER_VERSION = "1.0.0"
class AdapterDisabledError(Exception):
"""Raised when the adapter is called in disabled mode."""
pass
class CompatibilityAdapter:
"""Maps v3 intelligence records to v2 format for downstream consumers.
The adapter is gated by AdapterMode — it refuses to produce output when
disabled, ensuring v3 records cannot accidentally affect production
consumers until explicitly enabled.
"""
def __init__(self, mode: AdapterMode = AdapterMode.DISABLED) -> None:
self._mode = mode
@property
def mode(self) -> AdapterMode:
return self._mode
@property
def version(self) -> str:
return ADAPTER_VERSION
def map_to_v2(
self, v3_record: V3IntelligenceRecord
) -> tuple[V2IntelligenceRecord, AdapterLineage]:
"""Map an approved v3 record to v2 intelligence + impact records.
Returns:
A tuple of (V2IntelligenceRecord, AdapterLineage).
Raises:
AdapterDisabledError: If the adapter is in disabled mode.
"""
if not is_adapter_enabled(self._mode):
raise AdapterDisabledError(
f"Adapter is disabled (mode={self._mode.value}). "
"Enable replay, shadow, canary, or production mode to use."
)
v2_id = str(uuid.uuid4())
# Map each company signal to a v2 impact record
impact_records = [
self._map_company_signal(signal) for signal in v3_record.company_signals
]
v2_record = V2IntelligenceRecord(
id=v2_id,
document_id=v3_record.document_id,
summary=v3_record.summary,
macro_themes=v3_record.macro_themes,
novelty_score=v3_record.novelty_score,
confidence=v3_record.confidence,
model_provider="hybrid",
model_name="intelligence-pipeline-v3",
prompt_version=f"adapter-{ADAPTER_VERSION}",
schema_version="3.0.0",
impact_records=impact_records,
)
lineage = AdapterLineage(
adapter_version=ADAPTER_VERSION,
pipeline_version=v3_record.pipeline_version,
v3_document_id=v3_record.document_id,
v2_intelligence_id=v2_id,
stage_runs=v3_record.stage_runs,
mapping_notes=[
f"Mapped {len(v3_record.company_signals)} company signals",
f"Mode: {self._mode.value}",
],
)
return v2_record, lineage
def _map_company_signal(self, signal: V3CompanySignal) -> V2ImpactRecord:
"""Map a single v3 company signal to a v2 impact record."""
return V2ImpactRecord(
company_id=signal.company_id,
ticker=signal.ticker,
relevance=signal.relevance_probability,
sentiment=self._map_sentiment(signal.sentiment),
impact_score=self._map_impact_score(signal),
impact_horizon=self._map_horizon(signal.horizon_probabilities),
catalyst_type=self._map_catalyst_type(signal.event_classes),
evidence_spans=signal.evidence_spans,
)
@staticmethod
def _map_sentiment(dist: V3SentimentDistribution) -> str:
"""Map probability distribution to legacy sentiment enum.
Logic:
- If max probability is neutral and ≥ 0.5 → neutral
- If positive and negative are both ≥ 0.3 → mixed
- Otherwise take the argmax of positive/negative/neutral
"""
pos, neg, neu = dist.positive, dist.negative, dist.neutral
# Mixed detection: both positive and negative have significant mass
if pos >= 0.3 and neg >= 0.3:
return "mixed"
# Argmax
max_val = max(pos, neg, neu)
if max_val == neu:
return "neutral"
elif max_val == pos:
return "positive"
else:
return "negative"
@staticmethod
def _map_impact_score(signal: V3CompanySignal) -> float:
"""Map v3 expected_magnitude to legacy impact_score in [-1, 1].
The v3 expected_magnitude is already a signed value representing
expected market response. We clamp to [-1, 1] for legacy compatibility.
If expected_magnitude is None, derive a conservative estimate from
direction probabilities.
"""
if signal.expected_magnitude is not None:
return max(-1.0, min(1.0, signal.expected_magnitude))
# Fallback: derive from direction probabilities
dp = signal.direction_probabilities
# Signed score: positive_prob - negative_prob, scaled to [-1, 1]
signed = dp.positive - dp.negative
return max(-1.0, min(1.0, signed))
@staticmethod
def _map_horizon(probs: V3HorizonProbabilities) -> str:
"""Map horizon probability distribution to single legacy horizon string.
Returns the horizon with the highest probability (argmax).
Ties are broken by preferring shorter horizons.
"""
horizon_map = {
"intraday": probs.intraday,
"1d": probs.one_day,
"7d": probs.seven_day,
"30d": probs.thirty_day,
"90d": probs.ninety_day,
}
# argmax with tie-breaking by order (shortest first)
return max(horizon_map, key=lambda k: horizon_map[k])
@staticmethod
def _map_catalyst_type(event_classes: list[str]) -> str:
"""Map v3 event taxonomy to legacy catalyst_type enum.
Uses the first matching event class. Falls back to 'other'.
"""
# Mapping from v3 event classes to legacy CatalystType values
event_to_catalyst: dict[str, str] = {
"earnings_beat": "earnings",
"earnings_miss": "earnings",
"guidance_raise": "earnings",
"guidance_cut": "earnings",
"product_launch": "product",
"legal_regulatory": "legal",
"ma_announcement": "m_and_a",
"supply_chain": "supply_chain",
"rating_change": "rating_change",
"macro_event": "macro",
"management_change": "other",
"dividend_change": "other",
"buyback": "other",
}
for event_class in event_classes:
if event_class in event_to_catalyst:
return event_to_catalyst[event_class]
return "other"