feat: Intelligence Pipeline v3 — full implementation

Multi-stage evidence-grounded inference architecture replacing the
monolithic 9B model extraction pipeline. CPU-first specialist services
handle routine extraction while the 9B vLLM model is preserved for
semantic adjudication of ambiguous cases.

Key components:
- Capability-aware inference gateway (OpenAI-compatible + Ollama)
- Endpoint registry with DB migrations and REST API
- Sentence-aware document segmenter (property tests)
- Deterministic financial parsing with offset integrity
- Symbol resolution with ambiguity detection
- Specialist service (GLiNER2, dynamic batching, K8s deployment)
- Company-specific sentiment (FinBERT, calibration)
- Retrieval-based novelty and duplicate detection
- Confidence calibration pipeline
- Deterministic routing engine (property tests)
- 9B adjudication layer with VRAM gating
- Stock-specific impact model (features, labels, baseline, trained)
- Pipeline orchestrator (state machine, queues, leases, feature flags)
- Bounded parallelism (async workers, semaphore, load shedding)
- Observability (tracing, metrics, alerts)
- Compatibility adapter (v3→v2 golden mapping tests)
- Shadow/canary promotion framework
- Active learning and fine-tuning pipeline

Test results: 1,161 tests pass, ruff lint clean.
All 282 spec tasks completed.
This commit is contained in:
Celes Renata
2026-07-13 02:14:59 +00:00
parent 84634a365e
commit a72f336ad1
227 changed files with 50403 additions and 0 deletions
@@ -0,0 +1,296 @@
"""Benchmark runner scaffold for extraction configuration comparisons.
Provides the framework for running extraction benchmarks across different
configurations. The actual model invocations require the cluster, but
results can be stored and compared locally.
Validates: Requirements 16.2, 16.3
"""
from __future__ import annotations
import json
import time
from datetime import datetime, timezone
from pathlib import Path
from typing import Any
from pydantic import BaseModel, Field
from services.intelligence_pipeline_v3.benchmark.configurations import (
BenchmarkConfig,
)
# ---------------------------------------------------------------------------
# Result Models
# ---------------------------------------------------------------------------
class BenchmarkDocumentResult(BaseModel):
"""Result of running one document through one benchmark configuration."""
document_id: str = Field(description="Identifier of the source document")
raw_output: str | None = Field(
default=None,
description="Raw model output text (before parsing)",
)
parsed_output: dict[str, Any] | None = Field(
default=None,
description="Parsed JSON output if extraction succeeded",
)
schema_valid: bool = Field(
default=False,
description="Whether the output passed JSON Schema validation",
)
retries: int = Field(
default=0,
ge=0,
description="Number of retries needed to get valid output",
)
duration_ms: int = Field(
default=0,
ge=0,
description="Total wall-clock time in milliseconds",
)
input_tokens: int = Field(
default=0,
ge=0,
description="Input tokens consumed",
)
output_tokens: int = Field(
default=0,
ge=0,
description="Output tokens generated",
)
error: str | None = Field(
default=None,
description="Error message if extraction failed",
)
class BenchmarkRun(BaseModel):
"""A complete benchmark run: one configuration applied to multiple documents."""
config_name: str = Field(description="Configuration used for this run")
timestamp: datetime = Field(
default_factory=lambda: datetime.now(timezone.utc),
description="When this run was executed",
)
document_ids: list[str] = Field(
default_factory=list,
description="Documents included in this run",
)
results: list[BenchmarkDocumentResult] = Field(
default_factory=list,
description="Per-document results",
)
@property
def success_count(self) -> int:
"""Number of documents that produced valid output."""
return sum(1 for r in self.results if r.schema_valid and r.error is None)
@property
def failure_count(self) -> int:
"""Number of documents that failed or produced invalid output."""
return len(self.results) - self.success_count
@property
def schema_validity_rate(self) -> float:
"""Fraction of results that passed schema validation."""
if not self.results:
return 0.0
return self.success_count / len(self.results)
@property
def mean_duration_ms(self) -> float:
"""Average duration across all results."""
if not self.results:
return 0.0
return sum(r.duration_ms for r in self.results) / len(self.results)
@property
def total_input_tokens(self) -> int:
"""Total input tokens across all results."""
return sum(r.input_tokens for r in self.results)
@property
def total_output_tokens(self) -> int:
"""Total output tokens across all results."""
return sum(r.output_tokens for r in self.results)
# ---------------------------------------------------------------------------
# Artifact Storage
# ---------------------------------------------------------------------------
_DEFAULT_ARTIFACT_DIR = Path("artifacts/benchmark")
def _ensure_artifact_dir(base: Path) -> Path:
"""Create artifact directory if it does not exist."""
base.mkdir(parents=True, exist_ok=True)
return base
def save_benchmark_run(
run: BenchmarkRun,
artifact_dir: Path | None = None,
) -> Path:
"""Persist a benchmark run as a JSON artifact.
Args:
run: The benchmark run to save.
artifact_dir: Directory to write to. Defaults to artifacts/benchmark/.
Returns:
Path to the written JSON file.
"""
base = artifact_dir or _DEFAULT_ARTIFACT_DIR
_ensure_artifact_dir(base)
ts = run.timestamp.strftime("%Y%m%d_%H%M%S")
filename = f"{run.config_name}_{ts}.json"
path = base / filename
path.write_text(
run.model_dump_json(indent=2),
encoding="utf-8",
)
return path
def load_benchmark_run(path: Path) -> BenchmarkRun:
"""Load a benchmark run from a JSON artifact.
Args:
path: Path to the JSON artifact file.
Returns:
Deserialized BenchmarkRun.
"""
data = json.loads(path.read_text(encoding="utf-8"))
return BenchmarkRun.model_validate(data)
# ---------------------------------------------------------------------------
# Runner
# ---------------------------------------------------------------------------
class BenchmarkRunner:
"""Runs extraction benchmarks using a given configuration.
The runner provides the scaffolding for executing benchmarks.
Actual LLM invocation is delegated to an inference callable.
When no inference callable is provided, results are recorded as
errors (useful for dry-run / configuration testing).
"""
def __init__(
self,
config: BenchmarkConfig,
artifact_dir: Path | None = None,
inference_fn: Any | None = None,
) -> None:
"""Initialize the benchmark runner.
Args:
config: Benchmark configuration to use for all runs.
artifact_dir: Where to store result artifacts.
inference_fn: Optional async callable(document_text, config) -> dict.
If None, documents are recorded as not-run errors.
"""
self.config = config
self.artifact_dir = artifact_dir or _DEFAULT_ARTIFACT_DIR
self._inference_fn = inference_fn
async def run_single_document(
self,
document_id: str,
document_text: str,
json_schema: dict[str, Any] | None = None,
) -> BenchmarkDocumentResult:
"""Run a single document through the configured extraction.
Args:
document_id: Unique document identifier.
document_text: Full document text to extract from.
json_schema: Optional JSON Schema for validation.
Returns:
BenchmarkDocumentResult with extraction outcome.
"""
if self._inference_fn is None:
return BenchmarkDocumentResult(
document_id=document_id,
error="No inference function configured (dry-run mode)",
)
start = time.perf_counter()
try:
result = await self._inference_fn(document_text, self.config)
duration_ms = int((time.perf_counter() - start) * 1000)
raw_output = result.get("raw_output", "")
parsed_output = result.get("parsed_output")
schema_valid = result.get("schema_valid", False)
input_tokens = result.get("input_tokens", 0)
output_tokens = result.get("output_tokens", 0)
retries = result.get("retries", 0)
return BenchmarkDocumentResult(
document_id=document_id,
raw_output=raw_output,
parsed_output=parsed_output,
schema_valid=schema_valid,
retries=retries,
duration_ms=duration_ms,
input_tokens=input_tokens,
output_tokens=output_tokens,
)
except Exception as exc:
duration_ms = int((time.perf_counter() - start) * 1000)
return BenchmarkDocumentResult(
document_id=document_id,
duration_ms=duration_ms,
error=str(exc),
)
async def run_batch(
self,
documents: list[tuple[str, str]],
json_schema: dict[str, Any] | None = None,
save_artifacts: bool = True,
) -> BenchmarkRun:
"""Run a batch of documents through the configured extraction.
Args:
documents: List of (document_id, document_text) tuples.
json_schema: Optional JSON Schema for validation.
save_artifacts: Whether to persist results as JSON artifacts.
Returns:
BenchmarkRun with all document results.
"""
results: list[BenchmarkDocumentResult] = []
document_ids: list[str] = []
for doc_id, doc_text in documents:
document_ids.append(doc_id)
result = await self.run_single_document(
document_id=doc_id,
document_text=doc_text,
json_schema=json_schema,
)
results.append(result)
run = BenchmarkRun(
config_name=self.config.config_name,
document_ids=document_ids,
results=results,
)
if save_artifacts:
save_benchmark_run(run, self.artifact_dir)
return run