feat: Intelligence Pipeline v3 — full implementation
Multi-stage evidence-grounded inference architecture replacing the monolithic 9B model extraction pipeline. CPU-first specialist services handle routine extraction while the 9B vLLM model is preserved for semantic adjudication of ambiguous cases. Key components: - Capability-aware inference gateway (OpenAI-compatible + Ollama) - Endpoint registry with DB migrations and REST API - Sentence-aware document segmenter (property tests) - Deterministic financial parsing with offset integrity - Symbol resolution with ambiguity detection - Specialist service (GLiNER2, dynamic batching, K8s deployment) - Company-specific sentiment (FinBERT, calibration) - Retrieval-based novelty and duplicate detection - Confidence calibration pipeline - Deterministic routing engine (property tests) - 9B adjudication layer with VRAM gating - Stock-specific impact model (features, labels, baseline, trained) - Pipeline orchestrator (state machine, queues, leases, feature flags) - Bounded parallelism (async workers, semaphore, load shedding) - Observability (tracing, metrics, alerts) - Compatibility adapter (v3→v2 golden mapping tests) - Shadow/canary promotion framework - Active learning and fine-tuning pipeline Test results: 1,161 tests pass, ruff lint clean. All 282 spec tasks completed.
This commit is contained in:
@@ -0,0 +1,134 @@
|
||||
"""Benchmark configuration definitions for controlled extraction comparisons.
|
||||
|
||||
Defines the standard configurations used to attribute improvement sources:
|
||||
- BASELINE_CURRENT: Current production settings (temperature 0.7, no schema constraint)
|
||||
- BASELINE_TEMP_ZERO: Same model, temperature 0, no schema constraint
|
||||
- BASELINE_STRICT_SCHEMA: Same model, temperature 0, strict JSON Schema
|
||||
|
||||
Validates: Requirements 16.2, 16.3, 16.5
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from enum import Enum
|
||||
from typing import Any
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field
|
||||
|
||||
|
||||
class StructuredOutputMode(str, Enum):
|
||||
"""Structured output constraint modes for extraction."""
|
||||
|
||||
NONE = "none"
|
||||
JSON_OBJECT = "json_object"
|
||||
JSON_SCHEMA = "json_schema"
|
||||
|
||||
|
||||
class BenchmarkConfig(BaseModel):
|
||||
"""Configuration for a single benchmark extraction run.
|
||||
|
||||
Captures all parameters that affect extraction behavior so that
|
||||
differences between runs can be attributed to specific settings.
|
||||
"""
|
||||
|
||||
model_config = ConfigDict(frozen=True)
|
||||
|
||||
config_name: str = Field(
|
||||
description="Unique identifier for this configuration",
|
||||
)
|
||||
description: str = Field(
|
||||
description="Human-readable description of what this configuration tests",
|
||||
)
|
||||
model_name: str = Field(
|
||||
description="Served model name (e.g. 'AxionML/Qwen3.5-9B-NVFP4')",
|
||||
)
|
||||
temperature: float = Field(
|
||||
ge=0.0,
|
||||
le=2.0,
|
||||
description="Sampling temperature; 0.0 = deterministic",
|
||||
)
|
||||
max_output_tokens: int = Field(
|
||||
gt=0,
|
||||
description="Maximum tokens in generated output",
|
||||
)
|
||||
structured_output_mode: StructuredOutputMode = Field(
|
||||
description="How output structure is constrained",
|
||||
)
|
||||
seed: int | None = Field(
|
||||
default=None,
|
||||
description="Random seed for reproducibility (None = not pinned)",
|
||||
)
|
||||
additional_params: dict[str, Any] = Field(
|
||||
default_factory=dict,
|
||||
description="Provider-specific extra parameters",
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Standard Benchmark Configurations
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# The 9B model currently deployed on the cluster
|
||||
_DEFAULT_MODEL = "AxionML/Qwen3.5-9B-NVFP4"
|
||||
_DEFAULT_MAX_OUTPUT_TOKENS = 2048
|
||||
|
||||
BASELINE_CURRENT = BenchmarkConfig(
|
||||
config_name="baseline_current",
|
||||
description=(
|
||||
"Current production settings: temperature 0.7, response_format json_object "
|
||||
"only (schema not enforced on generation), no seed pinning. "
|
||||
"Represents the unchanged request as deployed."
|
||||
),
|
||||
model_name=_DEFAULT_MODEL,
|
||||
temperature=0.7,
|
||||
max_output_tokens=_DEFAULT_MAX_OUTPUT_TOKENS,
|
||||
structured_output_mode=StructuredOutputMode.JSON_OBJECT,
|
||||
seed=None,
|
||||
additional_params={},
|
||||
)
|
||||
|
||||
BASELINE_TEMP_ZERO = BenchmarkConfig(
|
||||
config_name="baseline_temp_zero",
|
||||
description=(
|
||||
"Same 9B model with temperature set to 0.0 for deterministic generation. "
|
||||
"Still uses json_object mode without strict schema enforcement. "
|
||||
"Isolates the effect of removing sampling stochasticity."
|
||||
),
|
||||
model_name=_DEFAULT_MODEL,
|
||||
temperature=0.0,
|
||||
max_output_tokens=_DEFAULT_MAX_OUTPUT_TOKENS,
|
||||
structured_output_mode=StructuredOutputMode.JSON_OBJECT,
|
||||
seed=0,
|
||||
additional_params={},
|
||||
)
|
||||
|
||||
BASELINE_STRICT_SCHEMA = BenchmarkConfig(
|
||||
config_name="baseline_strict_schema",
|
||||
description=(
|
||||
"Same 9B model with temperature 0.0 AND strict JSON Schema output "
|
||||
"enforcement via vLLM structured output backend. "
|
||||
"Isolates the combined effect of deterministic generation plus "
|
||||
"grammar-constrained decoding."
|
||||
),
|
||||
model_name=_DEFAULT_MODEL,
|
||||
temperature=0.0,
|
||||
max_output_tokens=_DEFAULT_MAX_OUTPUT_TOKENS,
|
||||
structured_output_mode=StructuredOutputMode.JSON_SCHEMA,
|
||||
seed=0,
|
||||
additional_params={},
|
||||
)
|
||||
|
||||
# Registry of all standard configurations
|
||||
_STANDARD_CONFIGURATIONS: dict[str, BenchmarkConfig] = {
|
||||
BASELINE_CURRENT.config_name: BASELINE_CURRENT,
|
||||
BASELINE_TEMP_ZERO.config_name: BASELINE_TEMP_ZERO,
|
||||
BASELINE_STRICT_SCHEMA.config_name: BASELINE_STRICT_SCHEMA,
|
||||
}
|
||||
|
||||
|
||||
def list_configurations() -> list[BenchmarkConfig]:
|
||||
"""Return all registered benchmark configurations.
|
||||
|
||||
Returns:
|
||||
List of BenchmarkConfig instances in definition order.
|
||||
"""
|
||||
return list(_STANDARD_CONFIGURATIONS.values())
|
||||
Reference in New Issue
Block a user