feat: Intelligence Pipeline v3 — full implementation

Multi-stage evidence-grounded inference architecture replacing the
monolithic 9B model extraction pipeline. CPU-first specialist services
handle routine extraction while the 9B vLLM model is preserved for
semantic adjudication of ambiguous cases.

Key components:
- Capability-aware inference gateway (OpenAI-compatible + Ollama)
- Endpoint registry with DB migrations and REST API
- Sentence-aware document segmenter (property tests)
- Deterministic financial parsing with offset integrity
- Symbol resolution with ambiguity detection
- Specialist service (GLiNER2, dynamic batching, K8s deployment)
- Company-specific sentiment (FinBERT, calibration)
- Retrieval-based novelty and duplicate detection
- Confidence calibration pipeline
- Deterministic routing engine (property tests)
- 9B adjudication layer with VRAM gating
- Stock-specific impact model (features, labels, baseline, trained)
- Pipeline orchestrator (state machine, queues, leases, feature flags)
- Bounded parallelism (async workers, semaphore, load shedding)
- Observability (tracing, metrics, alerts)
- Compatibility adapter (v3→v2 golden mapping tests)
- Shadow/canary promotion framework
- Active learning and fine-tuning pipeline

Test results: 1,161 tests pass, ruff lint clean.
All 282 spec tasks completed.
This commit is contained in:
Celes Renata
2026-07-13 02:14:59 +00:00
parent 84634a365e
commit a72f336ad1
227 changed files with 50403 additions and 0 deletions
+89
View File
@@ -0,0 +1,89 @@
"""Normalized error categories for the inference gateway.
Maps provider-specific failures into a protocol-agnostic taxonomy
so that retry logic, alerting, and metrics work uniformly across
Ollama, OpenAI-compatible, and specialist endpoints.
Requirements: 2.1, 2.9
"""
from __future__ import annotations
from enum import Enum
class InferenceErrorCategory(str, Enum):
"""Normalized error categories for inference failures."""
# Network / transport
TIMEOUT = "timeout"
CONNECTION_REFUSED = "connection_refused"
CONNECTION_ERROR = "connection_error"
# Authentication / authorization
AUTH_FAILED = "auth_failed"
FORBIDDEN = "forbidden"
# Rate limiting
RATE_LIMITED = "rate_limited"
# Server errors
SERVER_ERROR = "server_error"
SERVICE_UNAVAILABLE = "service_unavailable"
# Client errors
BAD_REQUEST = "bad_request"
MODEL_NOT_FOUND = "model_not_found"
INVALID_REQUEST = "invalid_request"
# Response problems
INVALID_RESPONSE = "invalid_response"
EMPTY_RESPONSE = "empty_response"
SCHEMA_VIOLATION = "schema_violation"
# Capability / policy
CAPABILITY_UNAVAILABLE = "capability_unavailable"
POLICY_VIOLATION = "policy_violation"
# Ollama-specific
STALL_DETECTED = "stall_detected"
# Unknown
UNKNOWN = "unknown"
@property
def retryable(self) -> bool:
"""Whether this error category should generally be retried."""
return self in _RETRYABLE_CATEGORIES
_RETRYABLE_CATEGORIES = frozenset({
InferenceErrorCategory.TIMEOUT,
InferenceErrorCategory.CONNECTION_ERROR,
InferenceErrorCategory.CONNECTION_REFUSED,
InferenceErrorCategory.SERVER_ERROR,
InferenceErrorCategory.SERVICE_UNAVAILABLE,
InferenceErrorCategory.RATE_LIMITED,
InferenceErrorCategory.STALL_DETECTED,
InferenceErrorCategory.EMPTY_RESPONSE,
})
class InferenceError(Exception):
"""Typed inference error with category and optional provider detail."""
def __init__(
self,
category: InferenceErrorCategory,
message: str = "",
*,
provider_detail: str | None = None,
status_code: int | None = None,
) -> None:
self.category = category
self.provider_detail = provider_detail
self.status_code = status_code
super().__init__(message or category.value)
@property
def retryable(self) -> bool:
return self.category.retryable