Multi-stage evidence-grounded inference architecture replacing the monolithic 9B model extraction pipeline. CPU-first specialist services handle routine extraction while the 9B vLLM model is preserved for semantic adjudication of ambiguous cases. Key components: - Capability-aware inference gateway (OpenAI-compatible + Ollama) - Endpoint registry with DB migrations and REST API - Sentence-aware document segmenter (property tests) - Deterministic financial parsing with offset integrity - Symbol resolution with ambiguity detection - Specialist service (GLiNER2, dynamic batching, K8s deployment) - Company-specific sentiment (FinBERT, calibration) - Retrieval-based novelty and duplicate detection - Confidence calibration pipeline - Deterministic routing engine (property tests) - 9B adjudication layer with VRAM gating - Stock-specific impact model (features, labels, baseline, trained) - Pipeline orchestrator (state machine, queues, leases, feature flags) - Bounded parallelism (async workers, semaphore, load shedding) - Observability (tracing, metrics, alerts) - Compatibility adapter (v3→v2 golden mapping tests) - Shadow/canary promotion framework - Active learning and fine-tuning pipeline Test results: 1,161 tests pass, ruff lint clean. All 282 spec tasks completed.
407 lines
12 KiB
Python
407 lines
12 KiB
Python
"""Confidence calibrator using isotonic or Platt scaling.
|
|
|
|
Maps confidence feature vectors to calibrated correctness probabilities.
|
|
Supports training on held-out Gold_Corpus data, cross-validation for
|
|
method comparison, and versioned artifact tracking.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
from typing import Literal
|
|
|
|
import numpy as np
|
|
|
|
from services.intelligence_pipeline_v3.confidence.models import (
|
|
CalibrationArtifactMetadata,
|
|
ConfidenceFeatures,
|
|
)
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
DEFAULT_VERSION = "uncalibrated"
|
|
|
|
|
|
class ConfidenceCalibrator:
|
|
"""Calibrates confidence features to correctness probabilities.
|
|
|
|
Supports isotonic regression and Platt (logistic) scaling.
|
|
The calibrator is fitted on labeled Gold_Corpus data where labels
|
|
indicate whether the extraction was correct (True) or not (False).
|
|
|
|
Parameters
|
|
----------
|
|
method
|
|
Calibration method: "isotonic" for non-parametric monotone fit,
|
|
"platt" for logistic regression scaling.
|
|
"""
|
|
|
|
def __init__(self, method: Literal["isotonic", "platt"] = "isotonic") -> None:
|
|
self._method: Literal["isotonic", "platt"] = method
|
|
self._version: str = DEFAULT_VERSION
|
|
self._fitted: bool = False
|
|
self._model: object | None = None
|
|
self._metadata: CalibrationArtifactMetadata | None = None
|
|
self._training_count: int = 0
|
|
|
|
@property
|
|
def method(self) -> str:
|
|
"""Return the calibration method."""
|
|
return self._method
|
|
|
|
@property
|
|
def version(self) -> str:
|
|
"""Return the calibration artifact version."""
|
|
return self._version
|
|
|
|
@property
|
|
def is_fitted(self) -> bool:
|
|
"""Return whether the calibrator has been fitted."""
|
|
return self._fitted
|
|
|
|
@property
|
|
def metadata(self) -> CalibrationArtifactMetadata | None:
|
|
"""Return the artifact metadata if fitted."""
|
|
return self._metadata
|
|
|
|
def fit(
|
|
self,
|
|
features: list[ConfidenceFeatures],
|
|
labels: list[bool],
|
|
method: str | None = None,
|
|
version: str = "v1.0.0",
|
|
training_range: str = "unknown",
|
|
) -> None:
|
|
"""Train the calibrator on labeled feature/correctness pairs.
|
|
|
|
Parameters
|
|
----------
|
|
features
|
|
List of confidence feature vectors from training data.
|
|
labels
|
|
True if the extraction was correct, False otherwise.
|
|
method
|
|
Override method for this fit (isotonic or platt).
|
|
If None, uses the instance default.
|
|
version
|
|
Version string for the resulting artifact.
|
|
training_range
|
|
Description of the training data date range.
|
|
|
|
Raises
|
|
------
|
|
ValueError
|
|
If features and labels have different lengths or are empty.
|
|
"""
|
|
if not features or not labels:
|
|
raise ValueError("features and labels must not be empty")
|
|
if len(features) != len(labels):
|
|
raise ValueError(
|
|
f"features ({len(features)}) and labels ({len(labels)}) must have the same length"
|
|
)
|
|
|
|
if method is not None:
|
|
if method not in ("isotonic", "platt"):
|
|
raise ValueError(f"method must be 'isotonic' or 'platt', got '{method}'")
|
|
self._method = method # type: ignore[assignment]
|
|
|
|
# Convert features to matrix
|
|
X = np.array([f.to_vector() for f in features], dtype=np.float64)
|
|
y = np.array(labels, dtype=np.float64)
|
|
|
|
if self._method == "isotonic":
|
|
self._fit_isotonic(X, y)
|
|
else:
|
|
self._fit_platt(X, y)
|
|
|
|
self._version = version
|
|
self._training_count = len(features)
|
|
self._fitted = True
|
|
|
|
# Compute calibration quality on training data (for metadata)
|
|
predictions = self._predict_batch(X)
|
|
ece = _compute_ece(predictions, y)
|
|
brier = _compute_brier(predictions, y)
|
|
|
|
self._metadata = CalibrationArtifactMetadata(
|
|
version=version,
|
|
method=self._method,
|
|
training_count=len(features),
|
|
training_range=training_range,
|
|
ece=ece,
|
|
brier_score=brier,
|
|
)
|
|
|
|
logger.info(
|
|
"ConfidenceCalibrator fitted: method=%s, n=%d, version=%s, ECE=%.4f, Brier=%.4f",
|
|
self._method,
|
|
len(features),
|
|
version,
|
|
ece,
|
|
brier,
|
|
)
|
|
|
|
def predict(self, features: ConfidenceFeatures) -> float:
|
|
"""Return calibrated probability of extraction correctness.
|
|
|
|
Parameters
|
|
----------
|
|
features
|
|
Confidence feature vector for a single extraction.
|
|
|
|
Returns
|
|
-------
|
|
float
|
|
Calibrated probability in [0, 1].
|
|
"""
|
|
if not self._fitted:
|
|
# Return a neutral default when uncalibrated
|
|
return 0.5
|
|
|
|
X = np.array([features.to_vector()], dtype=np.float64)
|
|
predictions = self._predict_batch(X)
|
|
return float(np.clip(predictions[0], 0.0, 1.0))
|
|
|
|
def predict_batch(self, features_list: list[ConfidenceFeatures]) -> list[float]:
|
|
"""Return calibrated probabilities for a batch of feature vectors.
|
|
|
|
Parameters
|
|
----------
|
|
features_list
|
|
List of confidence feature vectors.
|
|
|
|
Returns
|
|
-------
|
|
list[float]
|
|
Calibrated probabilities in [0, 1].
|
|
"""
|
|
if not self._fitted:
|
|
return [0.5] * len(features_list)
|
|
|
|
X = np.array([f.to_vector() for f in features_list], dtype=np.float64)
|
|
predictions = self._predict_batch(X)
|
|
return [float(np.clip(p, 0.0, 1.0)) for p in predictions]
|
|
|
|
def evaluate(
|
|
self,
|
|
features: list[ConfidenceFeatures],
|
|
labels: list[bool],
|
|
) -> tuple[float, float]:
|
|
"""Evaluate ECE and Brier score on held-out data.
|
|
|
|
Parameters
|
|
----------
|
|
features
|
|
Held-out feature vectors.
|
|
labels
|
|
True correctness labels.
|
|
|
|
Returns
|
|
-------
|
|
tuple[float, float]
|
|
(ECE, Brier_score) on the held-out set.
|
|
"""
|
|
if not features or not labels:
|
|
raise ValueError("features and labels must not be empty")
|
|
if len(features) != len(labels):
|
|
raise ValueError("features and labels must have the same length")
|
|
|
|
X = np.array([f.to_vector() for f in features], dtype=np.float64)
|
|
y = np.array(labels, dtype=np.float64)
|
|
|
|
if self._fitted:
|
|
predictions = self._predict_batch(X)
|
|
else:
|
|
predictions = np.full(len(y), 0.5)
|
|
|
|
ece = _compute_ece(predictions, y)
|
|
brier = _compute_brier(predictions, y)
|
|
return ece, brier
|
|
|
|
def _fit_isotonic(self, X: np.ndarray, y: np.ndarray) -> None:
|
|
"""Fit isotonic regression on aggregated feature scores."""
|
|
from sklearn.isotonic import IsotonicRegression
|
|
|
|
# Aggregate features into a single score for isotonic monotone fit
|
|
aggregated = X.mean(axis=1)
|
|
iso = IsotonicRegression(y_min=0.0, y_max=1.0, out_of_bounds="clip")
|
|
iso.fit(aggregated, y)
|
|
self._model = iso
|
|
|
|
def _fit_platt(self, X: np.ndarray, y: np.ndarray) -> None:
|
|
"""Fit logistic regression (Platt scaling) on the full feature vector."""
|
|
from sklearn.linear_model import LogisticRegression
|
|
|
|
y_int = y.astype(np.int32)
|
|
if len(np.unique(y_int)) < 2:
|
|
# Not enough class diversity — store a dummy model
|
|
self._model = _ConstantPredictor(float(y.mean()))
|
|
return
|
|
|
|
lr = LogisticRegression(solver="lbfgs", max_iter=1000, C=1.0)
|
|
lr.fit(X, y_int)
|
|
self._model = lr
|
|
|
|
def _predict_batch(self, X: np.ndarray) -> np.ndarray:
|
|
"""Internal prediction dispatch."""
|
|
if self._model is None:
|
|
return np.full(X.shape[0], 0.5)
|
|
|
|
if self._method == "isotonic":
|
|
# Isotonic uses aggregated score
|
|
aggregated = X.mean(axis=1)
|
|
return self._model.predict(aggregated) # type: ignore[union-attr]
|
|
else:
|
|
# Platt uses full feature vector
|
|
if isinstance(self._model, _ConstantPredictor):
|
|
return self._model.predict(X)
|
|
return self._model.predict_proba(X)[:, 1] # type: ignore[union-attr]
|
|
|
|
|
|
class _ConstantPredictor:
|
|
"""Fallback predictor when training data has only one class."""
|
|
|
|
def __init__(self, value: float) -> None:
|
|
self._value = value
|
|
|
|
def predict(self, X: np.ndarray) -> np.ndarray:
|
|
return np.full(X.shape[0], self._value)
|
|
|
|
|
|
def _compute_ece(
|
|
predictions: np.ndarray,
|
|
labels: np.ndarray,
|
|
n_bins: int = 10,
|
|
) -> float:
|
|
"""Compute Expected Calibration Error.
|
|
|
|
Partitions predictions into equal-width bins and computes the
|
|
weighted average of |avg_predicted - avg_actual| per bin.
|
|
|
|
Parameters
|
|
----------
|
|
predictions
|
|
Predicted probabilities.
|
|
labels
|
|
True binary labels (0 or 1).
|
|
n_bins
|
|
Number of equal-width bins.
|
|
|
|
Returns
|
|
-------
|
|
float
|
|
ECE value in [0, 1].
|
|
"""
|
|
if len(predictions) == 0:
|
|
return 0.0
|
|
|
|
bin_boundaries = np.linspace(0.0, 1.0, n_bins + 1)
|
|
ece = 0.0
|
|
n = len(predictions)
|
|
|
|
for i in range(n_bins):
|
|
lower = bin_boundaries[i]
|
|
upper = bin_boundaries[i + 1]
|
|
|
|
if i == n_bins - 1:
|
|
# Include right boundary in last bin
|
|
mask = (predictions >= lower) & (predictions <= upper)
|
|
else:
|
|
mask = (predictions >= lower) & (predictions < upper)
|
|
|
|
bin_count = mask.sum()
|
|
if bin_count == 0:
|
|
continue
|
|
|
|
avg_predicted = predictions[mask].mean()
|
|
avg_actual = labels[mask].mean()
|
|
ece += (bin_count / n) * abs(avg_predicted - avg_actual)
|
|
|
|
return float(ece)
|
|
|
|
|
|
def _compute_brier(predictions: np.ndarray, labels: np.ndarray) -> float:
|
|
"""Compute Brier score (mean squared error of probability predictions).
|
|
|
|
Parameters
|
|
----------
|
|
predictions
|
|
Predicted probabilities.
|
|
labels
|
|
True binary labels (0 or 1).
|
|
|
|
Returns
|
|
-------
|
|
float
|
|
Brier score in [0, 1].
|
|
"""
|
|
if len(predictions) == 0:
|
|
return 0.0
|
|
return float(np.mean((predictions - labels) ** 2))
|
|
|
|
|
|
def compare_methods(
|
|
features: list[ConfidenceFeatures],
|
|
labels: list[bool],
|
|
n_folds: int = 5,
|
|
) -> dict[str, dict[str, float]]:
|
|
"""Compare isotonic and Platt methods using k-fold cross-validation.
|
|
|
|
Parameters
|
|
----------
|
|
features
|
|
Full set of training features.
|
|
labels
|
|
Full set of correctness labels.
|
|
n_folds
|
|
Number of cross-validation folds.
|
|
|
|
Returns
|
|
-------
|
|
dict
|
|
Mapping of method name to {"ece": float, "brier": float} averages.
|
|
"""
|
|
if len(features) < n_folds * 2:
|
|
raise ValueError(
|
|
f"Need at least {n_folds * 2} samples for {n_folds}-fold CV, got {len(features)}"
|
|
)
|
|
|
|
results: dict[str, list[tuple[float, float]]] = {
|
|
"isotonic": [],
|
|
"platt": [],
|
|
}
|
|
|
|
indices = np.arange(len(features))
|
|
fold_size = len(features) // n_folds
|
|
|
|
for fold in range(n_folds):
|
|
val_start = fold * fold_size
|
|
val_end = val_start + fold_size if fold < n_folds - 1 else len(features)
|
|
|
|
val_indices = indices[val_start:val_end]
|
|
train_indices = np.concatenate([indices[:val_start], indices[val_end:]])
|
|
|
|
train_features = [features[i] for i in train_indices]
|
|
train_labels = [labels[i] for i in train_indices]
|
|
val_features = [features[i] for i in val_indices]
|
|
val_labels = [labels[i] for i in val_indices]
|
|
|
|
for method_name in ("isotonic", "platt"):
|
|
cal = ConfidenceCalibrator(method=method_name) # type: ignore[arg-type]
|
|
cal.fit(
|
|
train_features,
|
|
train_labels,
|
|
version=f"cv-fold-{fold}",
|
|
training_range="cross-validation",
|
|
)
|
|
ece, brier = cal.evaluate(val_features, val_labels)
|
|
results[method_name].append((ece, brier))
|
|
|
|
return {
|
|
method: {
|
|
"ece": float(np.mean([r[0] for r in scores])),
|
|
"brier": float(np.mean([r[1] for r in scores])),
|
|
}
|
|
for method, scores in results.items()
|
|
}
|