"""Benchmark configuration definitions for controlled extraction comparisons. Defines the standard configurations used to attribute improvement sources: - BASELINE_CURRENT: Current production settings (temperature 0.7, no schema constraint) - BASELINE_TEMP_ZERO: Same model, temperature 0, no schema constraint - BASELINE_STRICT_SCHEMA: Same model, temperature 0, strict JSON Schema Validates: Requirements 16.2, 16.3, 16.5 """ from __future__ import annotations from enum import Enum from typing import Any from pydantic import BaseModel, ConfigDict, Field class StructuredOutputMode(str, Enum): """Structured output constraint modes for extraction.""" NONE = "none" JSON_OBJECT = "json_object" JSON_SCHEMA = "json_schema" class BenchmarkConfig(BaseModel): """Configuration for a single benchmark extraction run. Captures all parameters that affect extraction behavior so that differences between runs can be attributed to specific settings. """ model_config = ConfigDict(frozen=True) config_name: str = Field( description="Unique identifier for this configuration", ) description: str = Field( description="Human-readable description of what this configuration tests", ) model_name: str = Field( description="Served model name (e.g. 'AxionML/Qwen3.5-9B-NVFP4')", ) temperature: float = Field( ge=0.0, le=2.0, description="Sampling temperature; 0.0 = deterministic", ) max_output_tokens: int = Field( gt=0, description="Maximum tokens in generated output", ) structured_output_mode: StructuredOutputMode = Field( description="How output structure is constrained", ) seed: int | None = Field( default=None, description="Random seed for reproducibility (None = not pinned)", ) additional_params: dict[str, Any] = Field( default_factory=dict, description="Provider-specific extra parameters", ) # --------------------------------------------------------------------------- # Standard Benchmark Configurations # --------------------------------------------------------------------------- # The 9B model currently deployed on the cluster _DEFAULT_MODEL = "AxionML/Qwen3.5-9B-NVFP4" _DEFAULT_MAX_OUTPUT_TOKENS = 2048 BASELINE_CURRENT = BenchmarkConfig( config_name="baseline_current", description=( "Current production settings: temperature 0.7, response_format json_object " "only (schema not enforced on generation), no seed pinning. " "Represents the unchanged request as deployed." ), model_name=_DEFAULT_MODEL, temperature=0.7, max_output_tokens=_DEFAULT_MAX_OUTPUT_TOKENS, structured_output_mode=StructuredOutputMode.JSON_OBJECT, seed=None, additional_params={}, ) BASELINE_TEMP_ZERO = BenchmarkConfig( config_name="baseline_temp_zero", description=( "Same 9B model with temperature set to 0.0 for deterministic generation. " "Still uses json_object mode without strict schema enforcement. " "Isolates the effect of removing sampling stochasticity." ), model_name=_DEFAULT_MODEL, temperature=0.0, max_output_tokens=_DEFAULT_MAX_OUTPUT_TOKENS, structured_output_mode=StructuredOutputMode.JSON_OBJECT, seed=0, additional_params={}, ) BASELINE_STRICT_SCHEMA = BenchmarkConfig( config_name="baseline_strict_schema", description=( "Same 9B model with temperature 0.0 AND strict JSON Schema output " "enforcement via vLLM structured output backend. " "Isolates the combined effect of deterministic generation plus " "grammar-constrained decoding." ), model_name=_DEFAULT_MODEL, temperature=0.0, max_output_tokens=_DEFAULT_MAX_OUTPUT_TOKENS, structured_output_mode=StructuredOutputMode.JSON_SCHEMA, seed=0, additional_params={}, ) # Registry of all standard configurations _STANDARD_CONFIGURATIONS: dict[str, BenchmarkConfig] = { BASELINE_CURRENT.config_name: BASELINE_CURRENT, BASELINE_TEMP_ZERO.config_name: BASELINE_TEMP_ZERO, BASELINE_STRICT_SCHEMA.config_name: BASELINE_STRICT_SCHEMA, } def list_configurations() -> list[BenchmarkConfig]: """Return all registered benchmark configurations. Returns: List of BenchmarkConfig instances in definition order. """ return list(_STANDARD_CONFIGURATIONS.values())