Compare commits
204
Commits
3b49aa2fa2
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c43865cc9c | ||
|
|
987b32d2ff | ||
|
|
e1d1c70d59 | ||
|
|
4f021511fa | ||
|
|
6bde9d2726 | ||
|
|
620a8f0d91 | ||
|
|
f9284ce0a8 | ||
|
|
d4f1b529a5 | ||
|
|
9617b088bd | ||
|
|
c0d7f2f438 | ||
|
|
7229273461 | ||
|
|
d18a63aab5 | ||
|
|
7df45f7e27 | ||
|
|
0d4a812e3d | ||
|
|
2fff553518 | ||
|
|
2d516a725d | ||
|
|
cdad58ae28 | ||
|
|
9645d9c261 | ||
|
|
05e6f21a16 | ||
|
|
d762a2b76f | ||
|
|
46d7699ed0 | ||
|
|
13729bda98 | ||
|
|
b5deeaa0d5 | ||
|
|
0aa72cbce1 | ||
|
|
a72f336ad1 | ||
|
|
84634a365e | ||
|
|
2f966b1caf | ||
|
|
81ed2f0803 | ||
|
|
5bd5c521dc | ||
|
|
9cbb559ac7 | ||
|
|
b13232a21f | ||
|
|
0cd052a0a7 | ||
|
|
a2f4398f8d | ||
|
|
76e06d5951 | ||
|
|
96eabf595b | ||
|
|
430554fe0c | ||
|
|
853b85a2af | ||
|
|
1d104187f4 | ||
|
|
d018c75182 | ||
|
|
ed474c521a | ||
|
|
e8e348d478 | ||
|
|
d9fbd5b660 | ||
|
|
ca712ad4a0 | ||
|
|
a4f51c00e1 | ||
|
|
cf93e2d547 | ||
|
|
88a6c8f9ee | ||
|
|
af7b654fb7 | ||
|
|
a322e00659 | ||
|
|
ecade0dd52 | ||
|
|
3a9894cd03 | ||
|
|
4ac10ab4f6 | ||
|
|
3996ec6ef5 | ||
|
|
d1533ed31d | ||
|
|
02eb8e619f | ||
|
|
a6aceebe18 | ||
|
|
14a9b4fcc1 | ||
|
|
b70304ad6c | ||
|
|
fdb54b77ce | ||
|
|
106b07c0f0 | ||
|
|
5d7aaacc9d | ||
|
|
b4bf0f2361 | ||
|
|
365bc5d4b7 | ||
|
|
cbd0e156e7 | ||
|
|
bf9b7e65dd | ||
|
|
b516d6805f | ||
|
|
a0eaa7d6c8 | ||
|
|
17a044de0f | ||
|
|
cf38e6aabc | ||
|
|
1ef016e652 | ||
|
|
649e039164 | ||
|
|
e5944dd755 | ||
|
|
3cc438abef | ||
|
|
7d2cb93f50 | ||
|
|
00a3b2bd48 | ||
|
|
e1fe935ee4 | ||
|
|
0810699a9c | ||
|
|
69a95c05d7 | ||
|
|
3d40a6ebff | ||
|
|
6b470bab32 | ||
|
|
ee6991576d | ||
|
|
f861eae6e2 | ||
|
|
d92f5c3403 | ||
|
|
c82eba858e | ||
|
|
acb8751e92 | ||
|
|
39ff49dc47 | ||
|
|
ed71a2cd54 | ||
|
|
e174de3416 | ||
|
|
8ee519d47b | ||
|
|
16a1e16184 | ||
|
|
f5efad13a9 | ||
|
|
869630588e | ||
|
|
30733b591d | ||
|
|
456187bdab | ||
|
|
19d602699c | ||
|
|
ba72879945 | ||
|
|
db82b8a025 | ||
|
|
5ed9f001b1 | ||
|
|
18f857e1a3 | ||
|
|
61bcb5aa57 | ||
|
|
f468e30af0 | ||
|
|
7e2343ec2c | ||
|
|
3a8c6f6c80 | ||
|
|
139fbd6342 | ||
|
|
63b62c5e9f | ||
|
|
3acfdfd11f | ||
|
|
ab212db8be | ||
|
|
7102358f51 | ||
|
|
bc077bfcc8 | ||
|
|
376fcb4bb4 | ||
|
|
cc21fd9e8f | ||
|
|
affb65d7f4 | ||
|
|
32535540fe | ||
|
|
751cce0509 | ||
|
|
0732894414 | ||
|
|
d9110d03a6 | ||
|
|
76f6bd5677 | ||
|
|
32d290bea7 | ||
|
|
7fcc8a6c07 | ||
|
|
5d2ffd9163 | ||
|
|
6169efdc89 | ||
|
|
c42f2223d8 | ||
|
|
1f08820f11 | ||
|
|
2f2ea65fb4 | ||
|
|
9a60ce127b | ||
|
|
414f476620 | ||
|
|
a5f2bcde55 | ||
|
|
34ffdad00c | ||
|
|
601b85764b | ||
|
|
51b6f3d34a | ||
|
|
bddaf44ffc | ||
|
|
5209cc522e | ||
|
|
fa18b1a7c2 | ||
|
|
4b254adad2 | ||
|
|
af2e554edd | ||
|
|
facce5dbb5 | ||
|
|
2f2c0d24f6 | ||
|
|
861423c1e3 | ||
|
|
13f863ef30 | ||
|
|
2538da3f1e | ||
|
|
fa4ad6b15a | ||
|
|
5109c85a3e | ||
|
|
9975c2098b | ||
|
|
e976363259 | ||
|
|
b6e2718007 | ||
|
|
cb3eb230d6 | ||
|
|
963a5c462c | ||
|
|
82892b7a3e | ||
|
|
6f54fd07fa | ||
|
|
99b7dcee98 | ||
|
|
4f7358f4e3 | ||
|
|
0665cef7e3 | ||
|
|
48eca672a9 | ||
|
|
f159b20c87 | ||
|
|
97fe2249fe | ||
|
|
951b733ac3 | ||
|
|
531e33b0ce | ||
|
|
24c753f6e6 | ||
|
|
6880f11c26 | ||
|
|
eead4f1381 | ||
|
|
007189c0a5 | ||
|
|
f9ee1532dc | ||
|
|
ac29e62033 | ||
|
|
7eecd71a0d | ||
|
|
bb40a3cb8e | ||
|
|
4e010bc048 | ||
|
|
8c3c1aab43 | ||
|
|
cfcfd655e7 | ||
|
|
aaf8cee927 | ||
|
|
f264e924f0 | ||
|
|
a36702e5f3 | ||
|
|
01d77c153d | ||
|
|
8d227b62f6 | ||
|
|
b38fb24f14 | ||
|
|
5c64043892 | ||
|
|
11c6457559 | ||
|
|
f151747d56 | ||
|
|
49bff9de50 | ||
|
|
27b84fcd2e | ||
|
|
7e8d518946 | ||
|
|
23f2134754 | ||
|
|
4954318f7b | ||
|
|
3b22f5e1fc | ||
|
|
c188677330 | ||
|
|
58613955e4 | ||
|
|
b1770f37df | ||
|
|
2e4a9b1e08 | ||
|
|
416206e37b | ||
|
|
0a009cdc99 | ||
|
|
2ab52afc73 | ||
|
|
1aae36382c | ||
|
|
98bbec9b8d | ||
|
|
24db0e97f6 | ||
|
|
226d799eb2 | ||
|
|
e360b66c3e | ||
|
|
0437943863 | ||
|
|
f7ae34ef3b | ||
|
|
4bee7a7874 | ||
|
|
5cf60be76d | ||
|
|
6909ac5e50 | ||
|
|
117b693b19 | ||
|
|
63e4fb96ea | ||
|
|
e517cea081 | ||
|
|
88ad1e8d99 | ||
|
|
f251c53f92 |
@@ -0,0 +1,148 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* Minimal MCP server for OpenAI chat completions.
|
||||
* Accepts ANY model string (gpt-5.2, gpt-5.4, etc.) — no hardcoded enum.
|
||||
* Communicates over stdio using JSON-RPC (MCP protocol).
|
||||
*/
|
||||
|
||||
import { createInterface } from "readline";
|
||||
|
||||
const OPENAI_API_KEY = process.env.OPENAI_API_KEY;
|
||||
if (!OPENAI_API_KEY) {
|
||||
process.stderr.write("ERROR: OPENAI_API_KEY environment variable is required\n");
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const SERVER_INFO = {
|
||||
name: "openai-chat",
|
||||
version: "1.0.0",
|
||||
};
|
||||
|
||||
const TOOLS = [
|
||||
{
|
||||
name: "openai_chat",
|
||||
description:
|
||||
"Send messages to OpenAI chat completions API. Supports all OpenAI models including GPT-5.x series.",
|
||||
inputSchema: {
|
||||
type: "object",
|
||||
properties: {
|
||||
model: {
|
||||
type: "string",
|
||||
description:
|
||||
"OpenAI model name (e.g. gpt-5.2, gpt-5.4, gpt-4o, etc.)",
|
||||
default: "gpt-5.2",
|
||||
},
|
||||
messages: {
|
||||
type: "array",
|
||||
description: "Array of chat messages",
|
||||
items: {
|
||||
type: "object",
|
||||
properties: {
|
||||
role: {
|
||||
type: "string",
|
||||
enum: ["system", "user", "assistant"],
|
||||
},
|
||||
content: { type: "string" },
|
||||
},
|
||||
required: ["role", "content"],
|
||||
},
|
||||
},
|
||||
},
|
||||
required: ["messages"],
|
||||
},
|
||||
},
|
||||
];
|
||||
|
||||
async function callOpenAI(model, messages) {
|
||||
const resp = await fetch("https://api.openai.com/v1/chat/completions", {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
Authorization: `Bearer ${OPENAI_API_KEY}`,
|
||||
},
|
||||
body: JSON.stringify({ model, messages }),
|
||||
});
|
||||
|
||||
if (!resp.ok) {
|
||||
const errText = await resp.text();
|
||||
throw new Error(`OpenAI API error ${resp.status}: ${errText}`);
|
||||
}
|
||||
|
||||
const data = await resp.json();
|
||||
return data.choices?.[0]?.message?.content ?? "(no response)";
|
||||
}
|
||||
|
||||
function jsonRpcResponse(id, result) {
|
||||
return JSON.stringify({ jsonrpc: "2.0", id, result });
|
||||
}
|
||||
|
||||
function jsonRpcError(id, code, message) {
|
||||
return JSON.stringify({ jsonrpc: "2.0", id, error: { code, message } });
|
||||
}
|
||||
|
||||
async function handleRequest(req) {
|
||||
const { id, method, params } = req;
|
||||
|
||||
switch (method) {
|
||||
case "initialize":
|
||||
return jsonRpcResponse(id, {
|
||||
protocolVersion: "2024-11-05",
|
||||
capabilities: { tools: {} },
|
||||
serverInfo: SERVER_INFO,
|
||||
});
|
||||
|
||||
case "notifications/initialized":
|
||||
return null; // no response needed for notifications
|
||||
|
||||
case "tools/list":
|
||||
return jsonRpcResponse(id, { tools: TOOLS });
|
||||
|
||||
case "tools/call": {
|
||||
const toolName = params?.name;
|
||||
if (toolName !== "openai_chat") {
|
||||
return jsonRpcError(id, -32602, `Unknown tool: ${toolName}`);
|
||||
}
|
||||
const args = params?.arguments ?? {};
|
||||
const model = args.model || "gpt-5.2";
|
||||
const messages = args.messages || [];
|
||||
|
||||
if (!messages.length) {
|
||||
return jsonRpcError(id, -32602, "messages array is required");
|
||||
}
|
||||
|
||||
try {
|
||||
const content = await callOpenAI(model, messages);
|
||||
return jsonRpcResponse(id, {
|
||||
content: [{ type: "text", text: content }],
|
||||
});
|
||||
} catch (err) {
|
||||
return jsonRpcResponse(id, {
|
||||
content: [{ type: "text", text: `Error: ${err.message}` }],
|
||||
isError: true,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
case "ping":
|
||||
return jsonRpcResponse(id, {});
|
||||
|
||||
default:
|
||||
if (method?.startsWith("notifications/")) return null;
|
||||
return jsonRpcError(id, -32601, `Method not found: ${method}`);
|
||||
}
|
||||
}
|
||||
|
||||
// stdio transport
|
||||
const rl = createInterface({ input: process.stdin });
|
||||
|
||||
rl.on("line", async (line) => {
|
||||
try {
|
||||
const req = JSON.parse(line);
|
||||
const resp = await handleRequest(req);
|
||||
if (resp) {
|
||||
process.stdout.write(resp + "\n");
|
||||
}
|
||||
} catch (err) {
|
||||
process.stderr.write(`Parse error: ${err.message}\n`);
|
||||
}
|
||||
});
|
||||
@@ -0,0 +1 @@
|
||||
{"specId": "e433350c-baf0-4f4f-a30e-3724f6654090", "workflowType": "requirements-first", "specType": "feature"}
|
||||
@@ -0,0 +1,377 @@
|
||||
# Design Document: Comprehensive Quality & Documentation
|
||||
|
||||
## Overview
|
||||
|
||||
This design covers three pillars for the Stonks Oracle platform:
|
||||
|
||||
1. **Test Coverage** — Close unit test gaps in the scheduler and ingestion services, fix pre-existing test failures in the extractor module, and achieve a fully green test suite (Requirements 1–4).
|
||||
2. **Docker Deployment** — Extend `docker-compose.yml` to include all 13 application services plus the frontend, enabling full-platform local development without Kubernetes (Requirement 5).
|
||||
3. **Documentation** — Produce comprehensive documentation covering per-service features, API references, Helm chart configuration, Docker deployment, three Mermaid architecture diagrams, AI agent building, backup/restore, observability, and README resource links (Requirements 6–16).
|
||||
|
||||
### Design Rationale
|
||||
|
||||
The platform has mature production code across 13 services but uneven test coverage and documentation. The scheduler and ingestion services lack dedicated unit tests — their logic is only exercised through integration tests. Four extractor-related test files have pre-existing failures that block CI. Documentation exists only as a local dev setup guide, a pipeline overview, and a runbook. This initiative fills those gaps systematically.
|
||||
|
||||
The approach prioritizes:
|
||||
- **Test isolation**: Mock all external dependencies (PostgreSQL, Redis, MinIO, Ollama) so unit tests run fast and deterministically.
|
||||
- **Documentation from source**: Generate API references by inspecting actual FastAPI route definitions, Helm values from `values.yaml`, and metrics from `services/shared/metrics.py`.
|
||||
- **Docker parity with Kubernetes**: Mirror the Helm chart's service definitions in Docker Compose so both deployment modes stay in sync.
|
||||
|
||||
## Architecture
|
||||
|
||||
The work does not change the platform's runtime architecture. It adds:
|
||||
|
||||
1. **New test files** in `tests/` for scheduler and ingestion unit tests.
|
||||
2. **Fixes** to existing test files and/or production code to resolve failures.
|
||||
3. **New service definitions** in `docker-compose.yml` using the existing `docker/Dockerfile` with `SERVICE_CMD` build args.
|
||||
4. **New documentation files** in `docs/` organized by topic.
|
||||
5. **Updated `README.md`** with a documentation index and Mermaid diagram.
|
||||
|
||||
```mermaid
|
||||
graph TD
|
||||
subgraph "Test Coverage (Reqs 1-4)"
|
||||
T1[tests/test_scheduler_unit.py]
|
||||
T2[tests/test_ingestion_unit.py]
|
||||
T3[Fix test_extractor_prompts.py]
|
||||
T4[Fix test_extractor_schemas.py]
|
||||
T5[Fix test_ollama_client.py]
|
||||
T6[Fix test_filings_adapter.py]
|
||||
end
|
||||
|
||||
subgraph "Docker (Req 5)"
|
||||
D1[docker-compose.yml<br/>+ 13 app services + frontend]
|
||||
end
|
||||
|
||||
subgraph "Documentation (Reqs 6-16)"
|
||||
DOC1[docs/services.md]
|
||||
DOC2[docs/api-reference.md]
|
||||
DOC3[docs/helm-reference.md]
|
||||
DOC4[docs/docker-deployment.md]
|
||||
DOC5[docs/architecture-kubernetes.md]
|
||||
DOC6[docs/architecture-docker-compose.md]
|
||||
DOC7[docs/architecture-data-pipeline.md]
|
||||
DOC8[docs/ai-agents.md]
|
||||
DOC9[docs/backup-restore.md]
|
||||
DOC10[docs/observability.md]
|
||||
DOC11[README.md update]
|
||||
end
|
||||
```
|
||||
|
||||
## Components and Interfaces
|
||||
|
||||
### 1. Scheduler Unit Tests (Requirement 1)
|
||||
|
||||
**Target module**: `services/scheduler/app.py`
|
||||
|
||||
**Functions to test in isolation**:
|
||||
- `get_cadence_for_source(source_type, config)` — Returns polling interval from config or defaults.
|
||||
- `compute_backoff(retry_count)` — Exponential backoff with cap.
|
||||
- `is_source_due(...)` — Core scheduling logic: determines if a source needs polling based on last run status, timing, retry state.
|
||||
- `build_job_payload(source, aliases, now)` — Constructs the ingestion job dict.
|
||||
- `schedule_cycle(pool, rds)` — Full scheduling pass (mocked DB/Redis).
|
||||
- `check_rate_limit(rds, source_type, now)` — Rate limiting with per-type and global Polygon limits.
|
||||
- `recover_stale_documents(pool, rds)` — Re-enqueue orphaned parsed documents.
|
||||
- `retry_failed_extractions(pool, rds)` — Re-enqueue failed extractions.
|
||||
|
||||
**Mocking strategy**:
|
||||
- `asyncpg.Pool` → `AsyncMock` with `.fetch()`, `.fetchrow()`, `.fetchval()`, `.execute()` returning canned records.
|
||||
- `redis.asyncio.Redis` → `AsyncMock` with `.rpush()`, `.set()`, `.get()`, `.incr()`, `.expire()`, `.decr()`, `.delete()` tracking calls.
|
||||
- Use `unittest.mock.patch` for module-level imports where needed.
|
||||
|
||||
**Test file**: `tests/test_scheduler_unit.py`
|
||||
|
||||
### 2. Ingestion Unit Tests (Requirement 2)
|
||||
|
||||
**Target module**: `services/ingestion/worker.py`
|
||||
|
||||
**Functions to test**:
|
||||
- `process_job(job, pool, rds, minio_client, adapters)` — Main job processing with various adapter outcomes.
|
||||
- Error handling paths: adapter returns `AdapterResult(error=...)`, retry exhaustion, dead-letter routing.
|
||||
- Deduplication: content hash already seen in Redis, cross-source document dedup via `dedupe_items`.
|
||||
|
||||
**Mocking strategy**:
|
||||
- Adapters → `AsyncMock` returning `AdapterResult` with controlled `error`, `items`, `content_hash`, `raw_payload`.
|
||||
- `asyncpg.Pool` → `AsyncMock` for `ingestion_runs` INSERT/UPDATE, `persist_ingestion_items`, `record_retrieval_failure`.
|
||||
- `redis.asyncio.Redis` → `AsyncMock` for dedupe checks, queue pushes, DLQ routing.
|
||||
- `minio.Minio` → `MagicMock` for `upload_raw_artifact`.
|
||||
|
||||
**Test file**: `tests/test_ingestion_unit.py`
|
||||
|
||||
### 3. Extractor Test Fixes (Requirement 3)
|
||||
|
||||
**Target files**:
|
||||
- `tests/test_extractor_prompts.py`
|
||||
- `tests/test_extractor_schemas.py`
|
||||
- `tests/test_ollama_client.py`
|
||||
- `tests/test_filings_adapter.py`
|
||||
|
||||
**Approach**: Run each file individually, diagnose failures, and fix either the test setup (mock configuration, fixture data) or the production code. Preserve original test intent and assertions. If production code changes are needed, add regression tests.
|
||||
|
||||
### 4. Full Test Suite Green (Requirement 4)
|
||||
|
||||
**Verification**: Run `pytest tests/ -x --tb=short -q` and `ruff check services/` after all fixes. All existing `test_pbt_*` files must remain passing. Any production code fix must include a regression test.
|
||||
|
||||
### 5. Docker Compose Application Services (Requirement 5)
|
||||
|
||||
**Current state**: `docker-compose.yml` defines 7 infrastructure services (postgres, redis, minio, minio-init, ollama, trino, hive-metastore, superset).
|
||||
|
||||
**Addition**: 14 new service definitions (13 app services + frontend dashboard):
|
||||
|
||||
| Service | Image Build | Command | Port | Depends On |
|
||||
|---------|------------|---------|------|------------|
|
||||
| scheduler | `docker/Dockerfile.scheduler` | `python -m services.scheduler.app` | — | postgres, redis |
|
||||
| symbol-registry | `docker/Dockerfile` | `uvicorn services.symbol_registry.app:app --host 0.0.0.0 --port 8000` | 8001:8000 | postgres |
|
||||
| ingestion | `docker/Dockerfile` | `python -m services.ingestion.worker` | — | postgres, redis, minio |
|
||||
| parser | `docker/Dockerfile` | `python -m services.parser.worker` | — | postgres, redis |
|
||||
| extractor | `docker/Dockerfile` | `python -m services.extractor.main` | — | postgres, redis, ollama |
|
||||
| aggregation | `docker/Dockerfile` | `python -m services.aggregation.main` | — | postgres, redis |
|
||||
| recommendation | `docker/Dockerfile` | `python -m services.recommendation.main` | — | postgres, redis |
|
||||
| trading-engine | `docker/Dockerfile` | `uvicorn services.trading.app:app --host 0.0.0.0 --port 8000` | 8002:8000 | postgres, redis |
|
||||
| risk-engine | `docker/Dockerfile` | `uvicorn services.risk.app:app --host 0.0.0.0 --port 8000` | 8003:8000 | postgres |
|
||||
| broker-adapter | `docker/Dockerfile` | `python -m services.adapters.broker_service` | — | postgres, redis |
|
||||
| lake-publisher | `docker/Dockerfile` | `python -m services.lake_publisher.jobs` | — | postgres, minio |
|
||||
| query-api | `docker/Dockerfile` | `uvicorn services.api.app:app --host 0.0.0.0 --port 8000` | 8004:8000 | postgres, redis, minio |
|
||||
| dashboard | `frontend/Dockerfile` | nginx (built-in) | 3000:8080 | query-api |
|
||||
|
||||
**Common environment block** (shared via `x-app-env` YAML anchor):
|
||||
```yaml
|
||||
POSTGRES_HOST: postgres
|
||||
POSTGRES_PORT: "5432"
|
||||
POSTGRES_DB: stonks
|
||||
POSTGRES_USER: stonks
|
||||
POSTGRES_PASSWORD: stonks_dev
|
||||
REDIS_HOST: redis
|
||||
REDIS_PORT: "6379"
|
||||
MINIO_ENDPOINT: minio:9000
|
||||
MINIO_ACCESS_KEY: minioadmin
|
||||
MINIO_SECRET_KEY: minioadmin
|
||||
OLLAMA_BASE_URL: http://ollama:11434
|
||||
```
|
||||
|
||||
**`.env` file support**: `MARKET_DATA_API_KEY`, `BROKER_API_KEY`, `BROKER_API_SECRET`, `BROKER_BASE_URL` loaded via `env_file: .env` on services that need them (ingestion, broker-adapter, trading-engine).
|
||||
|
||||
**Health checks**: FastAPI services use `curl -f http://localhost:8000/health`; workers use process liveness checks. Infrastructure `depends_on` uses `condition: service_healthy`.
|
||||
|
||||
### 6. Documentation Structure (Requirements 6–16)
|
||||
|
||||
All documentation files are Markdown in `docs/`. The structure:
|
||||
|
||||
```
|
||||
docs/
|
||||
├── services.md # Req 6: Per-service feature docs
|
||||
├── api-reference.md # Req 7: All 4 FastAPI API references
|
||||
├── helm-reference.md # Req 8: Helm chart values reference
|
||||
├── docker-deployment.md # Req 9: Docker deployment guide
|
||||
├── architecture-kubernetes.md # Req 10: K8s Mermaid diagram
|
||||
├── architecture-docker-compose.md # Req 11: Docker Compose Mermaid diagram
|
||||
├── architecture-data-pipeline.md # Req 12: Data pipeline Mermaid diagram
|
||||
├── ai-agents.md # Req 13: AI agent building guide
|
||||
├── backup-restore.md # Req 14: Backup and restore guide
|
||||
├── observability.md # Req 15: Observability & metrics reference
|
||||
├── LOCAL_DEV_SETUP.md # (existing)
|
||||
├── llm-to-trade-pipeline.md # (existing)
|
||||
└── notes/
|
||||
└── runbook.md # (existing)
|
||||
```
|
||||
|
||||
#### 6a. Service Feature Documentation (`docs/services.md`) — Req 6
|
||||
|
||||
For each of the 13 services, document:
|
||||
- **Purpose**: What the service does in the pipeline.
|
||||
- **Entry point**: Module path (e.g., `services.scheduler.app`).
|
||||
- **Configuration**: Environment variables from `services/shared/config.py` relevant to this service.
|
||||
- **Database tables**: Tables read/written by this service.
|
||||
- **Redis queues**: Queue names consumed from and published to (from `services/shared/redis_keys.py`).
|
||||
- **Queue message schema**: JSON structure of messages.
|
||||
- **Signal layers**: For aggregation/recommendation, document the three signal layers (company, macro, competitive), their toggles (`macro_enabled`, `competitive_enabled` in `risk_configs`), and weight configurations.
|
||||
- **Trading engine features**: For the trading service, document position sizing, circuit breakers, reserve pool, risk tier auto-adjustment, backtesting, and notification configuration.
|
||||
|
||||
Queue topology reference (from `redis_keys.py`):
|
||||
| Queue | Producer | Consumer |
|
||||
|-------|----------|----------|
|
||||
| `stonks:queue:ingestion` | scheduler | ingestion |
|
||||
| `stonks:queue:parsing` | ingestion | parser |
|
||||
| `stonks:queue:extraction` | parser | extractor |
|
||||
| `stonks:queue:macro_classification` | parser, scheduler | extractor |
|
||||
| `stonks:queue:aggregation` | extractor | aggregation |
|
||||
| `stonks:queue:recommendation` | aggregation | recommendation |
|
||||
| `stonks:queue:lake_publish` | various | lake-publisher |
|
||||
| `stonks:queue:broker_orders` | trading-engine, trading API | broker-adapter |
|
||||
| `stonks:queue:trading_decisions` | recommendation | trading-engine |
|
||||
|
||||
#### 6b. API Reference (`docs/api-reference.md`) — Req 7
|
||||
|
||||
Document all endpoints from the four FastAPI services by inspecting their route definitions:
|
||||
|
||||
**Query API** (`services/api/app.py`): ~40+ endpoints covering companies, documents, trends, recommendations, evidence drill-down, orders, positions, portfolio, global events, macro impacts, competitive signals, trend projections, agents, dead-letter queues, pipeline control, SQL explorer, saved queries, audit trail, DevOps metrics, and Prometheus metrics.
|
||||
|
||||
**Symbol Registry API** (`services/symbol_registry/app.py`): Companies CRUD, aliases, watchlists, sources, exposure profiles, competitor relationships, competitor inference.
|
||||
|
||||
**Trading API** (`services/trading/app.py`): Health/readiness, engine status, config update, pause/resume, reset, decisions audit, performance metrics/history, backtesting, notifications config/history, override orders, debug state.
|
||||
|
||||
**Risk API** (`services/risk/app.py`): Order evaluation (`POST /evaluate`), health, pending approvals, approval review, approval expiration.
|
||||
|
||||
For each endpoint: method, path, query parameters (type, default, constraints), request body schema, response schema, error codes (4xx/5xx).
|
||||
|
||||
#### 6c. Helm Chart Reference (`docs/helm-reference.md`) — Req 8
|
||||
|
||||
Document from `infra/helm/stonks-oracle/values.yaml`:
|
||||
- `image` block: registry, pullPolicy, tag
|
||||
- `pipelineEnabled`: toggle and effect on worker replicas
|
||||
- `services` block: per-service structure (replicas, image, command, tier, port, secrets, resources, probes)
|
||||
- `config` block: all ConfigMap environment variables with defaults and descriptions
|
||||
- `secrets` block: core, broker, market, gmail, dashboard — injection via `--set` flags
|
||||
- `ingress` block: className, clusterIssuer, host mappings
|
||||
- Analytics stack: trino, hiveMetastore, superset toggles and resources
|
||||
- `networkPolicies.enabled`: default-deny-ingress behavior
|
||||
- Value override files: `values-beta.yaml`, `values-paper.yaml` and their deployment stages
|
||||
|
||||
#### 6d. Docker Deployment Guide (`docs/docker-deployment.md`) — Req 9
|
||||
|
||||
- Complete service inventory with images, ports, volumes, environment variables
|
||||
- `.env` file format with all required/optional variables
|
||||
- Volume mounts and data persistence (pgdata, miniodata, ollama_models, hive_data, superset_data)
|
||||
- Health check configurations
|
||||
- Dockerfile build arguments (`SERVICE_CMD`)
|
||||
- Operational commands: start, stop, restart, logs, scale, reset (`docker compose down -v`)
|
||||
|
||||
#### 6e. Architecture Diagrams (Reqs 10–12)
|
||||
|
||||
**Kubernetes diagram** (`docs/architecture-kubernetes.md`):
|
||||
- `stonks-oracle` namespace with all 13 services grouped by tier (api, processing, trading, orchestration, analytics, frontend)
|
||||
- External cluster services in their namespaces (postgresql-service, redis-service, minio-service, ollama-service)
|
||||
- Traefik ingress routes to external domains
|
||||
- Network policy boundaries
|
||||
- Analytics plane (Trino, Hive Metastore, Superset)
|
||||
- Helm-managed secrets (core, broker, market, gmail) with consumer mapping
|
||||
- Service tier distinction (API with ingress, pipeline workers, trading)
|
||||
|
||||
**Docker Compose diagram** (`docs/architecture-docker-compose.md`):
|
||||
- All infrastructure + application containers
|
||||
- Host port mappings
|
||||
- `depends_on` relationships and health check dependencies
|
||||
- Named volumes and mount points
|
||||
- `.env` file providing API keys
|
||||
- Internal Docker network connectivity
|
||||
|
||||
**Data Pipeline diagram** (`docs/architecture-data-pipeline.md`):
|
||||
- External sources → ingestion → parsing → extraction → aggregation → recommendation → risk → trading → broker
|
||||
- Redis queue topology with queue names
|
||||
- Three signal layers as distinct paths merging at aggregation
|
||||
- Data stores at each stage (MinIO, PostgreSQL, Redis)
|
||||
- Trading engine decision loop
|
||||
- Analytical branch (lake publisher → MinIO/Parquet → Trino → Superset/Dashboard)
|
||||
- External integrations (Ollama, Alpaca, AWS SNS, Gmail)
|
||||
|
||||
#### 6f. AI Agent Guide (`docs/ai-agents.md`) — Req 13
|
||||
|
||||
- Three built-in agents: document-extractor, event-classifier, thesis-rewriter
|
||||
- Per-agent: purpose, input data, output schema, default model, system prompt structure, user prompt template
|
||||
- `ai_agents` table schema and registration (system-seeded vs API-created)
|
||||
- `agent_variants` table: create, activate, deactivate variants for A/B testing
|
||||
- `AgentConfigResolver` module: TTL cache (60s default), COALESCE-based variant override, fallback behavior
|
||||
- Performance logging: `agent_performance_log` table, querying for variant comparison
|
||||
- API endpoints: CRUD on `/api/agents`, test endpoint `/api/agents/{id}/test`
|
||||
- Step-by-step guide: creating a new variant with different model/prompt and activating it
|
||||
|
||||
#### 6g. Backup & Restore Guide (`docs/backup-restore.md`) — Req 14
|
||||
|
||||
Scripts in `scripts/`:
|
||||
- `backup-db.sh`: PostgreSQL dump, CLI args, storage location, retention (keeps last 7)
|
||||
- `restore-db.sh`: PostgreSQL restore, service scale-down/up, data loss implications
|
||||
- `backup-redis.sh`: Redis RDB snapshot backup
|
||||
- `backup.sh`: Combined backup (DB + Redis), `--upload-minio` option
|
||||
- `restore.sh`: Combined restore
|
||||
- Full nuke-and-rebuild procedure (connection termination, DB drop, Redis flush, redeploy, re-seed)
|
||||
- Recommended backup schedules and automation (cron, Kubernetes CronJobs)
|
||||
|
||||
#### 6h. Observability Reference (`docs/observability.md`) — Req 15
|
||||
|
||||
- `/metrics` endpoint on query-api, Prometheus scrape configuration
|
||||
- All metrics from `services/shared/metrics.py`:
|
||||
- **Ingestion**: `stonks_ingestion_jobs_total`, `stonks_ingestion_items_fetched_total`, `stonks_ingestion_items_new_total`, `stonks_ingestion_items_deduped_total`, `stonks_ingestion_errors_total`, `stonks_ingestion_adapter_duration_seconds`
|
||||
- **Parsing**: `stonks_parse_jobs_total`, `stonks_parse_quality_score`, `stonks_parse_low_quality_total`, `stonks_parse_duration_seconds`
|
||||
- **Extraction**: `stonks_extraction_jobs_total`, `stonks_extraction_attempts_total`, `stonks_extraction_retries_total`, `stonks_extraction_duration_seconds`, `stonks_extraction_confidence`, `stonks_extraction_validation_errors_total`, `stonks_extraction_tokens_total`
|
||||
- **Aggregation**: `stonks_aggregation_windows_total`, `stonks_aggregation_signals_total`, `stonks_aggregation_contradiction_score`, `stonks_aggregation_duration_seconds`
|
||||
- **Recommendation**: `stonks_recommendations_total`, `stonks_recommendations_suppressed_total`, `stonks_recommendation_confidence`
|
||||
- **Lake**: `stonks_lake_facts_published_total`, `stonks_lake_publish_duration_seconds`, `stonks_lake_publish_errors_total`, `stonks_lake_publish_bytes_total`
|
||||
- **Trading**: `stonks_orders_submitted_total`, `stonks_orders_rejected_total`, `stonks_orders_filled_total`, `stonks_orders_duplicates_prevented_total`, `stonks_risk_evaluations_total`, `stonks_risk_check_failures_total`, `stonks_positions_synced_total`
|
||||
- **Alerting**: `stonks_alerts_fired_total`, `stonks_alerts_resolved_total`, `stonks_alert_check_duration_seconds`, `stonks_alert_active`
|
||||
- **DLQ**: `stonks_dlq_items_total`, `stonks_dlq_replayed_total`, `stonks_dlq_depth`
|
||||
- **Active**: `stonks_active_jobs`
|
||||
- Alerting module (`services/shared/alerting.py`): 4 alert rules (source_failures, schema_failure_spike, analytical_lag, broker_issues), thresholds, evaluation windows, ConfigMap variables
|
||||
- Structured JSON logging format, trace context (trace_id, span_id)
|
||||
- Dead-letter queue system: queue names (`stonks:dlq:<queue>`), routing, replay tooling
|
||||
- Recommended Prometheus/Grafana queries
|
||||
|
||||
#### 6i. README Update — Req 16
|
||||
|
||||
- Add "Documentation" section with links to all docs
|
||||
- Replace ASCII architecture diagram with Mermaid or link to diagram docs
|
||||
- Preserve all existing content (license, features, tech stack, project structure, deployment)
|
||||
|
||||
## Data Models
|
||||
|
||||
No new database tables or schema changes are introduced. This initiative works with existing tables:
|
||||
|
||||
**Tables referenced in test coverage work**:
|
||||
- `sources`, `companies`, `company_aliases` — scheduler source polling
|
||||
- `ingestion_runs` — scheduler run tracking, ingestion job recording
|
||||
- `documents`, `document_company_mentions` — ingestion persistence, stale document recovery
|
||||
- `document_intelligence`, `document_impact_records` — extractor test fixtures
|
||||
- `model_performance_metrics` — extractor schema validation metrics
|
||||
|
||||
**Tables documented** (not modified):
|
||||
- All tables listed above plus `trend_windows`, `trend_history`, `trend_projections`, `recommendations`, `recommendation_evidence`, `risk_evaluations`, `orders`, `order_events`, `positions`, `portfolio_snapshots`, `trading_decisions`, `circuit_breaker_events`, `reserve_pool_ledger`, `risk_tier_history`, `backtest_runs`, `backtest_trades`, `notifications`, `global_events`, `macro_impact_records`, `exposure_profiles`, `competitor_relationships`, `competitive_signal_records`, `ai_agents`, `agent_variants`, `agent_performance_log`, `audit_events`, `watchlists`, `watchlist_members`, `retention_policies`, `market_snapshots`
|
||||
|
||||
## Error Handling
|
||||
|
||||
### Test Coverage
|
||||
- **Mock failures**: Unit tests must verify that scheduler and ingestion services handle database/Redis connection failures gracefully (no crashes, proper logging).
|
||||
- **Adapter errors**: Ingestion unit tests must verify retry logic with exponential backoff and dead-letter queue routing after retry exhaustion.
|
||||
- **Test fix approach**: When fixing pre-existing failures, prefer fixing test setup over changing production code. If production code changes are needed, add regression tests to prevent re-introduction.
|
||||
|
||||
### Docker Compose
|
||||
- **Health check failures**: Application services use `depends_on` with `condition: service_healthy` to wait for infrastructure. Health checks have `interval`, `timeout`, `retries`, and `start_period` configured.
|
||||
- **Missing `.env` file**: Services that need API keys (ingestion, broker-adapter, trading-engine) will start but log warnings about missing keys. The platform runs in a degraded mode without external API access.
|
||||
- **Build failures**: Each service uses the same base Dockerfile with `SERVICE_CMD` build arg. Build errors are isolated per service.
|
||||
|
||||
### Documentation
|
||||
- **Stale documentation**: Documentation is generated from source code inspection. If the codebase changes after documentation is written, the docs may drift. The README links section serves as a single index to find and update docs.
|
||||
- **Diagram accuracy**: Mermaid diagrams are hand-authored based on current architecture. They should be updated when services are added or removed.
|
||||
|
||||
## Testing Strategy
|
||||
|
||||
### PBT Applicability Assessment
|
||||
|
||||
Property-based testing is **NOT applicable** to this feature. The work consists of:
|
||||
1. **Unit tests for existing services** — These are example-based tests with mocked dependencies, not pure functions with universal properties.
|
||||
2. **Fixing pre-existing test failures** — Bug fixes to existing tests/code.
|
||||
3. **Docker Compose configuration** — Declarative infrastructure configuration.
|
||||
4. **Documentation** — Markdown files with no executable logic.
|
||||
|
||||
None of these involve new pure functions, parsers, serializers, or business logic where PBT would add value. The existing `test_pbt_*` files (22 files covering trading, aggregation, competitive intelligence, etc.) already provide PBT coverage for the platform's core logic and must remain passing.
|
||||
|
||||
### Unit Testing Strategy
|
||||
|
||||
**New test files**:
|
||||
- `tests/test_scheduler_unit.py` — 8+ test cases covering all scheduler pure functions and the `schedule_cycle` orchestration with mocked dependencies.
|
||||
- `tests/test_ingestion_unit.py` — 6+ test cases covering adapter error handling, retry logic, deduplication, and dead-letter queue routing.
|
||||
|
||||
**Test fix files** (existing, to be repaired):
|
||||
- `tests/test_extractor_prompts.py`
|
||||
- `tests/test_extractor_schemas.py`
|
||||
- `tests/test_ollama_client.py`
|
||||
- `tests/test_filings_adapter.py`
|
||||
|
||||
**Test framework**: pytest + pytest-asyncio (already configured in the project).
|
||||
|
||||
**Mocking approach**: `unittest.mock.AsyncMock` for async dependencies, `unittest.mock.MagicMock` for sync dependencies, `unittest.mock.patch` for module-level state.
|
||||
|
||||
### Verification Criteria
|
||||
|
||||
1. `pytest tests/ -x --tb=short -q` → zero failures
|
||||
2. `ruff check services/` → zero violations
|
||||
3. All 22 existing `test_pbt_*` files pass unchanged
|
||||
4. `docker compose config` validates the updated docker-compose.yml
|
||||
5. All documentation files render valid Markdown with working internal links
|
||||
@@ -0,0 +1,236 @@
|
||||
# Requirements Document
|
||||
|
||||
## Introduction
|
||||
|
||||
This initiative covers three pillars for the Stonks Oracle platform: (1) closing unit test coverage gaps across all 13 services, fixing pre-existing test failures, and ensuring every feature has proper automated tests; (2) updating the Docker Compose deployment to include all application services so users can run the full platform without Kubernetes; and (3) producing comprehensive documentation covering every feature, all API endpoints, Helm chart configuration, Docker deployment options, and three Mermaid architecture diagrams (Kubernetes deployment, Docker Compose deployment, and data pipeline), with the README updated to link to all resources.
|
||||
|
||||
## Glossary
|
||||
|
||||
- **Test_Suite**: The collection of pytest unit tests, property-based tests, and integration tests in the `tests/` directory
|
||||
- **Docker_Compose_Stack**: The `docker-compose.yml` file and associated Dockerfiles that define the local development environment
|
||||
- **Helm_Chart**: The Kubernetes deployment configuration at `infra/helm/stonks-oracle/` including `values.yaml`, value overrides, and templates
|
||||
- **Query_API**: The FastAPI REST service at `services/api/app.py` serving analytics and dashboard queries
|
||||
- **Symbol_Registry_API**: The FastAPI REST service at `services/symbol_registry/app.py` managing companies, watchlists, sources, exposure profiles, and competitor relationships
|
||||
- **Trading_API**: The FastAPI REST service at `services/trading/app.py` controlling the autonomous trading engine
|
||||
- **Risk_API**: The FastAPI REST service at `services/risk/app.py` evaluating order risk and managing approval workflows
|
||||
- **Scheduler_Service**: The service at `services/scheduler/` that triggers ingestion cycles on a cadence
|
||||
- **Ingestion_Service**: The queue worker at `services/ingestion/` that fetches market data, news, filings, and macro events
|
||||
- **Extractor_Service**: The queue worker at `services/extractor/` that performs LLM-based intelligence extraction and event classification
|
||||
- **Documentation_Set**: The collection of Markdown files in `docs/` that describe features, APIs, deployment, and architecture
|
||||
- **Architecture_Diagram**: A Mermaid-syntax diagram showing services, data stores, external integrations, and data flow. Three diagrams are produced: Kubernetes deployment, Docker Compose deployment, and data pipeline
|
||||
- **README**: The root `README.md` file serving as the project entry point
|
||||
|
||||
## Requirements
|
||||
|
||||
### Requirement 1: Scheduler Service Unit Tests
|
||||
|
||||
**User Story:** As a developer, I want the scheduler service to have dedicated unit tests, so that scheduling logic, cadence management, and source polling behavior are verified independently of integration tests.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN the Test_Suite is executed for the scheduler module, THE Test_Suite SHALL include unit tests covering job enqueue logic, polling interval calculation, and source due-date evaluation
|
||||
2. WHEN a scheduler unit test is run, THE Test_Suite SHALL mock all external dependencies (PostgreSQL, Redis) and test scheduling logic in isolation
|
||||
3. THE Test_Suite SHALL verify that the scheduler correctly enqueues ingestion jobs for sources whose polling interval has elapsed
|
||||
4. IF a database or Redis connection fails during scheduling, THEN THE Test_Suite SHALL verify that the Scheduler_Service handles the error without crashing
|
||||
|
||||
### Requirement 2: Ingestion Service Unit Tests
|
||||
|
||||
**User Story:** As a developer, I want the ingestion service to have unit tests for adapter error handling and retry logic, so that data fetching resilience is verified beyond integration tests.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN the Test_Suite is executed for the ingestion module, THE Test_Suite SHALL include unit tests covering adapter error handling, retry logic, and deduplication behavior
|
||||
2. WHEN an external API returns an error response, THE Test_Suite SHALL verify that the Ingestion_Service retries according to the configured backoff policy
|
||||
3. WHEN a duplicate content hash is detected, THE Test_Suite SHALL verify that the Ingestion_Service skips re-processing the document
|
||||
4. IF all retry attempts are exhausted, THEN THE Test_Suite SHALL verify that the Ingestion_Service routes the failed job to the dead-letter queue
|
||||
|
||||
### Requirement 3: Extractor Test Failure Fixes
|
||||
|
||||
**User Story:** As a developer, I want the pre-existing test failures in the extractor module to be resolved, so that the full test suite passes cleanly in CI.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN the Test_Suite is executed, THE Test_Suite SHALL pass all tests in `test_extractor_prompts.py` without failures
|
||||
2. WHEN the Test_Suite is executed, THE Test_Suite SHALL pass all tests in `test_extractor_schemas.py` without failures
|
||||
3. WHEN the Test_Suite is executed, THE Test_Suite SHALL pass all tests in `test_ollama_client.py` without failures
|
||||
4. WHEN the Test_Suite is executed, THE Test_Suite SHALL pass all tests in `test_filings_adapter.py` without failures
|
||||
5. THE Test_Suite SHALL maintain the original test intent and assertions when fixing failures, modifying only the code under test or test setup as needed
|
||||
|
||||
### Requirement 4: Full Test Suite Green Status
|
||||
|
||||
**User Story:** As a developer, I want the entire test suite to pass, so that CI builds succeed and regressions are caught immediately.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN `pytest tests/ -x --tb=short -q` is executed, THE Test_Suite SHALL report zero failures across all test files
|
||||
2. WHEN `ruff check services/` is executed, THE Test_Suite SHALL report zero lint violations
|
||||
3. THE Test_Suite SHALL maintain all existing property-based tests (files prefixed `test_pbt_*`) in a passing state
|
||||
4. IF a test fix requires modifying production code, THEN THE Test_Suite SHALL include a regression test that validates the fix
|
||||
|
||||
### Requirement 5: Docker Compose Application Services
|
||||
|
||||
**User Story:** As a developer using Docker instead of Kubernetes, I want docker-compose.yml to include all 13 application services and the frontend, so that I can run the full platform locally with a single `docker compose up`.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Docker_Compose_Stack SHALL define service containers for all 13 application services: scheduler, symbol-registry, ingestion, parser, extractor, aggregation, recommendation, trading-engine, risk-engine, broker-adapter, lake-publisher, query-api, and dashboard
|
||||
2. THE Docker_Compose_Stack SHALL define a frontend container serving the React dashboard via nginx on port 8080
|
||||
3. WHEN `docker compose up` is executed, THE Docker_Compose_Stack SHALL start all infrastructure services (PostgreSQL, Redis, MinIO, Ollama, Trino, Hive Metastore, Superset) before application services using dependency ordering
|
||||
4. WHEN an application service container starts, THE Docker_Compose_Stack SHALL provide health checks that verify the service is ready to accept requests
|
||||
5. THE Docker_Compose_Stack SHALL configure environment variables for each service matching the defaults documented in `docs/LOCAL_DEV_SETUP.md`, with infrastructure hostnames pointing to Docker Compose service names
|
||||
6. THE Docker_Compose_Stack SHALL allow users to provide API keys (MARKET_DATA_API_KEY, BROKER_API_KEY, BROKER_API_SECRET) via a `.env` file without modifying docker-compose.yml
|
||||
7. IF an infrastructure dependency (PostgreSQL, Redis) is not yet healthy, THEN THE Docker_Compose_Stack SHALL delay application service startup using `depends_on` with `condition: service_healthy`
|
||||
|
||||
### Requirement 6: Service Feature Documentation
|
||||
|
||||
**User Story:** As a user or contributor, I want every service documented with its purpose, configuration, queue interactions, and database tables, so that I can understand how each part of the platform works.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Documentation_Set SHALL include a dedicated document for each of the 13 services describing its purpose, inputs, outputs, configuration environment variables, and database tables used
|
||||
2. WHEN a service consumes from or publishes to a Redis queue, THE Documentation_Set SHALL document the queue name, message schema, and processing behavior
|
||||
3. WHEN a service exposes HTTP endpoints, THE Documentation_Set SHALL reference the API documentation for that service
|
||||
4. THE Documentation_Set SHALL describe the three signal layers (company, macro, competitive) with their data flow, toggle mechanisms, and weight configurations
|
||||
5. THE Documentation_Set SHALL document the trading engine features including position sizing, circuit breakers, reserve pool management, risk tier auto-adjustment, backtesting, and notification configuration
|
||||
|
||||
### Requirement 7: API Reference Documentation
|
||||
|
||||
**User Story:** As a developer integrating with Stonks Oracle, I want a complete API reference for all four FastAPI services, so that I know every endpoint, its parameters, request/response schemas, and error codes.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Documentation_Set SHALL include an API reference document covering all endpoints of the Query_API, including path, method, query parameters, response schema, and error codes
|
||||
2. THE Documentation_Set SHALL include an API reference document covering all endpoints of the Symbol_Registry_API, including CRUD operations for companies, aliases, watchlists, sources, exposure profiles, and competitor relationships
|
||||
3. THE Documentation_Set SHALL include an API reference document covering all endpoints of the Trading_API, including engine control, decision audit, performance metrics, backtesting, notifications, and manual override orders
|
||||
4. THE Documentation_Set SHALL include an API reference document covering all endpoints of the Risk_API, including order evaluation, approval workflow, and approval expiration
|
||||
5. WHEN an endpoint accepts query parameters or a request body, THE Documentation_Set SHALL document each parameter with its type, default value, and constraints
|
||||
6. WHEN an endpoint returns an error, THE Documentation_Set SHALL document the HTTP status code and error response format
|
||||
|
||||
### Requirement 8: Helm Chart Configuration Reference
|
||||
|
||||
**User Story:** As an operator deploying Stonks Oracle on Kubernetes, I want a complete reference for all Helm chart values, so that I can configure services, resources, secrets, ingress, network policies, and analytics components.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Documentation_Set SHALL include a Helm configuration reference documenting every key in `values.yaml` with its type, default value, and description
|
||||
2. THE Documentation_Set SHALL document the `services` block structure including replicas, image, command, tier, port, secrets, resources, and probes for each service
|
||||
3. THE Documentation_Set SHALL document the `config` block with all ConfigMap environment variables, their defaults, and what they control
|
||||
4. THE Documentation_Set SHALL document the `secrets` block structure (core, broker, market, gmail, dashboard) and how secrets are injected via `--set` flags during deployment
|
||||
5. THE Documentation_Set SHALL document the `ingress` block including className, clusterIssuer, and host mappings
|
||||
6. THE Documentation_Set SHALL document the analytics stack toggles (trino.enabled, hiveMetastore.enabled, superset.enabled) and their resource configurations
|
||||
7. THE Documentation_Set SHALL document the `pipelineEnabled` toggle and its effect on worker service replicas
|
||||
8. THE Documentation_Set SHALL document the `networkPolicies.enabled` toggle and the default-deny-ingress behavior
|
||||
9. THE Documentation_Set SHALL document the value override files (`values-beta.yaml`, `values-paper.yaml`) and their intended deployment stages
|
||||
|
||||
### Requirement 9: Docker Deployment Guide
|
||||
|
||||
**User Story:** As a developer deploying with Docker Compose, I want a guide explaining all Docker deployment options, environment variables, volume mounts, and operational commands, so that I can run and manage the platform without Kubernetes.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Documentation_Set SHALL include a Docker deployment guide documenting every service defined in docker-compose.yml with its image, ports, volumes, and environment variables
|
||||
2. THE Documentation_Set SHALL document the `.env` file format with all required and optional environment variables, their defaults, and descriptions
|
||||
3. THE Documentation_Set SHALL document volume mounts and data persistence behavior, including how to reset data with `docker compose down -v`
|
||||
4. THE Documentation_Set SHALL document health check configurations and how to verify all services are running
|
||||
5. THE Documentation_Set SHALL document the Dockerfile build arguments (SERVICE_CMD) and how to build custom service images
|
||||
6. THE Documentation_Set SHALL document operational commands for starting, stopping, restarting individual services, viewing logs, and scaling replicas
|
||||
|
||||
### Requirement 10: Kubernetes Architecture Diagram
|
||||
|
||||
**User Story:** As an operator deploying on Kubernetes, I want a Mermaid diagram showing how Stonks Oracle runs in a K8s cluster, so that I can understand the deployment topology, networking, and infrastructure dependencies.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Documentation_Set SHALL include a Mermaid diagram showing all 13 application services deployed as Kubernetes Deployments within the `stonks-oracle` namespace
|
||||
2. THE diagram SHALL show external cluster services (PostgreSQL, Redis, MinIO, Ollama) in their respective namespaces with cross-namespace service references
|
||||
3. THE diagram SHALL show Traefik ingress routes mapping external domains to internal services (stonks.celestium.life → dashboard, stonks-api.celestium.life → query-api, etc.)
|
||||
4. THE diagram SHALL show network policy boundaries indicating which services can communicate with each other
|
||||
5. THE diagram SHALL show the analytics plane (Trino, Hive Metastore, Superset) deployed within the stonks-oracle namespace and their connections to MinIO
|
||||
6. THE diagram SHALL show Helm-managed secrets (core, broker, market, gmail) and which services consume them
|
||||
7. THE diagram SHALL distinguish between API-tier services (with ingress), pipeline-tier workers (queue-driven), and trading-tier services
|
||||
|
||||
### Requirement 11: Docker Compose Architecture Diagram
|
||||
|
||||
**User Story:** As a developer running the platform locally with Docker Compose, I want a Mermaid diagram showing how all containers are wired together, so that I can understand port mappings, volume mounts, and service dependencies.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Documentation_Set SHALL include a Mermaid diagram showing all infrastructure containers (PostgreSQL, Redis, MinIO, Ollama, Trino, Hive Metastore, Superset) and all 13 application service containers as defined in docker-compose.yml
|
||||
2. THE diagram SHALL show host port mappings for externally accessible services (PostgreSQL:5432, Redis:6379, MinIO:9000/9001, Ollama:11434, Trino:8080, Superset:8088, Dashboard:8080, Query API:8000)
|
||||
3. THE diagram SHALL show Docker Compose `depends_on` relationships and health check dependencies between infrastructure and application services
|
||||
4. THE diagram SHALL show named volumes (pgdata, miniodata, ollama_models, hive_data, superset_data) and which containers mount them
|
||||
5. THE diagram SHALL show the `.env` file providing API keys (MARKET_DATA_API_KEY, BROKER_API_KEY, BROKER_API_SECRET) to relevant service containers
|
||||
6. THE diagram SHALL show internal Docker network connectivity between containers using Docker Compose service names as hostnames
|
||||
|
||||
### Requirement 12: Data Pipeline Architecture Diagram
|
||||
|
||||
**User Story:** As a user or contributor, I want a Mermaid diagram showing the end-to-end data pipeline from external data sources through signal processing to trade execution, so that I can understand how data flows through the system.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Documentation_Set SHALL include a Mermaid diagram showing the complete data pipeline from external sources (Polygon.io, news APIs, SEC filings, macro news sources) through ingestion, parsing, extraction, aggregation, recommendation, risk evaluation, and trade execution
|
||||
2. THE diagram SHALL show the Redis queue topology connecting pipeline stages (ingestion → parsing → extraction → aggregation → recommendation → broker) with queue names
|
||||
3. THE diagram SHALL show the three signal layers (company, macro, competitive) as distinct processing paths that merge in the aggregation stage
|
||||
4. THE diagram SHALL show data stores at each stage: MinIO for raw artifacts, PostgreSQL for structured data, Redis for queues and caching
|
||||
5. THE diagram SHALL show the trading engine decision loop: recommendation polling → position sizing → risk evaluation → order execution → broker submission → fill tracking
|
||||
6. THE diagram SHALL show the analytical branch: lake publisher writing Parquet fact tables to MinIO, queryable via Trino, visualized in Superset and the React dashboard
|
||||
7. THE diagram SHALL show external integrations at their connection points: Ollama for LLM extraction, Alpaca for trade execution, AWS SNS and Gmail for notifications
|
||||
|
||||
### Requirement 13: AI Agent Building Guide
|
||||
|
||||
**User Story:** As a user or contributor, I want a guide explaining how each of the three AI agents works — document extractor, event classifier, and thesis rewriter — including how to configure them, create variants, tune prompts, and monitor performance, so that I can customize and extend the AI capabilities of the platform.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Documentation_Set SHALL include an AI agent guide documenting the three built-in agents: `document-extractor` (structured intelligence extraction from news/filings), `event-classifier` (macro/geopolitical event classification), and `thesis-rewriter` (LLM-enhanced recommendation thesis generation)
|
||||
2. FOR each agent, THE Documentation_Set SHALL document its purpose, input data, output schema, default model, system prompt structure, and user prompt template
|
||||
3. THE Documentation_Set SHALL document the `ai_agents` database table schema and how agents are registered (system-seeded vs user-created via the API)
|
||||
4. THE Documentation_Set SHALL document the `agent_variants` table and how to create, activate, and deactivate variants for A/B testing different models or prompts
|
||||
5. THE Documentation_Set SHALL document the `AgentConfigResolver` module including the TTL cache (60-second default), COALESCE-based variant override logic, and fallback behavior when no DB config exists
|
||||
6. THE Documentation_Set SHALL document the agent performance logging system and how to query `agent_performance_log` to compare variant effectiveness
|
||||
7. THE Documentation_Set SHALL document the API endpoints for managing agents (CRUD on `/api/agents`) and testing agent configurations (`/api/agents/{id}/test`)
|
||||
8. THE Documentation_Set SHALL include a step-by-step guide for creating a new agent variant with a different model or prompt and activating it for live traffic
|
||||
|
||||
### Requirement 14: Backup and Restore Guide
|
||||
|
||||
**User Story:** As an operator, I want a guide documenting all backup and restore scripts, their options, storage locations, and retention policies, so that I can protect data and recover from failures.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Documentation_Set SHALL include a backup and restore guide documenting every script in `scripts/` related to backup and restore: `backup-db.sh`, `restore-db.sh`, `backup-redis.sh`, `backup.sh`, and `restore.sh`
|
||||
2. FOR each backup script, THE Documentation_Set SHALL document its CLI arguments, what data it captures, where backups are stored, and retention/pruning behavior (e.g., keeps last 7)
|
||||
3. FOR each restore script, THE Documentation_Set SHALL document its CLI arguments, what it restores, the service scale-down/scale-up procedure it performs, and any data loss implications
|
||||
4. THE Documentation_Set SHALL document the MinIO upload option (`--upload-minio`) for off-host backup storage
|
||||
5. THE Documentation_Set SHALL document the full database nuke and rebuild procedure including connection termination, database drop, Redis flush, redeploy, and re-seed steps
|
||||
6. THE Documentation_Set SHALL document recommended backup schedules and how to automate backups via cron or Kubernetes CronJobs
|
||||
|
||||
### Requirement 15: Observability and Prometheus Metrics Reference
|
||||
|
||||
**User Story:** As an operator, I want a reference documenting all Prometheus metrics exposed by the platform, the alerting rules, and how to monitor pipeline health, so that I can set up dashboards and respond to incidents.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Documentation_Set SHALL include an observability reference documenting the `/metrics` endpoint on the query API and how to configure Prometheus to scrape it
|
||||
2. THE Documentation_Set SHALL document all Prometheus counters, gauges, and histograms emitted by each service, including metric name, labels, and what they measure (e.g., `EXTRACTION_ATTEMPTS`, `EXTRACTION_DURATION`, `AGGREGATION_WINDOWS_COMPUTED`, `AGGREGATION_SIGNALS_PROCESSED`, `RECOMMENDATION_GENERATED`, `RECOMMENDATION_CONFIDENCE`, alerting counters)
|
||||
3. THE Documentation_Set SHALL document the alerting module (`services/shared/alerting.py`) including all alert rules, their thresholds, evaluation windows, and the ConfigMap environment variables that control them (`ALERT_SOURCE_FAILURE_THRESHOLD`, `ALERT_SCHEMA_FAILURE_RATE_THRESHOLD`, `ALERT_LAKE_LAG_THRESHOLD_MINUTES`, `ALERT_BROKER_ERROR_THRESHOLD`, etc.)
|
||||
4. THE Documentation_Set SHALL document the structured JSON logging format, trace context propagation (trace_id, span_id), and how to query logs for debugging pipeline issues
|
||||
5. THE Documentation_Set SHALL document the dead-letter queue system including queue names, how failed jobs are routed there, and how to replay them using the dead-letter tooling
|
||||
6. THE Documentation_Set SHALL document recommended Prometheus/Grafana dashboard configurations or queries for monitoring ingestion throughput, extraction latency, aggregation volume, recommendation generation rate, and trading engine activity
|
||||
|
||||
### Requirement 16: README Resource Links
|
||||
|
||||
**User Story:** As a user landing on the repository, I want the README to link to all documentation resources, so that I can navigate to any guide, reference, or diagram from a single entry point.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN the README is updated, THE README SHALL include a documentation section with links to every document in the Documentation_Set
|
||||
2. THE README SHALL link to the API reference documents for all four FastAPI services
|
||||
3. THE README SHALL link to the Helm chart configuration reference
|
||||
4. THE README SHALL link to the Docker deployment guide
|
||||
5. THE README SHALL link to all three architecture diagram documents (Kubernetes, Docker Compose, and Data Pipeline)
|
||||
6. THE README SHALL link to the per-service feature documentation
|
||||
7. THE README SHALL link to the AI agent building guide
|
||||
8. THE README SHALL link to the backup and restore guide
|
||||
9. THE README SHALL link to the observability and Prometheus metrics reference
|
||||
10. THE README SHALL replace the existing ASCII architecture diagram with the Mermaid architecture diagram or link to it
|
||||
11. THE README SHALL preserve all existing content (license, features, tech stack, project structure, deployment instructions) while adding the new documentation links
|
||||
@@ -0,0 +1,223 @@
|
||||
# Implementation Plan: Comprehensive Quality & Documentation
|
||||
|
||||
## Overview
|
||||
|
||||
This plan implements three pillars for the Stonks Oracle platform: (1) unit test coverage for the scheduler and ingestion services plus fixing pre-existing test failures, (2) extending docker-compose.yml with all 13 application services and the frontend, and (3) producing comprehensive documentation covering services, APIs, Helm configuration, Docker deployment, architecture diagrams, AI agents, backup/restore, observability, and README resource links. Tasks are ordered so tests come first (catch regressions early), then Docker Compose (infrastructure), then documentation (references verified code).
|
||||
|
||||
## Tasks
|
||||
|
||||
- [x] 1. Write scheduler service unit tests
|
||||
- [x] 1.1 Create `tests/test_scheduler_unit.py` with unit tests for scheduler pure functions and orchestration
|
||||
- Import scheduler functions from `services/scheduler/app.py`
|
||||
- Mock `asyncpg.Pool` (`.fetch()`, `.fetchrow()`, `.fetchval()`, `.execute()`) and `redis.asyncio.Redis` (`.rpush()`, `.set()`, `.get()`, `.incr()`, `.expire()`, `.decr()`, `.delete()`)
|
||||
- Write 8+ test cases covering: `get_cadence_for_source`, `compute_backoff`, `is_source_due`, `build_job_payload`, `schedule_cycle` (mocked DB/Redis), `check_rate_limit`, `recover_stale_documents`, `retry_failed_extractions`
|
||||
- Verify error handling: DB/Redis connection failures handled without crashing
|
||||
- Use `pytest-asyncio` for async test functions, `unittest.mock.AsyncMock` and `unittest.mock.patch`
|
||||
- _Requirements: 1.1, 1.2, 1.3, 1.4_
|
||||
|
||||
- [x] 1.2 Write additional edge-case unit tests for scheduler
|
||||
- Test boundary conditions: zero polling interval, max retry count, empty source list
|
||||
- Test rate limiting edge cases: global Polygon limit, per-type limits
|
||||
- _Requirements: 1.3, 1.4_
|
||||
|
||||
- [x] 2. Write ingestion service unit tests
|
||||
- [x] 2.1 Create `tests/test_ingestion_unit.py` with unit tests for ingestion worker
|
||||
- Import ingestion functions from `services/ingestion/worker.py`
|
||||
- Mock adapters as `AsyncMock` returning `AdapterResult` with controlled `error`, `items`, `content_hash`, `raw_payload`
|
||||
- Mock `asyncpg.Pool` for `ingestion_runs` INSERT/UPDATE, `persist_ingestion_items`, `record_retrieval_failure`
|
||||
- Mock `redis.asyncio.Redis` for dedupe checks, queue pushes, DLQ routing
|
||||
- Mock `minio.Minio` for `upload_raw_artifact`
|
||||
- Write 6+ test cases covering: successful job processing, adapter error with retry, retry exhaustion → dead-letter queue, content hash deduplication skip, cross-source dedup via `dedupe_items`, error handling paths
|
||||
- _Requirements: 2.1, 2.2, 2.3, 2.4_
|
||||
|
||||
- [x] 2.2 Write additional edge-case unit tests for ingestion
|
||||
- Test empty adapter response, partial failures, multiple items in single job
|
||||
- _Requirements: 2.1, 2.4_
|
||||
|
||||
- [x] 3. Checkpoint — Verify new unit tests pass
|
||||
- Run `pytest tests/test_scheduler_unit.py tests/test_ingestion_unit.py -x --tb=short -q`
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [x] 4. Fix pre-existing test failures
|
||||
- [x] 4.1 Fix `tests/test_extractor_prompts.py`
|
||||
- Run the file individually to diagnose failures
|
||||
- Fix test setup (mock configuration, fixture data) or production code as needed
|
||||
- Preserve original test intent and assertions
|
||||
- If production code changes are needed, add regression tests
|
||||
- _Requirements: 3.1, 3.5_
|
||||
|
||||
- [x] 4.2 Fix `tests/test_extractor_schemas.py`
|
||||
- Run the file individually to diagnose failures
|
||||
- Fix test setup or production code as needed
|
||||
- Preserve original test intent and assertions
|
||||
- _Requirements: 3.2, 3.5_
|
||||
|
||||
- [x] 4.3 Fix `tests/test_ollama_client.py`
|
||||
- Run the file individually to diagnose failures
|
||||
- Fix test setup or production code as needed
|
||||
- Preserve original test intent and assertions
|
||||
- _Requirements: 3.3, 3.5_
|
||||
|
||||
- [x] 4.4 Fix `tests/test_filings_adapter.py`
|
||||
- Run the file individually to diagnose failures
|
||||
- Fix test setup or production code as needed
|
||||
- Preserve original test intent and assertions
|
||||
- _Requirements: 3.4, 3.5_
|
||||
|
||||
- [x] 5. Checkpoint — Full test suite green
|
||||
- Run `pytest tests/ -x --tb=short -q` and verify zero failures
|
||||
- Run `ruff check services/` and verify zero violations
|
||||
- Verify all `test_pbt_*` files pass unchanged
|
||||
- If any production code was modified, confirm regression tests exist
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
- _Requirements: 4.1, 4.2, 4.3, 4.4_
|
||||
|
||||
- [x] 6. Add application services to docker-compose.yml
|
||||
- [x] 6.1 Add shared environment anchor and all 14 service definitions to `docker-compose.yml`
|
||||
- Define `x-app-env` YAML anchor with common environment variables (POSTGRES_HOST, POSTGRES_PORT, POSTGRES_DB, POSTGRES_USER, POSTGRES_PASSWORD, REDIS_HOST, REDIS_PORT, MINIO_ENDPOINT, MINIO_ACCESS_KEY, MINIO_SECRET_KEY, OLLAMA_BASE_URL)
|
||||
- Add 13 application service definitions: scheduler (using `docker/Dockerfile.scheduler`), symbol-registry, ingestion, parser, extractor, aggregation, recommendation, trading-engine, risk-engine, broker-adapter, lake-publisher, query-api — each using `docker/Dockerfile` with appropriate `SERVICE_CMD` build arg
|
||||
- Add dashboard service using `frontend/Dockerfile` on port 3000:8080
|
||||
- Configure `depends_on` with `condition: service_healthy` for infrastructure dependencies
|
||||
- Add health checks: FastAPI services use `curl -f http://localhost:8000/health`, workers use process liveness
|
||||
- Configure `env_file: .env` on services needing API keys (ingestion, broker-adapter, trading-engine)
|
||||
- Map host ports: symbol-registry:8001, trading-engine:8002, risk-engine:8003, query-api:8004, dashboard:3000
|
||||
- _Requirements: 5.1, 5.2, 5.3, 5.4, 5.5, 5.6, 5.7_
|
||||
|
||||
- [x] 6.2 Validate docker-compose.yml configuration
|
||||
- Run `docker compose config` to verify the updated file parses correctly
|
||||
- _Requirements: 5.1_
|
||||
|
||||
- [x] 7. Checkpoint — Tests and Docker Compose validated
|
||||
- Run `pytest tests/ -x --tb=short -q` to confirm no regressions
|
||||
- Run `docker compose config` to confirm valid YAML
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [x] 8. Write per-service feature documentation
|
||||
- [x] 8.1 Create `docs/services.md` documenting all 13 services
|
||||
- For each service: purpose, entry point module path, configuration environment variables, database tables read/written, Redis queues consumed/published with message schemas
|
||||
- Include queue topology table (queue name → producer → consumer)
|
||||
- Document the three signal layers (company, macro, competitive) with data flow, toggles, and weight configurations
|
||||
- Document trading engine features: position sizing, circuit breakers, reserve pool, risk tier auto-adjustment, backtesting, notifications
|
||||
- Cross-reference API documentation for services with HTTP endpoints
|
||||
- _Requirements: 6.1, 6.2, 6.3, 6.4, 6.5_
|
||||
|
||||
- [x] 9. Write API reference documentation
|
||||
- [x] 9.1 Create `docs/api-reference.md` covering all four FastAPI services
|
||||
- Document all Query API endpoints (~40+): path, method, query parameters (type, default, constraints), request body schema, response schema, error codes
|
||||
- Document all Symbol Registry API endpoints: companies CRUD, aliases, watchlists, sources, exposure profiles, competitor relationships, competitor inference
|
||||
- Document all Trading API endpoints: health/readiness, engine status, config update, pause/resume, reset, decisions audit, performance metrics/history, backtesting, notifications config/history, override orders, debug state
|
||||
- Document all Risk API endpoints: order evaluation (POST /evaluate), health, pending approvals, approval review, approval expiration
|
||||
- Inspect actual route definitions in `services/api/app.py`, `services/symbol_registry/app.py`, `services/trading/app.py`, `services/risk/app.py`
|
||||
- _Requirements: 7.1, 7.2, 7.3, 7.4, 7.5, 7.6_
|
||||
|
||||
- [x] 10. Write Helm chart configuration reference
|
||||
- [x] 10.1 Create `docs/helm-reference.md` documenting all Helm values
|
||||
- Document `image` block: registry, pullPolicy, tag
|
||||
- Document `pipelineEnabled` toggle and effect on worker replicas
|
||||
- Document `services` block: per-service structure (replicas, image, command, tier, port, secrets, resources, probes)
|
||||
- Document `config` block: all ConfigMap environment variables with defaults and descriptions
|
||||
- Document `secrets` block: core, broker, market, gmail, dashboard — injection via `--set` flags
|
||||
- Document `ingress` block: className, clusterIssuer, host mappings
|
||||
- Document analytics stack toggles: trino.enabled, hiveMetastore.enabled, superset.enabled with resources
|
||||
- Document `networkPolicies.enabled` and default-deny-ingress behavior
|
||||
- Document value override files: `values-beta.yaml`, `values-paper.yaml` and deployment stages
|
||||
- _Requirements: 8.1, 8.2, 8.3, 8.4, 8.5, 8.6, 8.7, 8.8, 8.9_
|
||||
|
||||
- [x] 11. Write Docker deployment guide
|
||||
- [x] 11.1 Create `docs/docker-deployment.md` with complete Docker deployment guide
|
||||
- Document every service with image, ports, volumes, environment variables
|
||||
- Document `.env` file format with all required/optional variables, defaults, descriptions
|
||||
- Document volume mounts and data persistence (pgdata, miniodata, ollama_models, hive_data, superset_data), reset with `docker compose down -v`
|
||||
- Document health check configurations and verification commands
|
||||
- Document Dockerfile build arguments (`SERVICE_CMD`) and custom image builds
|
||||
- Document operational commands: start, stop, restart, logs, scale, reset
|
||||
- _Requirements: 9.1, 9.2, 9.3, 9.4, 9.5, 9.6_
|
||||
|
||||
- [x] 12. Checkpoint — Documentation progress check
|
||||
- Verify `docs/services.md`, `docs/api-reference.md`, `docs/helm-reference.md`, `docs/docker-deployment.md` exist and render valid Markdown
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [x] 13. Write architecture diagrams
|
||||
- [x] 13.1 Create `docs/architecture-kubernetes.md` with Kubernetes deployment Mermaid diagram
|
||||
- Show all 13 services in `stonks-oracle` namespace grouped by tier (api, processing, trading, orchestration, analytics, frontend)
|
||||
- Show external cluster services (PostgreSQL, Redis, MinIO, Ollama) in their namespaces
|
||||
- Show Traefik ingress routes to external domains
|
||||
- Show network policy boundaries
|
||||
- Show analytics plane (Trino, Hive Metastore, Superset) and MinIO connections
|
||||
- Show Helm-managed secrets (core, broker, market, gmail) with consumer mapping
|
||||
- Distinguish API-tier (with ingress), pipeline-tier (queue-driven), and trading-tier services
|
||||
- _Requirements: 10.1, 10.2, 10.3, 10.4, 10.5, 10.6, 10.7_
|
||||
|
||||
- [x] 13.2 Create `docs/architecture-docker-compose.md` with Docker Compose Mermaid diagram
|
||||
- Show all infrastructure + application containers
|
||||
- Show host port mappings for externally accessible services
|
||||
- Show `depends_on` relationships and health check dependencies
|
||||
- Show named volumes and mount points
|
||||
- Show `.env` file providing API keys to relevant containers
|
||||
- Show internal Docker network connectivity
|
||||
- _Requirements: 11.1, 11.2, 11.3, 11.4, 11.5, 11.6_
|
||||
|
||||
- [x] 13.3 Create `docs/architecture-data-pipeline.md` with data pipeline Mermaid diagram
|
||||
- Show complete pipeline: external sources → ingestion → parsing → extraction → aggregation → recommendation → risk → trading → broker
|
||||
- Show Redis queue topology with queue names
|
||||
- Show three signal layers as distinct paths merging at aggregation
|
||||
- Show data stores at each stage (MinIO, PostgreSQL, Redis)
|
||||
- Show trading engine decision loop
|
||||
- Show analytical branch: lake publisher → MinIO/Parquet → Trino → Superset/Dashboard
|
||||
- Show external integrations: Ollama, Alpaca, AWS SNS, Gmail
|
||||
- _Requirements: 12.1, 12.2, 12.3, 12.4, 12.5, 12.6, 12.7_
|
||||
|
||||
- [x] 14. Write AI agent building guide
|
||||
- [x] 14.1 Create `docs/ai-agents.md` with AI agent guide
|
||||
- Document three built-in agents: document-extractor, event-classifier, thesis-rewriter — purpose, input data, output schema, default model, system prompt structure, user prompt template
|
||||
- Document `ai_agents` table schema and registration (system-seeded vs API-created)
|
||||
- Document `agent_variants` table: create, activate, deactivate variants for A/B testing
|
||||
- Document `AgentConfigResolver` module: TTL cache (60s), COALESCE-based variant override, fallback behavior
|
||||
- Document performance logging: `agent_performance_log` table, querying for variant comparison
|
||||
- Document API endpoints: CRUD on `/api/agents`, test endpoint `/api/agents/{id}/test`
|
||||
- Include step-by-step guide: creating a new variant with different model/prompt and activating it
|
||||
- _Requirements: 13.1, 13.2, 13.3, 13.4, 13.5, 13.6, 13.7, 13.8_
|
||||
|
||||
- [x] 15. Write backup and restore guide
|
||||
- [x] 15.1 Create `docs/backup-restore.md` with backup and restore guide
|
||||
- Document all scripts in `scripts/`: `backup-db.sh`, `restore-db.sh`, `backup-redis.sh`, `backup.sh`, `restore.sh`
|
||||
- For each backup script: CLI arguments, data captured, storage location, retention/pruning (keeps last 7)
|
||||
- For each restore script: CLI arguments, what it restores, service scale-down/up procedure, data loss implications
|
||||
- Document MinIO upload option (`--upload-minio`) for off-host storage
|
||||
- Document full nuke-and-rebuild procedure: connection termination, DB drop, Redis flush, redeploy, re-seed
|
||||
- Document recommended backup schedules and automation (cron, Kubernetes CronJobs)
|
||||
- _Requirements: 14.1, 14.2, 14.3, 14.4, 14.5, 14.6_
|
||||
|
||||
- [x] 16. Write observability and metrics reference
|
||||
- [x] 16.1 Create `docs/observability.md` with observability reference
|
||||
- Document `/metrics` endpoint on query-api and Prometheus scrape configuration
|
||||
- Document all Prometheus counters, gauges, histograms from `services/shared/metrics.py` — ingestion, parsing, extraction, aggregation, recommendation, lake, trading, alerting, DLQ, active jobs metrics with names, labels, descriptions
|
||||
- Document alerting module (`services/shared/alerting.py`): 4 alert rules, thresholds, evaluation windows, ConfigMap variables
|
||||
- Document structured JSON logging format, trace context (trace_id, span_id), log querying
|
||||
- Document dead-letter queue system: queue names (`stonks:dlq:<queue>`), routing, replay tooling
|
||||
- Document recommended Prometheus/Grafana queries for monitoring
|
||||
- _Requirements: 15.1, 15.2, 15.3, 15.4, 15.5, 15.6_
|
||||
|
||||
- [x] 17. Update README with documentation links
|
||||
- [x] 17.1 Update `README.md` with documentation section and resource links
|
||||
- Add "Documentation" section with links to all docs: services.md, api-reference.md, helm-reference.md, docker-deployment.md, architecture-kubernetes.md, architecture-docker-compose.md, architecture-data-pipeline.md, ai-agents.md, backup-restore.md, observability.md
|
||||
- Replace ASCII architecture diagram with Mermaid diagram or link to architecture diagram docs
|
||||
- Preserve all existing content: license, features, tech stack, project structure, deployment instructions
|
||||
- _Requirements: 16.1, 16.2, 16.3, 16.4, 16.5, 16.6, 16.7, 16.8, 16.9, 16.10, 16.11_
|
||||
|
||||
- [x] 18. Final checkpoint — Full verification
|
||||
- Run `pytest tests/ -x --tb=short -q` — zero failures
|
||||
- Run `ruff check services/` — zero violations
|
||||
- Run `docker compose config` — validates successfully
|
||||
- Verify all `test_pbt_*` files pass unchanged
|
||||
- Verify all documentation files exist in `docs/` and render valid Markdown
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
## Notes
|
||||
|
||||
- Tasks marked with `*` are optional and can be skipped for faster MVP
|
||||
- Each task references specific requirements for traceability
|
||||
- Checkpoints ensure incremental validation
|
||||
- No property-based tests are included — the design assessment confirmed PBT is not applicable to this feature
|
||||
- Existing `test_pbt_*` files (22 files) must remain passing throughout
|
||||
- The implementation language is Python (with Markdown for documentation), matching the existing codebase
|
||||
@@ -0,0 +1 @@
|
||||
{"specId": "d76705a8-fb91-4fce-b59e-c4b3b0dbbd83", "workflowType": "requirements-first", "specType": "feature"}
|
||||
@@ -0,0 +1,723 @@
|
||||
# Design Document — Dual-Pipeline Signal Engine
|
||||
|
||||
## Overview
|
||||
|
||||
The dual-pipeline signal engine is a new service at `services/signal_engine/` that runs as an independent Kubernetes deployment alongside the existing aggregation → recommendation pipeline. It implements a concurrent dual-pipeline architecture where both a heuristic (deterministic scoring) and probabilistic (Bayesian inference) pipeline evaluate the same normalized inputs per ticker per evaluation tick, producing independent BUY/WATCH/SKIP verdicts. A delta analyzer compares the two verdicts, and an output formatter assembles a structured `SignalOutput` contract published to the existing `trading_decisions` Redis queue.
|
||||
|
||||
The engine introduces several new components — Input Normalizer, Signal Library (Fibonacci, MA Stack, RSI, Cup & Handle, Elliott Wave), Multi-Timeframe Engine, Hard Filter Engine, Exit Engine, Delta Analyzer, and Output Formatter — while reusing existing infrastructure: `compute_signal_weight`, `compute_bayesian_posterior`, `classify_regime`, `WeightedSignal`, `BayesianPosterior`, and `RegimeClassification` from `services/aggregation/`.
|
||||
|
||||
The service is toggled via `dual_pipeline_enabled` in the `risk_configs` table (default: false, fail-safe). When disabled, the existing pipeline operates unchanged. When enabled, the signal engine runs alongside the existing pipeline with support for shadow mode (dual-pipeline output persisted but not forwarded to trading).
|
||||
|
||||
### Design Rationale
|
||||
|
||||
- **Separate service, not inline extension**: The signal engine has a fundamentally different evaluation cadence (multi-timeframe technical signals) and data flow (OHLCV bars, not document intelligence). Embedding it in the aggregation worker would couple two distinct concerns.
|
||||
- **Reuse existing math**: The Bayesian posterior, regime classification, and signal weighting functions are battle-tested. The probabilistic pipeline wraps them with regime-based priors and likelihood ratio accumulation rather than reimplementing.
|
||||
- **Concurrent pipelines via asyncio.gather**: Both pipelines share the same `NormalizedInput` reference and run concurrently. If one fails, the other completes normally with the failed pipeline producing a SKIP verdict.
|
||||
- **Signal clustering for correlation penalty**: The Bayesian pipeline groups signals into four clusters (momentum, structure, volatility, fundamentals) and applies exponential decay within each cluster to prevent likelihood ratio stacking inflation from correlated signals.
|
||||
|
||||
---
|
||||
|
||||
## Architecture
|
||||
|
||||
### High-Level Flow
|
||||
|
||||
```mermaid
|
||||
graph TD
|
||||
A[Evaluation Tick<br/>Redis queue: signal_engine] --> B[Input Normalizer]
|
||||
B --> C[Hard Filter Engine]
|
||||
C -->|filtered out| D[SKIP verdict for both pipelines]
|
||||
C -->|passed| E[Signal Library]
|
||||
E --> F[Multi-Timeframe Engine]
|
||||
F --> G{asyncio.gather}
|
||||
G --> H[Heuristic Pipeline]
|
||||
G --> I[Probabilistic Pipeline]
|
||||
H --> J[Delta Analyzer]
|
||||
I --> J
|
||||
J --> K[Output Formatter]
|
||||
K --> L[SignalOutput]
|
||||
L --> M[Redis: trading_decisions queue]
|
||||
L --> N[PostgreSQL: signal_engine_outputs]
|
||||
|
||||
subgraph Exit Path
|
||||
B --> O[Exit Engine]
|
||||
O --> K
|
||||
end
|
||||
```
|
||||
|
||||
### Trigger Mechanism
|
||||
|
||||
The signal engine polls a new Redis queue `stonks:queue:signal_engine`. Evaluation ticks are enqueued by the scheduler service after aggregation completes for a ticker. The queue message contains `{"ticker": "AAPL", "triggered_at": "2024-01-15T10:00:00Z"}`.
|
||||
|
||||
### Integration Points
|
||||
|
||||
| Component | Integration | Direction |
|
||||
|---|---|---|
|
||||
| Scheduler | Enqueues ticks to `signal_engine` queue | Scheduler → Signal Engine |
|
||||
| Market data tables | OHLCV bars, closing prices, returns | Signal Engine reads |
|
||||
| `macro_impact_records` | Macro bias computation | Signal Engine reads |
|
||||
| `trend_windows` | Fundamental/valuation context | Signal Engine reads |
|
||||
| `risk_configs` | Feature flags, thresholds | Signal Engine reads |
|
||||
| `classify_regime()` | Regime classification for priors | Signal Engine calls |
|
||||
| `compute_signal_weight()` | Heuristic signal weighting | Signal Engine calls |
|
||||
| `compute_bayesian_posterior()` | Bayesian accumulation | Signal Engine calls |
|
||||
| Redis `trading_decisions` | SignalOutput publication | Signal Engine → Trading Engine |
|
||||
| `signal_engine_outputs` table | Persistence for audit | Signal Engine writes |
|
||||
| Redis rolling agreement | Delta analyzer metrics | Signal Engine writes |
|
||||
|
||||
---
|
||||
|
||||
## Components and Interfaces
|
||||
|
||||
### Module Structure
|
||||
|
||||
```
|
||||
services/signal_engine/
|
||||
├── __init__.py
|
||||
├── main.py # Entry point: asyncio event loop, queue polling
|
||||
├── worker.py # Top-level orchestrator per evaluation tick
|
||||
├── config.py # SignalEngineConfig, loaded from risk_configs + env
|
||||
├── models.py # All Pydantic models (NormalizedInput, SignalResult, etc.)
|
||||
├── normalizer.py # Input Normalizer — fetches and assembles NormalizedInput
|
||||
├── signals/
|
||||
│ ├── __init__.py
|
||||
│ ├── base.py # SignalEvaluator protocol, SignalResult model
|
||||
│ ├── fibonacci.py # Fibonacci retracement evaluator
|
||||
│ ├── ma_stack.py # Moving average stack evaluator
|
||||
│ ├── rsi.py # RSI evaluator
|
||||
│ ├── cup_handle.py # Cup & Handle pattern detector
|
||||
│ └── elliott_wave.py # Elliott Wave detector
|
||||
├── confluence.py # Multi-Timeframe Confluence Engine
|
||||
├── hard_filter.py # Hard Filter Engine
|
||||
├── heuristic.py # Heuristic Pipeline (Pipeline A)
|
||||
├── probabilistic.py # Probabilistic Pipeline (Pipeline B)
|
||||
├── correlation.py # Signal cluster classification + correlation penalty
|
||||
├── exit_engine.py # Exit Engine — position-level exit management
|
||||
├── delta.py # Delta Analyzer
|
||||
├── formatter.py # Output Formatter
|
||||
└── persistence.py # Database persistence for signal_engine_outputs
|
||||
```
|
||||
|
||||
### Key Function Signatures
|
||||
|
||||
#### `main.py` — Entry Point
|
||||
|
||||
```python
|
||||
async def main() -> None:
|
||||
"""Start the signal engine worker loop.
|
||||
|
||||
Connects to PostgreSQL and Redis, loads config from risk_configs,
|
||||
and polls the signal_engine queue indefinitely.
|
||||
"""
|
||||
```
|
||||
|
||||
#### `worker.py` — Orchestrator
|
||||
|
||||
```python
|
||||
async def evaluate_tick(
|
||||
pool: asyncpg.Pool,
|
||||
redis: redis.asyncio.Redis,
|
||||
ticker: str,
|
||||
config: SignalEngineConfig,
|
||||
) -> SignalOutput | None:
|
||||
"""Run a full evaluation tick for a single ticker.
|
||||
|
||||
1. Normalize inputs
|
||||
2. Evaluate exit conditions for open positions
|
||||
3. Run hard filters
|
||||
4. Evaluate signals across timeframes
|
||||
5. Run both pipelines concurrently
|
||||
6. Compute delta analysis
|
||||
7. Format and publish output
|
||||
|
||||
Returns None if the ticker is hard-filtered or both pipelines fail.
|
||||
"""
|
||||
```
|
||||
|
||||
#### `normalizer.py` — Input Normalizer
|
||||
|
||||
```python
|
||||
async def normalize_input(
|
||||
pool: asyncpg.Pool,
|
||||
ticker: str,
|
||||
config: SignalEngineConfig,
|
||||
) -> NormalizedInput:
|
||||
"""Fetch and assemble all data needed for a single evaluation tick.
|
||||
|
||||
Sources:
|
||||
- OHLCV bars from market_data_bars (M30, H1, H4, D, W, M)
|
||||
- Fundamental metrics from trend_windows + companies
|
||||
- Macro context from macro_impact_records + global_events
|
||||
- Open position state from the trading engine's portfolio
|
||||
|
||||
Missing data sources produce sentinel values (None/empty list)
|
||||
with a logged warning.
|
||||
"""
|
||||
```
|
||||
|
||||
#### `signals/base.py` — Signal Evaluator Protocol
|
||||
|
||||
```python
|
||||
from typing import Protocol
|
||||
|
||||
class SignalEvaluator(Protocol):
|
||||
"""Protocol for all signal evaluators in the Signal Library."""
|
||||
|
||||
def evaluate(
|
||||
self,
|
||||
bars: list[OHLCVBar],
|
||||
timeframe: str,
|
||||
) -> SignalResult | None:
|
||||
"""Evaluate a signal on a single timeframe's bar data.
|
||||
|
||||
Returns None when insufficient data is available.
|
||||
"""
|
||||
...
|
||||
```
|
||||
|
||||
#### `confluence.py` — Multi-Timeframe Engine
|
||||
|
||||
```python
|
||||
def compute_confluence(
|
||||
signal_results: dict[str, dict[str, SignalResult]],
|
||||
weights: dict[str, float],
|
||||
) -> list[ConfluenceSignal]:
|
||||
"""Compute weighted confluence scores across timeframes.
|
||||
|
||||
Args:
|
||||
signal_results: {signal_type: {timeframe: SignalResult}}
|
||||
weights: {timeframe: weight} e.g. {"M30": 0.03, "D": 0.30, ...}
|
||||
|
||||
Returns:
|
||||
List of ConfluenceSignal objects that pass the minimum
|
||||
confluence threshold (≥2 timeframes, ≥1 of D/W/M).
|
||||
"""
|
||||
```
|
||||
|
||||
#### `hard_filter.py` — Hard Filter Engine
|
||||
|
||||
```python
|
||||
def evaluate_hard_filters(
|
||||
normalized: NormalizedInput,
|
||||
config: HardFilterConfig,
|
||||
) -> HardFilterResult:
|
||||
"""Evaluate pre-pipeline hard filters.
|
||||
|
||||
Checks:
|
||||
- macro_bias == -1.0 → SKIP
|
||||
- valuation_score < threshold → SKIP
|
||||
- earnings_proximity_days <= threshold → SKIP
|
||||
|
||||
Returns HardFilterResult with filtered=True/False and all triggered reasons.
|
||||
"""
|
||||
```
|
||||
|
||||
#### `heuristic.py` — Heuristic Pipeline
|
||||
|
||||
```python
|
||||
def run_heuristic_pipeline(
|
||||
normalized: NormalizedInput,
|
||||
confluence_signals: list[ConfluenceSignal],
|
||||
config: HeuristicConfig,
|
||||
) -> HeuristicResult:
|
||||
"""Run the deterministic heuristic pipeline.
|
||||
|
||||
Computes S_total = S_company + S_macro + S_competitive using
|
||||
existing compute_signal_weight() and weighted sentiment averaging.
|
||||
Produces BUY/WATCH/SKIP verdict based on confidence and score thresholds.
|
||||
"""
|
||||
```
|
||||
|
||||
#### `probabilistic.py` — Probabilistic Pipeline
|
||||
|
||||
```python
|
||||
def run_probabilistic_pipeline(
|
||||
normalized: NormalizedInput,
|
||||
confluence_signals: list[ConfluenceSignal],
|
||||
regime: RegimeClassification,
|
||||
config: ProbabilisticConfig,
|
||||
) -> ProbabilisticResult:
|
||||
"""Run the Bayesian probabilistic pipeline.
|
||||
|
||||
1. Initialize regime-based prior (bull=0.58, range=0.50, bear=0.42)
|
||||
2. Compute likelihood ratios per signal with correlation penalty
|
||||
3. Accumulate via log-odds: logit(P_post) = logit(P_prior) + Σ log(LR_i)
|
||||
4. Apply entropy gating
|
||||
5. Compute EV_R = P_up · E[win_R] - (1 - P_up) · 1.0
|
||||
6. Produce BUY/WATCH/SKIP verdict
|
||||
"""
|
||||
```
|
||||
|
||||
#### `correlation.py` — Signal Correlation Penalty
|
||||
|
||||
```python
|
||||
class SignalCluster(str, Enum):
|
||||
MOMENTUM = "momentum" # MA stack, RSI
|
||||
STRUCTURE = "structure" # Fibonacci, Elliott Wave
|
||||
VOLATILITY = "volatility" # ATR-based, Bollinger-derived
|
||||
FUNDAMENTALS = "fundamentals" # valuation, earnings, macro
|
||||
|
||||
def classify_signal(signal_type: str) -> SignalCluster:
|
||||
"""Map a signal type to its correlation cluster."""
|
||||
|
||||
def apply_correlation_penalty(
|
||||
likelihood_ratios: list[LikelihoodRatio],
|
||||
) -> list[LikelihoodRatio]:
|
||||
"""Apply within-cluster decay penalty to correlated signals.
|
||||
|
||||
Within each cluster, signals are ranked by LR magnitude.
|
||||
The strongest contributes at full weight; subsequent signals
|
||||
contribute at 0.5^(n-1) decay.
|
||||
|
||||
Cross-cluster signals are independent (no penalty).
|
||||
"""
|
||||
```
|
||||
|
||||
#### `exit_engine.py` — Exit Engine
|
||||
|
||||
```python
|
||||
def evaluate_exits(
|
||||
positions: list[OpenPositionState],
|
||||
current_prices: dict[str, float],
|
||||
config: ExitConfig,
|
||||
) -> list[ExitSignal]:
|
||||
"""Evaluate exit conditions for all open positions.
|
||||
|
||||
Checks: stop_loss hit, target_1 hit (EXIT_HALF), target_2 hit (EXIT_FULL),
|
||||
trailing stop hit (EXIT_FULL for remaining).
|
||||
|
||||
Trailing stop activates after EXIT_HALF and ratchets upward only.
|
||||
"""
|
||||
```
|
||||
|
||||
#### `delta.py` — Delta Analyzer
|
||||
|
||||
```python
|
||||
async def analyze_delta(
|
||||
heuristic: HeuristicResult,
|
||||
probabilistic: ProbabilisticResult,
|
||||
redis: redis.asyncio.Redis,
|
||||
ticker: str,
|
||||
) -> DeltaResult:
|
||||
"""Compare pipeline verdicts and track agreement metrics.
|
||||
|
||||
Computes agreement flag, confidence delta, disagreement reasons.
|
||||
Updates rolling 100-evaluation agreement rate in Redis.
|
||||
Logs warning when agreement rate drops below 0.50.
|
||||
"""
|
||||
```
|
||||
|
||||
#### `formatter.py` — Output Formatter
|
||||
|
||||
```python
|
||||
def format_output(
|
||||
ticker: str,
|
||||
price: float,
|
||||
heuristic: HeuristicResult,
|
||||
probabilistic: ProbabilisticResult,
|
||||
delta: DeltaResult,
|
||||
exit_signals: list[ExitSignal],
|
||||
config: SignalEngineConfig,
|
||||
) -> SignalOutput:
|
||||
"""Assemble the structured SignalOutput contract.
|
||||
|
||||
Populates trade_plan based on verdict combination:
|
||||
- Both BUY → dual_confirmed, full position sizing
|
||||
- Probabilistic-only BUY → probabilistic_only, 50% position sizing
|
||||
- Heuristic-only BUY → standard position sizing
|
||||
- No BUY → no trade_plan (WATCH/SKIP persisted for analysis)
|
||||
"""
|
||||
|
||||
def signal_output_to_recommendation(output: SignalOutput) -> Recommendation:
|
||||
"""Map a SignalOutput to the existing Recommendation schema.
|
||||
|
||||
Enables the trading engine to consume dual-pipeline outputs
|
||||
without modification to its core evaluate_recommendation logic.
|
||||
"""
|
||||
```
|
||||
|
||||
#### `persistence.py` — Database Persistence
|
||||
|
||||
```python
|
||||
async def persist_signal_output(
|
||||
pool: asyncpg.Pool,
|
||||
output: SignalOutput,
|
||||
) -> None:
|
||||
"""Persist a SignalOutput to the signal_engine_outputs table.
|
||||
|
||||
Logs and continues on database errors (persistence failure
|
||||
does not block signal emission to the trading queue).
|
||||
"""
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Data Models
|
||||
|
||||
All new data models are Pydantic `BaseModel` subclasses defined in `services/signal_engine/models.py`. Existing models (`WeightedSignal`, `BayesianPosterior`, `RegimeClassification`, `TrendSummary`, `Recommendation`, `PositionSizing`) are imported from `services/aggregation/` and `services/shared/schemas.py`.
|
||||
|
||||
### OHLCVBar
|
||||
|
||||
```python
|
||||
class OHLCVBar(BaseModel):
|
||||
"""Single OHLCV bar for a timeframe."""
|
||||
timestamp: datetime
|
||||
open: float
|
||||
high: float
|
||||
low: float
|
||||
close: float
|
||||
volume: float
|
||||
```
|
||||
|
||||
### NormalizedInput
|
||||
|
||||
```python
|
||||
class NormalizedInput(BaseModel):
|
||||
"""Unified input structure consumed by both pipelines."""
|
||||
ticker: str
|
||||
evaluated_at: datetime
|
||||
|
||||
# Multi-timeframe OHLCV bars
|
||||
bars: dict[str, list[OHLCVBar]] # {"M30": [...], "H1": [...], ...}
|
||||
|
||||
# Fundamental metrics
|
||||
valuation_score: float | None = None # [0.0, 1.0]
|
||||
earnings_proximity_days: int | None = None
|
||||
|
||||
# Macro context
|
||||
macro_bias: float = 0.0 # [-1.0, 1.0]
|
||||
|
||||
# Open position state (for exit engine)
|
||||
open_positions: list[OpenPositionState] = Field(default_factory=list)
|
||||
|
||||
# Market data for regime classification
|
||||
closing_prices: list[float] = Field(default_factory=list)
|
||||
returns: list[float] = Field(default_factory=list)
|
||||
|
||||
# Current price (latest close from shortest available timeframe)
|
||||
current_price: float | None = None
|
||||
```
|
||||
|
||||
### OpenPositionState
|
||||
|
||||
```python
|
||||
class OpenPositionState(BaseModel):
|
||||
"""Snapshot of an open position for exit evaluation."""
|
||||
position_id: str
|
||||
ticker: str
|
||||
entry_price: float
|
||||
current_price: float
|
||||
stop_loss: float
|
||||
target_1: float
|
||||
target_2: float
|
||||
trailing_stop: float | None = None
|
||||
partial_exit_done: bool = False
|
||||
atr: float | None = None
|
||||
```
|
||||
|
||||
### SignalResult
|
||||
|
||||
```python
|
||||
class SignalDirection(str, Enum):
|
||||
BULLISH = "bullish"
|
||||
BEARISH = "bearish"
|
||||
NEUTRAL = "neutral"
|
||||
|
||||
class SignalResult(BaseModel):
|
||||
"""Output from a single signal evaluator on a single timeframe."""
|
||||
signal_type: str # e.g. "fibonacci", "ma_stack", "rsi"
|
||||
timeframe: str # e.g. "D", "H4"
|
||||
strength: float = Field(ge=0.0, le=1.0)
|
||||
direction: SignalDirection
|
||||
confidence: float = Field(ge=0.0, le=1.0)
|
||||
metadata: dict = Field(default_factory=dict) # signal-specific details
|
||||
```
|
||||
|
||||
### ConfluenceSignal
|
||||
|
||||
```python
|
||||
class ConfluenceSignal(BaseModel):
|
||||
"""A signal that passed multi-timeframe confluence filtering."""
|
||||
signal_type: str
|
||||
direction: SignalDirection
|
||||
confluence_score: float # weighted sum across timeframes
|
||||
active_timeframes: list[str] # which timeframes triggered
|
||||
per_timeframe: dict[str, float] # {timeframe: strength}
|
||||
```
|
||||
|
||||
### Verdict
|
||||
|
||||
```python
|
||||
class Verdict(str, Enum):
|
||||
BUY = "BUY"
|
||||
WATCH = "WATCH"
|
||||
SKIP = "SKIP"
|
||||
```
|
||||
|
||||
### HeuristicResult
|
||||
|
||||
```python
|
||||
class HeuristicResult(BaseModel):
|
||||
"""Output from the heuristic (deterministic) pipeline."""
|
||||
verdict: Verdict
|
||||
confidence: float = Field(ge=0.0, le=1.0)
|
||||
s_total: float
|
||||
s_company: float
|
||||
s_macro: float
|
||||
s_competitive: float
|
||||
signal_weights: list[dict] = Field(default_factory=list)
|
||||
reasoning: list[str] = Field(default_factory=list)
|
||||
```
|
||||
|
||||
### LikelihoodRatio
|
||||
|
||||
```python
|
||||
class LikelihoodRatio(BaseModel):
|
||||
"""A single signal's likelihood ratio for Bayesian updating."""
|
||||
signal_type: str
|
||||
cluster: str # SignalCluster value
|
||||
lr: float # P(sig|up) / P(sig|down)
|
||||
log_lr: float # log(lr)
|
||||
penalized_log_lr: float # after correlation penalty
|
||||
hit_rate: float
|
||||
strength: float
|
||||
```
|
||||
|
||||
### ProbabilisticResult
|
||||
|
||||
```python
|
||||
class ProbabilisticResult(BaseModel):
|
||||
"""Output from the probabilistic (Bayesian) pipeline."""
|
||||
verdict: Verdict
|
||||
p_up: float = Field(ge=0.0, le=1.0)
|
||||
entropy: float = Field(ge=0.0, le=1.0)
|
||||
ev_r: float
|
||||
prior: float
|
||||
posterior: float
|
||||
likelihood_ratios: list[LikelihoodRatio] = Field(default_factory=list)
|
||||
regime: str
|
||||
reasoning: list[str] = Field(default_factory=list)
|
||||
```
|
||||
|
||||
### DeltaResult
|
||||
|
||||
```python
|
||||
class DeltaResult(BaseModel):
|
||||
"""Output from the delta analyzer comparing both pipelines."""
|
||||
agreement: bool
|
||||
confidence_delta: float
|
||||
heuristic_verdict: str
|
||||
probabilistic_verdict: str
|
||||
disagreement_reasons: list[str] = Field(default_factory=list)
|
||||
rolling_agreement_rate: float | None = None
|
||||
```
|
||||
|
||||
### ExitSignal
|
||||
|
||||
```python
|
||||
class ExitType(str, Enum):
|
||||
EXIT_HALF = "EXIT_HALF"
|
||||
EXIT_FULL = "EXIT_FULL"
|
||||
|
||||
class ExitSignal(BaseModel):
|
||||
"""An exit signal for an open position."""
|
||||
position_id: str
|
||||
ticker: str
|
||||
exit_type: ExitType
|
||||
reason: str # "stop_hit", "target_1_hit", "target_2_hit", "trailing_stop_hit"
|
||||
price: float
|
||||
```
|
||||
|
||||
### TradePlan
|
||||
|
||||
```python
|
||||
class TradePlan(BaseModel):
|
||||
"""Optional trade plan attached to a BUY signal."""
|
||||
entry_price: float
|
||||
stop_loss: float
|
||||
target_1: float
|
||||
target_2: float
|
||||
position_size_pct: float = Field(ge=0.0, le=1.0)
|
||||
max_loss_pct: float = Field(ge=0.0, le=1.0)
|
||||
dual_confirmed: bool = False
|
||||
probabilistic_only: bool = False
|
||||
```
|
||||
|
||||
### SignalOutput
|
||||
|
||||
```python
|
||||
class SignalOutput(BaseModel):
|
||||
"""The structured output contract consumed by the trading engine and audit systems."""
|
||||
output_id: str = Field(default_factory=lambda: str(uuid.uuid4()))
|
||||
ticker: str
|
||||
timestamp: datetime
|
||||
price: float
|
||||
|
||||
# Heuristic pipeline results
|
||||
heuristic_verdict: str
|
||||
heuristic_confidence: float
|
||||
heuristic_s_total: float
|
||||
|
||||
# Probabilistic pipeline results
|
||||
probabilistic_verdict: str
|
||||
probabilistic_p_up: float
|
||||
probabilistic_entropy: float
|
||||
probabilistic_ev_r: float
|
||||
|
||||
# Delta analysis
|
||||
delta_agreement: bool
|
||||
delta_confidence_delta: float
|
||||
delta_reasons: list[str] = Field(default_factory=list)
|
||||
|
||||
# Optional trade plan (populated when at least one pipeline says BUY)
|
||||
trade_plan: TradePlan | None = None
|
||||
|
||||
# Exit signals for open positions
|
||||
exit_signals: list[ExitSignal] = Field(default_factory=list)
|
||||
|
||||
# Full pipeline results for audit (stored as JSONB)
|
||||
heuristic_detail: dict = Field(default_factory=dict)
|
||||
probabilistic_detail: dict = Field(default_factory=dict)
|
||||
|
||||
# Pipeline mode metadata
|
||||
pipeline_mode: str = "dual_pipeline"
|
||||
shadow_mode: bool = False
|
||||
```
|
||||
|
||||
### SignalEngineConfig
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class SignalEngineConfig:
|
||||
"""Configuration loaded from risk_configs + environment."""
|
||||
dual_pipeline_enabled: bool = False
|
||||
heuristic_pipeline_enabled: bool = True
|
||||
probabilistic_pipeline_enabled: bool = True
|
||||
shadow_mode: bool = False
|
||||
|
||||
# Timeframe weights
|
||||
timeframe_weights: dict[str, float] = field(default_factory=lambda: {
|
||||
"M30": 0.03, "H1": 0.07, "H4": 0.15,
|
||||
"D": 0.30, "W": 0.30, "M": 0.15,
|
||||
})
|
||||
|
||||
# Hard filter thresholds
|
||||
hard_filter_valuation_min: float = 0.3
|
||||
hard_filter_earnings_days: int = 5
|
||||
hard_filter_macro_bias_skip: float = -1.0
|
||||
|
||||
# Heuristic verdict thresholds
|
||||
heuristic_buy_confidence: float = 0.70
|
||||
heuristic_buy_s_total: float = 1.2
|
||||
heuristic_buy_valuation_min: float = 0.5
|
||||
heuristic_watch_confidence: float = 0.55
|
||||
|
||||
# Probabilistic verdict thresholds
|
||||
prob_buy_p_up: float = 0.60
|
||||
prob_buy_entropy_max: float = 0.90
|
||||
prob_buy_ev_r_min: float = 1.5
|
||||
prob_buy_valuation_min: float = 0.5
|
||||
prob_watch_p_up: float = 0.55
|
||||
prob_watch_entropy_max: float = 0.95
|
||||
prob_entropy_skip: float = 0.95
|
||||
|
||||
# Regime priors
|
||||
regime_prior_bull: float = 0.58
|
||||
regime_prior_range: float = 0.50
|
||||
regime_prior_bear: float = 0.42
|
||||
|
||||
# Exit engine
|
||||
trailing_stop_atr_multiplier: float = 2.0
|
||||
|
||||
# Polling
|
||||
polling_interval_seconds: int = 30
|
||||
```
|
||||
|
||||
### HardFilterConfig / HeuristicConfig / ProbabilisticConfig / ExitConfig
|
||||
|
||||
These are derived from `SignalEngineConfig` fields for cleaner function signatures — simple `@dataclass` wrappers over the relevant subset of config values.
|
||||
|
||||
---
|
||||
|
||||
### Database Migration (039)
|
||||
|
||||
```sql
|
||||
-- Migration 039: Signal Engine Outputs
|
||||
-- Creates the signal_engine_outputs table for persisting dual-pipeline evaluations.
|
||||
|
||||
CREATE TABLE IF NOT EXISTS signal_engine_outputs (
|
||||
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||
ticker TEXT NOT NULL,
|
||||
evaluated_at TIMESTAMPTZ NOT NULL,
|
||||
price NUMERIC NOT NULL,
|
||||
|
||||
-- Heuristic pipeline
|
||||
heuristic_verdict TEXT NOT NULL,
|
||||
heuristic_confidence NUMERIC NOT NULL,
|
||||
heuristic_s_total NUMERIC NOT NULL,
|
||||
|
||||
-- Probabilistic pipeline
|
||||
probabilistic_verdict TEXT NOT NULL,
|
||||
probabilistic_p_up NUMERIC NOT NULL,
|
||||
probabilistic_entropy NUMERIC NOT NULL,
|
||||
probabilistic_ev_r NUMERIC NOT NULL,
|
||||
|
||||
-- Delta analysis
|
||||
delta_agreement BOOLEAN NOT NULL,
|
||||
delta_confidence_delta NUMERIC NOT NULL,
|
||||
delta_reasons JSONB NOT NULL DEFAULT '[]'::jsonb,
|
||||
|
||||
-- Trade plan (null when no BUY verdict)
|
||||
trade_plan JSONB,
|
||||
|
||||
-- Full output for audit
|
||||
full_output JSONB NOT NULL,
|
||||
|
||||
-- Exit signals
|
||||
exit_signals JSONB NOT NULL DEFAULT '[]'::jsonb,
|
||||
|
||||
-- Metadata
|
||||
pipeline_mode TEXT NOT NULL DEFAULT 'dual_pipeline',
|
||||
shadow_mode BOOLEAN NOT NULL DEFAULT FALSE,
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
|
||||
);
|
||||
|
||||
-- Index for per-ticker time-range queries
|
||||
CREATE INDEX IF NOT EXISTS idx_signal_engine_outputs_ticker_time
|
||||
ON signal_engine_outputs (ticker, evaluated_at);
|
||||
|
||||
-- Index for global time-range queries
|
||||
CREATE INDEX IF NOT EXISTS idx_signal_engine_outputs_evaluated
|
||||
ON signal_engine_outputs (evaluated_at);
|
||||
|
||||
-- Index for filtering by verdict
|
||||
CREATE INDEX IF NOT EXISTS idx_signal_engine_outputs_verdicts
|
||||
ON signal_engine_outputs (heuristic_verdict, probabilistic_verdict);
|
||||
```
|
||||
|
||||
### Helm / Deployment Configuration
|
||||
|
||||
Add to `values.yaml` under `services:`:
|
||||
|
||||
```yaml
|
||||
signalEngine:
|
||||
replicas: 1
|
||||
pipeline: true
|
||||
image: signal-engine
|
||||
command: "python -m services.signal_engine.main"
|
||||
tier: processing
|
||||
secrets: [stonks-core-secrets, stonks-market-secrets]
|
||||
resources:
|
||||
requests: { cpu: 100m, memory: 128Mi }
|
||||
limits: { cpu: 500m, memory: 256Mi }
|
||||
```
|
||||
|
||||
Add to `redis_keys.py`:
|
||||
|
||||
```python
|
||||
QUEUE_SIGNAL_ENGINE = "signal_engine"
|
||||
```
|
||||
|
||||
The service uses the existing `stonks-config` ConfigMap and `stonks-core-secrets` for database/Redis credentials. No new ingress or network policy is needed — the signal engine is a queue-polling worker with no HTTP interface.
|
||||
|
||||
---
|
||||
|
||||
@@ -0,0 +1,300 @@
|
||||
# Requirements Document — Dual-Pipeline Signal Engine
|
||||
|
||||
## Introduction
|
||||
|
||||
The Stonks Oracle platform currently operates a single aggregation pipeline that can run in either heuristic or probabilistic mode (toggled via `probabilistic_scoring_enabled`). This feature replaces the single-pipeline toggle with a dual-pipeline architecture where both pipelines run concurrently per evaluation tick, produce independent verdicts (BUY/WATCH/SKIP), and emit a structured output contract for downstream consumers (trading engine, delta analysis, dashboards).
|
||||
|
||||
The dual-pipeline engine introduces:
|
||||
- **Pipeline A (Heuristic)**: Deterministic scoring using the existing `S_total = S_company + S_macro + S_competitive` formula with signal weighting, producing a confidence-gated verdict.
|
||||
- **Pipeline B (Probabilistic)**: Bayesian inference using the existing `bayesian.py` infrastructure with regime-based priors, likelihood ratios, entropy gating, and expected value calculation.
|
||||
- **Hard Filter Engine**: Pre-pipeline filters that short-circuit both pipelines before evaluation.
|
||||
- **Multi-Timeframe Engine**: Signal evaluation across M30, H1, H4, D, W, M timeframes with weighted confluence scoring.
|
||||
- **Exit Engine**: Position-level exit management (stop hit, targets, trailing ATR-based).
|
||||
- **Delta Analyzer**: Compares heuristic vs probabilistic verdicts to generate training signals for future model tuning.
|
||||
- **Output Formatter**: Structured `SignalOutput` contract consumed by the trading engine and delta analysis.
|
||||
|
||||
The design must address the signal independence assumption in the Bayesian pipeline — correlated signals (MA+RSI, Fib+Elliott) require correlation penalty or signal clustering into categories (momentum, structure, volatility, fundamentals) to prevent likelihood ratio stacking inflation.
|
||||
|
||||
## Glossary
|
||||
|
||||
- **Signal_Engine**: The top-level orchestrator in `services/signal_engine/` that coordinates input normalization, hard filters, both pipelines, delta analysis, and output formatting per evaluation tick.
|
||||
- **Heuristic_Pipeline**: Pipeline A — deterministic scoring that computes `S_total = S_company + S_macro + S_competitive` with signal weighting and produces a confidence-gated BUY/WATCH/SKIP verdict.
|
||||
- **Probabilistic_Pipeline**: Pipeline B — Bayesian inference pipeline that computes posterior probability via log-likelihood accumulation with regime-based priors, entropy gating, and expected value calculation.
|
||||
- **Input_Normalizer**: The component that ingests multi-timeframe OHLCV data, fundamentals, macro context, and open positions into a unified `NormalizedInput` structure consumed by both pipelines.
|
||||
- **Signal_Library**: The collection of technical signal evaluators (Fibonacci retracement, MA stack, RSI, Cup & Handle, Elliott Wave) that produce scored signals per timeframe.
|
||||
- **Multi_Timeframe_Engine**: The component that evaluates signals across six timeframes (M30, H1, H4, D, W, M) and computes weighted confluence scores.
|
||||
- **Hard_Filter_Engine**: The pre-pipeline filter stage that evaluates macro bias, valuation score, and earnings proximity to short-circuit evaluation before either pipeline runs.
|
||||
- **Exit_Engine**: The position management component that evaluates stop hits, take-profit targets, and trailing ATR-based stops for open positions.
|
||||
- **Delta_Analyzer**: The component that compares heuristic and probabilistic verdicts, tracks agreement rates, measures confidence deltas, and records disagreement reasons as training signals.
|
||||
- **Output_Formatter**: The component that assembles the structured `SignalOutput` contract from both pipeline results, delta analysis, and optional trade plan.
|
||||
- **SignalOutput**: The structured output contract containing ticker, timestamp, price, heuristic verdict/confidence/S_total, probabilistic verdict/P_up/entropy/EV_R, delta analysis, and optional trade plan.
|
||||
- **Verdict**: A pipeline decision of BUY, WATCH, or SKIP with associated confidence and reasoning.
|
||||
- **Confluence**: The condition where a signal triggers across multiple timeframes; requires activation on at least 2 timeframes including at least one of D, W, or M.
|
||||
- **Entropy_Gate**: Shannon entropy threshold used in the probabilistic pipeline to detect high-uncertainty states and force SKIP verdicts.
|
||||
- **EV_R**: Expected value per unit of risk, computed as `P_up · E[win_R] - (1 - P_up) · 1.0`, used as a quality gate in the probabilistic pipeline.
|
||||
- **Signal_Cluster**: A grouping of correlated signals (momentum, structure, volatility, fundamentals) used to prevent likelihood ratio stacking inflation in the Bayesian pipeline.
|
||||
- **Likelihood_Ratio**: The ratio `P(signal|up) / P(signal|down)` used in Bayesian updating, where `P(sig|up) = h·s + (1-h)·(1-s)·0.5`.
|
||||
- **Regime_Prior**: The initial probability assigned based on market regime classification: bull=0.58, range=0.50, bear=0.42.
|
||||
- **OHLCV**: Open, High, Low, Close, Volume — standard market data bar format.
|
||||
- **ATR**: Average True Range — a volatility measure used for trailing stop calculations.
|
||||
- **Fibonacci_Retracement**: A technical analysis tool computing price levels as `L(r) = SH - r·(SH - SL)` where SH is swing high, SL is swing low, and r is a retracement ratio (0.236, 0.382, 0.5, 0.618, 0.786).
|
||||
|
||||
---
|
||||
|
||||
## Requirements
|
||||
|
||||
### Requirement 1: Input Normalization
|
||||
|
||||
**User Story:** As a signal engine operator, I want all market data, fundamentals, macro context, and open positions normalized into a single input structure, so that both pipelines consume identical inputs per evaluation tick.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN an evaluation tick is triggered for a ticker, THE Input_Normalizer SHALL construct a `NormalizedInput` containing multi-timeframe OHLCV bars (M30, H1, H4, D, W, M), fundamental metrics (valuation_score, earnings_proximity_days), macro context (macro_bias as float in [-1.0, 1.0]), and open position state (entry_price, current_price, stop_loss, targets).
|
||||
2. THE Input_Normalizer SHALL source OHLCV data from the existing market data tables, fundamental metrics from the existing company and trend data, and macro context from the existing `macro_impact_records` and `global_events` tables.
|
||||
3. IF any required data source is unavailable or returns an error, THEN THE Input_Normalizer SHALL populate the corresponding field with a sentinel value (`None` for optional fields, empty list for OHLCV bars) and log a warning identifying the missing source.
|
||||
4. THE Input_Normalizer SHALL validate that all OHLCV bars have monotonically increasing timestamps within each timeframe series.
|
||||
5. THE Input_Normalizer SHALL produce identical `NormalizedInput` instances for both pipelines within the same evaluation tick (shared reference, no independent fetches).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 2: Signal Library — Technical Signal Evaluation
|
||||
|
||||
**User Story:** As a quantitative analyst, I want a library of technical signal evaluators that produce scored signals per timeframe, so that both pipelines can consume standardized signal assessments.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Signal_Library SHALL implement Fibonacci retracement signal evaluation using the formula `L(r) = SH - r·(SH - SL)` for retracement ratios [0.236, 0.382, 0.5, 0.618, 0.786], where SH is the swing high and SL is the swing low within the evaluation window.
|
||||
2. THE Signal_Library SHALL implement moving average stack evaluation that detects bullish alignment (MA_10 > MA_20 > MA_50 > MA_200) and bearish alignment (MA_10 < MA_20 < MA_50 < MA_200), producing a signal strength proportional to the degree of alignment.
|
||||
3. THE Signal_Library SHALL implement RSI evaluation using the standard 14-period RSI formula, producing overbought signals (RSI > 70) and oversold signals (RSI < 30) with strength scaled by distance from the threshold.
|
||||
4. THE Signal_Library SHALL implement Cup & Handle pattern detection that identifies the cup formation (U-shaped price recovery) and handle (small consolidation), producing a signal with confidence proportional to pattern completeness.
|
||||
5. THE Signal_Library SHALL implement Elliott Wave detection that identifies impulse waves (5-wave structure) and corrective waves (3-wave structure), producing a signal with the current wave position and projected direction.
|
||||
6. WHEN a signal evaluator receives insufficient data for its calculation (fewer bars than the required lookback period), THE Signal_Library SHALL return a null signal with a reason code indicating insufficient data rather than producing a partial evaluation.
|
||||
7. FOR ALL signal evaluators, THE Signal_Library SHALL produce output conforming to a common `SignalResult` structure containing: signal_type, timeframe, strength (float in [0.0, 1.0]), direction (bullish/bearish/neutral), confidence (float in [0.0, 1.0]), and metadata specific to the signal type.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 3: Multi-Timeframe Confluence Engine
|
||||
|
||||
**User Story:** As a quantitative analyst, I want signals evaluated across multiple timeframes with weighted confluence scoring, so that the engine prioritizes signals confirmed across longer timeframes.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Multi_Timeframe_Engine SHALL evaluate each signal type across six timeframes with the following weights: M30=0.03, H1=0.07, H4=0.15, D=0.30, W=0.30, M=0.15.
|
||||
2. THE Multi_Timeframe_Engine SHALL compute a weighted confluence score as `C_confluence = Σ(w_tf · s_tf)` where `w_tf` is the timeframe weight and `s_tf` is the signal strength on that timeframe (0.0 if the signal did not trigger).
|
||||
3. WHEN a signal triggers on fewer than 2 timeframes, THE Multi_Timeframe_Engine SHALL discard the signal from further pipeline processing (minimum confluence threshold).
|
||||
4. WHEN a signal triggers on 2 or more timeframes but none of D, W, or M are included, THE Multi_Timeframe_Engine SHALL discard the signal from further pipeline processing (higher-timeframe anchor requirement).
|
||||
5. THE Multi_Timeframe_Engine SHALL pass the confluence-filtered signals and their weighted scores to both the Heuristic_Pipeline and Probabilistic_Pipeline.
|
||||
6. FOR ALL signal sets where a signal triggers on more timeframes with higher weights, THE Multi_Timeframe_Engine SHALL produce a higher confluence score (monotonicity with respect to timeframe activation count and weight).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 4: Hard Filter Engine — Pre-Pipeline Gating
|
||||
|
||||
**User Story:** As a risk manager, I want hard filters that short-circuit both pipelines before evaluation, so that clearly unfavorable conditions produce immediate SKIP verdicts without wasting computation.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN the macro_bias value from the NormalizedInput equals -1.0, THE Hard_Filter_Engine SHALL produce an immediate SKIP verdict for both pipelines with reason "macro_bias_negative".
|
||||
2. WHEN the valuation_score from the NormalizedInput is below 0.3, THE Hard_Filter_Engine SHALL produce an immediate SKIP verdict for both pipelines with reason "valuation_below_threshold".
|
||||
3. WHEN the earnings_proximity_days from the NormalizedInput is 5 or fewer, THE Hard_Filter_Engine SHALL produce an immediate SKIP verdict for both pipelines with reason "earnings_block".
|
||||
4. WHEN multiple hard filters trigger simultaneously, THE Hard_Filter_Engine SHALL record all triggered filter reasons in the SKIP verdict (not just the first).
|
||||
5. WHEN no hard filters trigger, THE Hard_Filter_Engine SHALL pass the NormalizedInput through to both pipelines without modification.
|
||||
6. THE Hard_Filter_Engine SHALL execute before either pipeline begins evaluation, and both pipelines SHALL receive the same filter decision.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 5: Heuristic Pipeline — Deterministic Scoring and Verdict
|
||||
|
||||
**User Story:** As a quantitative analyst, I want the heuristic pipeline to produce a deterministic BUY/WATCH/SKIP verdict based on composite scoring of company, macro, and competitive signals, so that the system maintains a transparent, auditable scoring path.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Heuristic_Pipeline SHALL compute a total score `S_total = S_company + S_macro + S_competitive` using the existing three-layer signal aggregation with the current `WeightedSignal` abstraction.
|
||||
2. THE Heuristic_Pipeline SHALL compute signal weights using the formula `W_signal = gate · recency · credibility · (1 + novelty) · market_context` consistent with the existing `compute_signal_weight` function in `scoring.py`.
|
||||
3. THE Heuristic_Pipeline SHALL compute a confidence value from the existing trend confidence formula incorporating source count, extraction confidence, signal agreement, and contradiction penalty.
|
||||
4. THE Heuristic_Pipeline SHALL produce a BUY verdict WHEN confidence >= 0.70 AND S_total >= 1.2 AND valuation_score >= 0.5 AND macro_bias > 0 AND earnings_proximity_days > 5.
|
||||
5. THE Heuristic_Pipeline SHALL produce a WATCH verdict WHEN confidence >= 0.55 AND the BUY conditions are not fully met.
|
||||
6. THE Heuristic_Pipeline SHALL produce a SKIP verdict WHEN confidence < 0.55.
|
||||
7. THE Heuristic_Pipeline SHALL emit a `HeuristicResult` containing: verdict (BUY/WATCH/SKIP), confidence (float), S_total (float), S_company (float), S_macro (float), S_competitive (float), signal_weights (list), and reasoning (list of strings explaining the verdict).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 6: Probabilistic Pipeline — Bayesian Inference and Verdict
|
||||
|
||||
**User Story:** As a quantitative analyst, I want the probabilistic pipeline to produce a Bayesian BUY/WATCH/SKIP verdict using regime-based priors, likelihood ratios, entropy gating, and expected value calculation, so that the system captures uncertainty structure and risk-adjusted expected outcomes.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Probabilistic_Pipeline SHALL initialize the prior probability based on the current market regime classification: bull regime → P_prior = 0.58, range regime → P_prior = 0.50, bear regime → P_prior = 0.42.
|
||||
2. THE Probabilistic_Pipeline SHALL compute likelihood ratios for each signal using `P(sig|up) = h·s + (1-h)·(1-s)·0.5` and `LR = P(sig|up) / P(sig|down)`, where h is the signal's historical hit rate and s is the signal strength.
|
||||
3. THE Probabilistic_Pipeline SHALL update the posterior using log-odds accumulation: `logit(P_post) = logit(P_prior) + Σ log(LR_i)`, converting back to probability via the sigmoid function.
|
||||
4. THE Probabilistic_Pipeline SHALL compute Shannon entropy `H = -P_up·log₂(P_up) - (1-P_up)·log₂(1-P_up)` and apply entropy gating: WHEN H > 0.95, THE Probabilistic_Pipeline SHALL force a SKIP verdict with reason "high_entropy".
|
||||
5. THE Probabilistic_Pipeline SHALL compute expected value per unit risk as `EV_R = P_up · E[win_R] - (1 - P_up) · 1.0` where `E[win_R]` is the expected win in risk units derived from signal strength and historical reward-risk ratios.
|
||||
6. THE Probabilistic_Pipeline SHALL produce a BUY verdict WHEN P_up >= 0.60 AND entropy <= 0.90 AND EV_R >= 1.5 AND macro_bias > 0 AND valuation_score >= 0.5.
|
||||
7. THE Probabilistic_Pipeline SHALL produce a WATCH verdict WHEN P_up >= 0.55 AND entropy <= 0.95 AND the BUY conditions are not fully met.
|
||||
8. THE Probabilistic_Pipeline SHALL produce a SKIP verdict in all other cases.
|
||||
9. THE Probabilistic_Pipeline SHALL emit a `ProbabilisticResult` containing: verdict (BUY/WATCH/SKIP), P_up (float), entropy (float), EV_R (float), prior (float), posterior (float), likelihood_ratios (list), regime (string), and reasoning (list of strings).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 7: Signal Correlation Penalty — Preventing LR Stacking Inflation
|
||||
|
||||
**User Story:** As a quantitative analyst, I want correlated signals grouped into clusters with a correlation penalty applied to prevent likelihood ratio stacking inflation, so that the Bayesian pipeline does not overstate confidence from redundant signals.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Probabilistic_Pipeline SHALL classify each signal into one of four clusters: momentum (MA stack, RSI), structure (Fibonacci retracement, Elliott Wave), volatility (ATR-based signals, Bollinger-derived), and fundamentals (valuation, earnings, macro).
|
||||
2. WHEN multiple signals within the same cluster produce likelihood ratios in the same direction, THE Probabilistic_Pipeline SHALL apply a within-cluster penalty: only the strongest LR in the cluster contributes at full weight, and subsequent LRs in the same cluster contribute at a decay factor of 0.5^(n-1) where n is the signal's rank within the cluster by LR magnitude.
|
||||
3. THE Probabilistic_Pipeline SHALL apply no penalty across different clusters (signals from different clusters are treated as independent).
|
||||
4. WHEN a cluster contains only one signal, THE Probabilistic_Pipeline SHALL apply no penalty to that signal.
|
||||
5. FOR ALL signal sets, THE Probabilistic_Pipeline SHALL produce a posterior probability that is less than or equal to the posterior computed without the correlation penalty (the penalty only reduces confidence, never inflates it).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 8: Exit Engine — Position Management
|
||||
|
||||
**User Story:** As a trader, I want the signal engine to evaluate exit conditions for open positions, so that stop hits, take-profit targets, and trailing stops are managed as part of the signal evaluation cycle.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN the current price of an open position hits or crosses below the stop_loss level, THE Exit_Engine SHALL emit an EXIT_FULL signal for that position with reason "stop_hit".
|
||||
2. WHEN the current price of an open position hits or crosses above the first take-profit target (target_1), THE Exit_Engine SHALL emit an EXIT_HALF signal for that position with reason "target_1_hit".
|
||||
3. WHEN the current price of an open position hits or crosses above the second take-profit target (target_2), THE Exit_Engine SHALL emit an EXIT_FULL signal for that position with reason "target_2_hit".
|
||||
4. WHEN a partial exit has been executed (EXIT_HALF), THE Exit_Engine SHALL activate a trailing stop at `current_price - ATR · trailing_multiplier` and update the trailing stop upward as the price advances (the trailing stop moves up but does not move down).
|
||||
5. WHEN the trailing stop is active and the current price crosses below the trailing stop level, THE Exit_Engine SHALL emit an EXIT_FULL signal for the remaining position with reason "trailing_stop_hit".
|
||||
6. THE Exit_Engine SHALL evaluate exit conditions before the signal pipelines run for new entry signals, so that exit signals take priority over new entry signals for the same ticker.
|
||||
7. THE Exit_Engine SHALL emit exit signals as part of the `SignalOutput` contract with the position identifier, exit type (EXIT_HALF/EXIT_FULL), and reason.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 9: Delta Analyzer — Pipeline Agreement Tracking
|
||||
|
||||
**User Story:** As a model developer, I want the delta analyzer to compare heuristic and probabilistic verdicts and record disagreement details, so that I can generate training signals for future model tuning.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN both pipelines produce verdicts for the same ticker and tick, THE Delta_Analyzer SHALL compute an agreement flag (true if both verdicts are identical, false otherwise).
|
||||
2. THE Delta_Analyzer SHALL compute a confidence delta as `|heuristic_confidence - probabilistic_P_up|` representing the magnitude of disagreement between the two pipelines.
|
||||
3. WHEN the pipelines disagree on verdict, THE Delta_Analyzer SHALL record the disagreement reason by identifying which conditions differed (e.g., "heuristic_confidence_below_threshold", "probabilistic_entropy_too_high", "EV_R_below_threshold").
|
||||
4. THE Delta_Analyzer SHALL track a rolling agreement rate over the last 100 evaluations per ticker, stored in Redis for dashboard consumption.
|
||||
5. THE Delta_Analyzer SHALL emit a `DeltaResult` containing: agreement (bool), confidence_delta (float), heuristic_verdict (string), probabilistic_verdict (string), disagreement_reasons (list of strings), and rolling_agreement_rate (float).
|
||||
6. WHEN the rolling agreement rate drops below 0.50 for a ticker, THE Delta_Analyzer SHALL log a warning indicating persistent pipeline disagreement for operator review.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 10: Output Formatter — Structured SignalOutput Contract
|
||||
|
||||
**User Story:** As a downstream system consumer, I want the signal engine to emit a structured `SignalOutput` contract, so that the trading engine, delta analysis dashboard, and audit systems can consume a consistent output format.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Output_Formatter SHALL produce a `SignalOutput` containing: ticker (string), timestamp (datetime), price (float), heuristic section (verdict, confidence, S_total), probabilistic section (verdict, P_up, entropy, EV_R), delta section (agreement, confidence_delta, disagreement_reasons), and optional trade_plan section.
|
||||
2. WHEN the heuristic pipeline produces a BUY verdict, THE Output_Formatter SHALL populate the trade_plan section with entry_price, stop_loss, target_1, target_2, and position_size derived from the heuristic confidence and existing position sizing logic.
|
||||
3. WHEN the probabilistic pipeline produces a BUY verdict but the heuristic pipeline does not, THE Output_Formatter SHALL populate the trade_plan section with a "probabilistic_only" flag and reduced position sizing (50% of standard).
|
||||
4. WHEN both pipelines produce a BUY verdict, THE Output_Formatter SHALL populate the trade_plan section with full position sizing and a "dual_confirmed" flag.
|
||||
5. THE Output_Formatter SHALL serialize the `SignalOutput` as a Pydantic model with JSON serialization support for Redis queue publishing and database persistence.
|
||||
6. FOR ALL valid pipeline results, THE Output_Formatter SHALL produce a `SignalOutput` that round-trips through JSON serialization and deserialization without data loss (parse(format(output)) produces an equivalent object).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 11: Dual Pipeline Orchestration
|
||||
|
||||
**User Story:** As a signal engine operator, I want both pipelines to run concurrently per evaluation tick sharing the same inputs, so that the system produces independent verdicts without redundant data fetching.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN an evaluation tick is triggered, THE Signal_Engine SHALL execute the Input_Normalizer once, then pass the resulting `NormalizedInput` to the Hard_Filter_Engine, then (if not filtered) execute both the Heuristic_Pipeline and Probabilistic_Pipeline concurrently using `asyncio.gather`.
|
||||
2. THE Signal_Engine SHALL enforce that both pipelines receive identical `NormalizedInput` references (no independent data fetches that could produce different snapshots).
|
||||
3. WHEN either pipeline raises an exception during evaluation, THE Signal_Engine SHALL catch the exception, log the error with full traceback, and produce a SKIP verdict for the failed pipeline with reason "pipeline_error" while allowing the other pipeline to complete normally.
|
||||
4. THE Signal_Engine SHALL measure and log the wall-clock execution time of each pipeline per tick for performance monitoring.
|
||||
5. THE Signal_Engine SHALL publish the assembled `SignalOutput` to the existing Redis queue (`stonks:queue:trading_decisions`) for consumption by the trading engine.
|
||||
6. THE Signal_Engine SHALL persist each `SignalOutput` to a database table for historical analysis and audit.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 12: Integration with Existing Trading Engine
|
||||
|
||||
**User Story:** As a platform operator, I want the dual-pipeline signal engine to integrate with the existing trading engine, so that the trading engine can consume `SignalOutput` verdicts and make execution decisions.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Signal_Engine SHALL publish `SignalOutput` to the existing `stonks:queue:trading_decisions` Redis queue in a format compatible with the existing `TradingEngine.evaluate_recommendation` interface.
|
||||
2. THE Signal_Engine SHALL map the `SignalOutput` trade_plan to the existing `Recommendation` schema fields (action, confidence, position_sizing) so that the trading engine can process dual-pipeline outputs without modification to its core evaluation logic.
|
||||
3. WHEN the `SignalOutput` has a "dual_confirmed" flag, THE Signal_Engine SHALL set the recommendation confidence to the maximum of heuristic_confidence and probabilistic_P_up.
|
||||
4. WHEN the `SignalOutput` has a "probabilistic_only" flag, THE Signal_Engine SHALL set the recommendation confidence to `probabilistic_P_up · 0.8` (20% confidence haircut for single-pipeline confirmation).
|
||||
5. WHEN neither pipeline produces a BUY verdict, THE Signal_Engine SHALL not publish a trading recommendation to the queue (WATCH and SKIP verdicts are persisted for analysis but not forwarded to the trading engine).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 13: Configuration and Feature Flags
|
||||
|
||||
**User Story:** As a platform operator, I want the dual-pipeline engine configurable via the existing `risk_configs` table and environment variables, so that I can tune thresholds, enable/disable individual pipelines, and adjust timeframe weights without code changes.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Signal_Engine SHALL support a `dual_pipeline_enabled` feature flag in `risk_configs` that toggles the entire dual-pipeline engine on or off, defaulting to false for safe rollout.
|
||||
2. THE Signal_Engine SHALL support independent enable/disable flags for each pipeline: `heuristic_pipeline_enabled` and `probabilistic_pipeline_enabled`, both defaulting to true when the dual-pipeline engine is enabled.
|
||||
3. THE Signal_Engine SHALL support configurable timeframe weights via a `timeframe_weights` JSON object in `risk_configs`, defaulting to `{"M30": 0.03, "H1": 0.07, "H4": 0.15, "D": 0.30, "W": 0.30, "M": 0.15}`.
|
||||
4. THE Signal_Engine SHALL support configurable hard filter thresholds: `hard_filter_valuation_min` (default 0.3), `hard_filter_earnings_days` (default 5), and `hard_filter_macro_bias_skip` (default -1.0).
|
||||
5. THE Signal_Engine SHALL support configurable verdict thresholds for both pipelines via `risk_configs` JSON, including heuristic confidence thresholds (BUY: 0.70, WATCH: 0.55) and probabilistic thresholds (P_up: 0.60, entropy: 0.90, EV_R: 1.5).
|
||||
6. IF the `dual_pipeline_enabled` flag fails to read from the database, THEN THE Signal_Engine SHALL default to disabled (fail-safe behavior) and log a warning.
|
||||
7. THE Signal_Engine SHALL log the active configuration at startup and on each configuration change for auditability.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 14: Regime-Based Prior Engine
|
||||
|
||||
**User Story:** As a quantitative analyst, I want the probabilistic pipeline's prior probability to adapt based on the current market regime, so that the Bayesian inference starts from a regime-appropriate baseline rather than a fixed 0.50.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Probabilistic_Pipeline SHALL use the existing `classify_regime` function from `services/aggregation/regime.py` to determine the current market regime for each ticker.
|
||||
2. THE Probabilistic_Pipeline SHALL map regime classifications to prior probabilities: trend_following with positive trend_indicator → 0.58 (bull), trend_following with negative trend_indicator → 0.42 (bear), mean_reversion → 0.50 (range), panic → 0.42 (bear), uncertainty → 0.50 (range).
|
||||
3. THE Probabilistic_Pipeline SHALL convert the regime prior to log-odds before accumulating likelihood ratios: `logit(P_prior) = log(P_prior / (1 - P_prior))`.
|
||||
4. WHEN market data is insufficient for regime classification (fewer than 100 days of price history), THE Probabilistic_Pipeline SHALL use the uncertainty prior of 0.50.
|
||||
5. THE Probabilistic_Pipeline SHALL record the regime classification and prior probability in the `ProbabilisticResult` for auditability.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 15: Database Schema for Signal Engine Output
|
||||
|
||||
**User Story:** As a platform operator, I want signal engine outputs persisted to a dedicated database table, so that historical evaluations are available for analysis, backtesting, and audit.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Signal_Engine SHALL persist each `SignalOutput` to a `signal_engine_outputs` table with columns for: id (UUID primary key), ticker (text), evaluated_at (timestamptz), price (numeric), heuristic_verdict (text), heuristic_confidence (numeric), heuristic_s_total (numeric), probabilistic_verdict (text), probabilistic_p_up (numeric), probabilistic_entropy (numeric), probabilistic_ev_r (numeric), delta_agreement (boolean), delta_confidence_delta (numeric), delta_reasons (JSONB), trade_plan (JSONB), full_output (JSONB), created_at (timestamptz).
|
||||
2. THE Signal_Engine SHALL create an index on `(ticker, evaluated_at)` for efficient time-range queries per ticker.
|
||||
3. THE Signal_Engine SHALL create an index on `evaluated_at` for efficient global time-range queries.
|
||||
4. WHEN persisting fails due to a database error, THE Signal_Engine SHALL log the error and continue processing (persistence failure does not block signal emission to the trading queue).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 16: Backward Compatibility and Migration Path
|
||||
|
||||
**User Story:** As a platform operator, I want the dual-pipeline engine to coexist with the existing single-pipeline aggregation, so that the rollout is incremental and reversible.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN `dual_pipeline_enabled` is false, THE Signal_Engine SHALL not run, and the existing aggregation pipeline SHALL continue to operate unchanged.
|
||||
2. WHEN `dual_pipeline_enabled` is true, THE Signal_Engine SHALL run alongside the existing aggregation pipeline, with the trading engine consuming `SignalOutput` from the dual-pipeline engine instead of `Recommendation` from the existing recommendation worker.
|
||||
3. THE Signal_Engine SHALL reuse the existing `WeightedSignal`, `BayesianPosterior`, `RegimeClassification`, and `TrendSummary` data structures from `services/aggregation/` rather than duplicating them.
|
||||
4. THE Signal_Engine SHALL reuse the existing `compute_signal_weight`, `compute_bayesian_posterior`, and `classify_regime` functions rather than reimplementing the underlying math.
|
||||
5. THE Signal_Engine SHALL add the new `signal_engine_outputs` table via a new database migration without modifying existing tables.
|
||||
6. THE Signal_Engine SHALL support running in "shadow mode" where both the existing pipeline and the dual-pipeline engine run, but only the existing pipeline's output is forwarded to the trading engine (dual-pipeline output is persisted for comparison only).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 17: Property-Based Testing for Dual-Pipeline Correctness
|
||||
|
||||
**User Story:** As a developer, I want comprehensive property-based tests validating the mathematical correctness and structural invariants of the dual-pipeline engine, so that edge cases and numerical stability issues are caught before deployment.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE test suite SHALL include property-based tests for the Fibonacci retracement formula verifying that `L(r) = SH - r·(SH - SL)` produces values in [SL, SH] for all r in [0, 1] and all SH > SL > 0.
|
||||
2. THE test suite SHALL include property-based tests for the Bayesian log-odds update verifying that `logit(P_post) = logit(P_prior) + Σ log(LR_i)` round-trips correctly: converting P_prior to logit, adding log-LRs, and converting back via sigmoid produces a valid probability in (0, 1).
|
||||
3. THE test suite SHALL include property-based tests for the entropy gate verifying that Shannon entropy is maximized at P_up = 0.5 and equals 0.0 at P_up = 0.0 or P_up = 1.0, and is symmetric around 0.5.
|
||||
4. THE test suite SHALL include property-based tests for the signal correlation penalty verifying that the penalized posterior is always less than or equal to the unpenalized posterior for any signal set with correlated signals.
|
||||
5. THE test suite SHALL include property-based tests for the multi-timeframe confluence score verifying monotonicity: activating a signal on an additional timeframe with non-zero weight always increases or maintains the confluence score.
|
||||
6. THE test suite SHALL include property-based tests for the `SignalOutput` contract verifying round-trip serialization: `SignalOutput.model_validate_json(output.model_dump_json())` produces an equivalent object for all valid outputs.
|
||||
7. THE test suite SHALL include property-based tests for the hard filter engine verifying that macro_bias = -1.0 always produces SKIP, valuation_score < 0.3 always produces SKIP, and earnings_proximity_days <= 5 always produces SKIP, regardless of all other input values.
|
||||
8. THE test suite SHALL include property-based tests for the EV_R calculation verifying that `EV_R = P_up · E[win_R] - (1 - P_up) · 1.0` is monotonically increasing with P_up for fixed E[win_R] > 0.
|
||||
@@ -0,0 +1,345 @@
|
||||
# Implementation Plan: Dual-Pipeline Signal Engine
|
||||
|
||||
## Overview
|
||||
|
||||
Implement the dual-pipeline signal engine as a new service at `services/signal_engine/` that runs as an independent Kubernetes deployment. The engine evaluates both a heuristic (deterministic scoring) and probabilistic (Bayesian inference) pipeline concurrently per ticker per evaluation tick, producing independent BUY/WATCH/SKIP verdicts. Implementation proceeds incrementally: infrastructure first, then core models, signal library, pipelines, orchestration, integration, and deployment.
|
||||
|
||||
## Tasks
|
||||
|
||||
- [x] 1. Project scaffolding, configuration, and data models
|
||||
- [x] 1.1 Create service directory structure and `__init__.py` files
|
||||
- Create `services/signal_engine/` with all subdirectories per the design module structure
|
||||
- Create `services/signal_engine/__init__.py`, `services/signal_engine/signals/__init__.py`
|
||||
- _Requirements: 11.1, 13.1_
|
||||
|
||||
- [x] 1.2 Implement `models.py` — all Pydantic data models
|
||||
- Define `OHLCVBar`, `NormalizedInput`, `OpenPositionState`, `SignalResult`, `SignalDirection`
|
||||
- Define `ConfluenceSignal`, `Verdict`, `HeuristicResult`, `LikelihoodRatio`, `ProbabilisticResult`
|
||||
- Define `DeltaResult`, `ExitSignal`, `ExitType`, `TradePlan`, `SignalOutput`
|
||||
- All models must use Pydantic `BaseModel` with proper field constraints (`ge`, `le`)
|
||||
- _Requirements: 1.1, 2.7, 5.7, 6.9, 9.5, 10.1, 10.5_
|
||||
|
||||
- [x] 1.3 Implement `config.py` — `SignalEngineConfig` and sub-configs
|
||||
- Define `SignalEngineConfig` dataclass with all fields from the design
|
||||
- Define `HardFilterConfig`, `HeuristicConfig`, `ProbabilisticConfig`, `ExitConfig` as derived sub-configs
|
||||
- Implement `load_config()` that reads from `risk_configs` table + environment variables
|
||||
- Default `dual_pipeline_enabled` to `False` (fail-safe)
|
||||
- _Requirements: 13.1, 13.2, 13.3, 13.4, 13.5, 13.6, 13.7_
|
||||
|
||||
- [x] 1.4 Add `QUEUE_SIGNAL_ENGINE` to `services/shared/redis_keys.py`
|
||||
- Add `QUEUE_SIGNAL_ENGINE = "signal_engine"` constant
|
||||
- _Requirements: 11.1_
|
||||
|
||||
- [x] 1.5 Write property test for `SignalOutput` round-trip serialization
|
||||
- **Requirement 17.6: SignalOutput round-trip serialization**
|
||||
- Generate arbitrary valid `SignalOutput` instances with Hypothesis
|
||||
- Verify `SignalOutput.model_validate_json(output.model_dump_json())` produces equivalent object
|
||||
- File: `tests/test_pbt_signal_engine_models.py`
|
||||
- _Requirements: 10.5, 17.6_
|
||||
|
||||
- [x] 2. Input Normalizer and Hard Filter Engine
|
||||
- [x] 2.1 Implement `normalizer.py` — Input Normalizer
|
||||
- Implement `normalize_input(pool, ticker, config) -> NormalizedInput`
|
||||
- Fetch OHLCV bars from `market_data_bars` for M30, H1, H4, D, W, M timeframes
|
||||
- Fetch fundamental metrics (valuation_score, earnings_proximity_days) from company/trend data
|
||||
- Fetch macro context (macro_bias) from `macro_impact_records` and `global_events`
|
||||
- Fetch open position state from trading engine portfolio tables
|
||||
- Populate sentinel values (`None`, empty list) for unavailable data with logged warnings
|
||||
- Validate monotonically increasing timestamps within each timeframe series
|
||||
- _Requirements: 1.1, 1.2, 1.3, 1.4, 1.5_
|
||||
|
||||
- [x] 2.2 Implement `hard_filter.py` — Hard Filter Engine
|
||||
- Implement `evaluate_hard_filters(normalized, config) -> HardFilterResult`
|
||||
- Check `macro_bias == -1.0` → SKIP with reason "macro_bias_negative"
|
||||
- Check `valuation_score < 0.3` → SKIP with reason "valuation_below_threshold"
|
||||
- Check `earnings_proximity_days <= 5` → SKIP with reason "earnings_block"
|
||||
- Record all triggered filter reasons (not just first)
|
||||
- Return `HardFilterResult` with `filtered: bool` and `reasons: list[str]`
|
||||
- _Requirements: 4.1, 4.2, 4.3, 4.4, 4.5, 4.6_
|
||||
|
||||
- [x] 2.3 Write property tests for hard filter engine
|
||||
- **Requirement 17.7: Hard filter determinism**
|
||||
- Generate arbitrary `NormalizedInput` with `macro_bias = -1.0` → always SKIP
|
||||
- Generate arbitrary `NormalizedInput` with `valuation_score < 0.3` → always SKIP
|
||||
- Generate arbitrary `NormalizedInput` with `earnings_proximity_days <= 5` → always SKIP
|
||||
- Verify these hold regardless of all other input values
|
||||
- File: `tests/test_pbt_signal_engine_hard_filter.py`
|
||||
- _Requirements: 4.1, 4.2, 4.3, 17.7_
|
||||
|
||||
- [x] 3. Checkpoint — Ensure all tests pass
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [x] 4. Signal Library — Technical Signal Evaluators
|
||||
- [x] 4.1 Implement `signals/base.py` — SignalEvaluator protocol
|
||||
- Define `SignalEvaluator` protocol with `evaluate(bars, timeframe) -> SignalResult | None`
|
||||
- Define common helper functions for swing high/low detection, lookback validation
|
||||
- _Requirements: 2.6, 2.7_
|
||||
|
||||
- [x] 4.2 Implement `signals/fibonacci.py` — Fibonacci retracement evaluator
|
||||
- Implement `L(r) = SH - r·(SH - SL)` for ratios [0.236, 0.382, 0.5, 0.618, 0.786]
|
||||
- Detect swing high and swing low within the evaluation window
|
||||
- Produce signal strength based on proximity of current price to retracement levels
|
||||
- Return `None` with reason code when insufficient data
|
||||
- _Requirements: 2.1, 2.6, 2.7_
|
||||
|
||||
- [x] 4.3 Write property test for Fibonacci retracement formula
|
||||
- **Requirement 17.1: Fibonacci retracement bounds**
|
||||
- For all `r` in [0, 1] and all `SH > SL > 0`, verify `L(r)` is in [SL, SH]
|
||||
- File: `tests/test_pbt_signal_engine_fibonacci.py`
|
||||
- _Requirements: 2.1, 17.1_
|
||||
|
||||
- [x] 4.4 Implement `signals/ma_stack.py` — Moving average stack evaluator
|
||||
- Detect bullish alignment (MA_10 > MA_20 > MA_50 > MA_200)
|
||||
- Detect bearish alignment (MA_10 < MA_20 < MA_50 < MA_200)
|
||||
- Produce signal strength proportional to degree of alignment
|
||||
- Return `None` when insufficient bars for MA_200 calculation
|
||||
- _Requirements: 2.2, 2.6, 2.7_
|
||||
|
||||
- [x] 4.5 Implement `signals/rsi.py` — RSI evaluator
|
||||
- Implement standard 14-period RSI formula
|
||||
- Produce overbought signals (RSI > 70) and oversold signals (RSI < 30)
|
||||
- Scale strength by distance from threshold
|
||||
- Return `None` when fewer than 14 bars available
|
||||
- _Requirements: 2.3, 2.6, 2.7_
|
||||
|
||||
- [x] 4.6 Implement `signals/cup_handle.py` — Cup & Handle pattern detector
|
||||
- Identify cup formation (U-shaped price recovery) and handle (small consolidation)
|
||||
- Produce signal with confidence proportional to pattern completeness
|
||||
- Return `None` when insufficient data or no pattern detected
|
||||
- _Requirements: 2.4, 2.6, 2.7_
|
||||
|
||||
- [x] 4.7 Implement `signals/elliott_wave.py` — Elliott Wave detector
|
||||
- Identify impulse waves (5-wave structure) and corrective waves (3-wave structure)
|
||||
- Produce signal with current wave position and projected direction
|
||||
- Return `None` when insufficient data or ambiguous wave count
|
||||
- _Requirements: 2.5, 2.6, 2.7_
|
||||
|
||||
- [x] 5. Multi-Timeframe Confluence Engine
|
||||
- [x] 5.1 Implement `confluence.py` — Multi-Timeframe Engine
|
||||
- Implement `compute_confluence(signal_results, weights) -> list[ConfluenceSignal]`
|
||||
- Compute weighted confluence score: `C_confluence = Σ(w_tf · s_tf)`
|
||||
- Apply minimum confluence threshold: discard signals triggering on < 2 timeframes
|
||||
- Apply higher-timeframe anchor: discard signals without at least one of D, W, or M
|
||||
- Return `ConfluenceSignal` objects with active timeframes and per-timeframe strengths
|
||||
- _Requirements: 3.1, 3.2, 3.3, 3.4, 3.5, 3.6_
|
||||
|
||||
- [x] 5.2 Write property test for confluence score monotonicity
|
||||
- **Requirement 17.5: Confluence score monotonicity**
|
||||
- Verify that activating a signal on an additional timeframe with non-zero weight always increases or maintains the confluence score
|
||||
- File: `tests/test_pbt_signal_engine_confluence.py`
|
||||
- _Requirements: 3.6, 17.5_
|
||||
|
||||
- [x] 6. Checkpoint — Ensure all tests pass
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [x] 7. Heuristic Pipeline (Pipeline A)
|
||||
- [x] 7.1 Implement `heuristic.py` — Heuristic Pipeline
|
||||
- Implement `run_heuristic_pipeline(normalized, confluence_signals, config) -> HeuristicResult`
|
||||
- Compute `S_total = S_company + S_macro + S_competitive` using existing `compute_signal_weight()`
|
||||
- Compute confidence from source count, extraction confidence, signal agreement, contradiction penalty
|
||||
- BUY verdict: confidence >= 0.70 AND S_total >= 1.2 AND valuation_score >= 0.5 AND macro_bias > 0 AND earnings_proximity_days > 5
|
||||
- WATCH verdict: confidence >= 0.55 AND BUY conditions not fully met
|
||||
- SKIP verdict: confidence < 0.55
|
||||
- Emit `HeuristicResult` with all required fields and reasoning
|
||||
- _Requirements: 5.1, 5.2, 5.3, 5.4, 5.5, 5.6, 5.7_
|
||||
|
||||
- [x] 7.2 Write unit tests for heuristic pipeline verdict logic
|
||||
- Test BUY threshold conditions
|
||||
- Test WATCH threshold conditions
|
||||
- Test SKIP conditions
|
||||
- Test edge cases at threshold boundaries
|
||||
- File: `tests/test_signal_engine_heuristic.py`
|
||||
- _Requirements: 5.4, 5.5, 5.6_
|
||||
|
||||
- [x] 8. Probabilistic Pipeline (Pipeline B) and Correlation Penalty
|
||||
- [x] 8.1 Implement `correlation.py` — Signal cluster classification and penalty
|
||||
- Define `SignalCluster` enum: MOMENTUM, STRUCTURE, VOLATILITY, FUNDAMENTALS
|
||||
- Implement `classify_signal(signal_type) -> SignalCluster`
|
||||
- Implement `apply_correlation_penalty(likelihood_ratios) -> list[LikelihoodRatio]`
|
||||
- Within-cluster decay: strongest LR at full weight, subsequent at 0.5^(n-1)
|
||||
- No penalty across different clusters
|
||||
- Single-signal clusters receive no penalty
|
||||
- _Requirements: 7.1, 7.2, 7.3, 7.4_
|
||||
|
||||
- [x] 8.2 Implement `probabilistic.py` — Probabilistic Pipeline
|
||||
- Implement `run_probabilistic_pipeline(normalized, confluence_signals, regime, config) -> ProbabilisticResult`
|
||||
- Initialize regime-based prior: bull=0.58, range=0.50, bear=0.42
|
||||
- Compute likelihood ratios: `P(sig|up) = h·s + (1-h)·(1-s)·0.5`, `LR = P(sig|up) / P(sig|down)`
|
||||
- Apply correlation penalty via `apply_correlation_penalty()`
|
||||
- Accumulate via log-odds: `logit(P_post) = logit(P_prior) + Σ log(LR_i)`
|
||||
- Compute Shannon entropy and apply entropy gating (H > 0.95 → SKIP)
|
||||
- Compute `EV_R = P_up · E[win_R] - (1 - P_up) · 1.0`
|
||||
- BUY: P_up >= 0.60 AND entropy <= 0.90 AND EV_R >= 1.5 AND macro_bias > 0 AND valuation_score >= 0.5
|
||||
- WATCH: P_up >= 0.55 AND entropy <= 0.95 AND BUY conditions not fully met
|
||||
- SKIP: all other cases
|
||||
- Use existing `classify_regime()` from `services/aggregation/regime.py`
|
||||
- _Requirements: 6.1, 6.2, 6.3, 6.4, 6.5, 6.6, 6.7, 6.8, 6.9, 14.1, 14.2, 14.3, 14.4, 14.5_
|
||||
|
||||
- [x] 8.3 Write property test for Bayesian log-odds round-trip
|
||||
- **Requirement 17.2: Bayesian log-odds update correctness**
|
||||
- Verify `logit(P_post) = logit(P_prior) + Σ log(LR_i)` round-trips correctly
|
||||
- Converting P_prior to logit, adding log-LRs, converting back via sigmoid produces valid probability in (0, 1)
|
||||
- File: `tests/test_pbt_signal_engine_bayesian.py`
|
||||
- _Requirements: 6.3, 17.2_
|
||||
|
||||
- [x] 8.4 Write property test for entropy gate
|
||||
- **Requirement 17.3: Entropy gate properties**
|
||||
- Verify Shannon entropy is maximized at P_up = 0.5
|
||||
- Verify entropy equals 0.0 at P_up = 0.0 or P_up = 1.0
|
||||
- Verify entropy is symmetric around 0.5
|
||||
- File: `tests/test_pbt_signal_engine_bayesian.py`
|
||||
- _Requirements: 6.4, 17.3_
|
||||
|
||||
- [x] 8.5 Write property test for signal correlation penalty
|
||||
- **Requirement 17.4: Correlation penalty reduces confidence**
|
||||
- Verify penalized posterior is always <= unpenalized posterior for any signal set with correlated signals
|
||||
- File: `tests/test_pbt_signal_engine_correlation.py`
|
||||
- _Requirements: 7.5, 17.4_
|
||||
|
||||
- [x] 8.6 Write property test for EV_R monotonicity
|
||||
- **Requirement 17.8: EV_R monotonically increasing with P_up**
|
||||
- Verify `EV_R = P_up · E[win_R] - (1 - P_up) · 1.0` is monotonically increasing with P_up for fixed E[win_R] > 0
|
||||
- File: `tests/test_pbt_signal_engine_bayesian.py`
|
||||
- _Requirements: 6.5, 17.8_
|
||||
|
||||
- [x] 9. Checkpoint — Ensure all tests pass
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [x] 10. Exit Engine
|
||||
- [x] 10.1 Implement `exit_engine.py` — Exit Engine
|
||||
- Implement `evaluate_exits(positions, current_prices, config) -> list[ExitSignal]`
|
||||
- Check stop_loss hit → EXIT_FULL with reason "stop_hit"
|
||||
- Check target_1 hit → EXIT_HALF with reason "target_1_hit"
|
||||
- Check target_2 hit → EXIT_FULL with reason "target_2_hit"
|
||||
- Trailing stop: activate after EXIT_HALF at `current_price - ATR · trailing_multiplier`
|
||||
- Trailing stop ratchets upward only (never moves down)
|
||||
- Trailing stop hit → EXIT_FULL with reason "trailing_stop_hit"
|
||||
- _Requirements: 8.1, 8.2, 8.3, 8.4, 8.5, 8.6, 8.7_
|
||||
|
||||
- [x] 10.2 Write unit tests for exit engine
|
||||
- Test stop_loss trigger
|
||||
- Test target_1 partial exit
|
||||
- Test target_2 full exit
|
||||
- Test trailing stop activation and ratchet behavior
|
||||
- File: `tests/test_signal_engine_exit.py`
|
||||
- _Requirements: 8.1, 8.2, 8.3, 8.4, 8.5_
|
||||
|
||||
- [x] 11. Delta Analyzer and Output Formatter
|
||||
- [x] 11.1 Implement `delta.py` — Delta Analyzer
|
||||
- Implement `analyze_delta(heuristic, probabilistic, redis, ticker) -> DeltaResult`
|
||||
- Compute agreement flag (both verdicts identical)
|
||||
- Compute confidence delta: `|heuristic_confidence - probabilistic_P_up|`
|
||||
- Record disagreement reasons when verdicts differ
|
||||
- Track rolling 100-evaluation agreement rate in Redis
|
||||
- Log warning when agreement rate drops below 0.50
|
||||
- _Requirements: 9.1, 9.2, 9.3, 9.4, 9.5, 9.6_
|
||||
|
||||
- [x] 11.2 Implement `formatter.py` — Output Formatter
|
||||
- Implement `format_output(ticker, price, heuristic, probabilistic, delta, exit_signals, config) -> SignalOutput`
|
||||
- Both BUY → `dual_confirmed`, full position sizing
|
||||
- Probabilistic-only BUY → `probabilistic_only`, 50% position sizing
|
||||
- Heuristic-only BUY → standard position sizing
|
||||
- No BUY → no trade_plan (WATCH/SKIP persisted for analysis)
|
||||
- Implement `signal_output_to_recommendation(output) -> Recommendation`
|
||||
- Map `SignalOutput` to existing `Recommendation` schema for trading engine compatibility
|
||||
- Dual confirmed: confidence = max(heuristic_confidence, probabilistic_P_up)
|
||||
- Probabilistic only: confidence = probabilistic_P_up · 0.8 (20% haircut)
|
||||
- _Requirements: 10.1, 10.2, 10.3, 10.4, 10.5, 10.6, 12.1, 12.2, 12.3, 12.4, 12.5_
|
||||
|
||||
- [x] 11.3 Write unit tests for output formatter
|
||||
- Test dual_confirmed trade plan generation
|
||||
- Test probabilistic_only trade plan with 50% sizing
|
||||
- Test heuristic-only trade plan
|
||||
- Test no-BUY case (no trade_plan)
|
||||
- Test `signal_output_to_recommendation` mapping
|
||||
- File: `tests/test_signal_engine_formatter.py`
|
||||
- _Requirements: 10.2, 10.3, 10.4, 12.3, 12.4_
|
||||
|
||||
- [x] 12. Orchestrator, Persistence, and Main Entry Point
|
||||
- [x] 12.1 Implement `persistence.py` — Database persistence
|
||||
- Implement `persist_signal_output(pool, output) -> None`
|
||||
- Insert into `signal_engine_outputs` table
|
||||
- Log and continue on database errors (non-blocking)
|
||||
- _Requirements: 15.1, 15.4_
|
||||
|
||||
- [x] 12.2 Implement `worker.py` — Top-level orchestrator
|
||||
- Implement `evaluate_tick(pool, redis, ticker, config) -> SignalOutput | None`
|
||||
- Step 1: Normalize inputs (single fetch, shared reference)
|
||||
- Step 2: Evaluate exit conditions for open positions
|
||||
- Step 3: Run hard filters (short-circuit if filtered)
|
||||
- Step 4: Evaluate signals across timeframes via Signal Library
|
||||
- Step 5: Compute confluence
|
||||
- Step 6: Classify regime via existing `classify_regime()`
|
||||
- Step 7: Run both pipelines concurrently via `asyncio.gather` with exception handling
|
||||
- Step 8: Compute delta analysis
|
||||
- Step 9: Format output
|
||||
- Step 10: Persist to database and publish to Redis queue
|
||||
- Catch pipeline exceptions → SKIP verdict for failed pipeline, other continues
|
||||
- Measure and log wall-clock execution time per pipeline
|
||||
- _Requirements: 11.1, 11.2, 11.3, 11.4, 11.5, 11.6_
|
||||
|
||||
- [x] 12.3 Implement `main.py` — Entry point with asyncio event loop
|
||||
- Connect to PostgreSQL (asyncpg pool) and Redis (redis.asyncio)
|
||||
- Load config from `risk_configs` table
|
||||
- Log active configuration at startup
|
||||
- Poll `stonks:queue:signal_engine` queue indefinitely
|
||||
- Check `dual_pipeline_enabled` flag; if disabled, sleep and retry
|
||||
- On config read failure, default to disabled (fail-safe)
|
||||
- Support shadow mode (persist but don't forward to trading queue)
|
||||
- _Requirements: 13.1, 13.6, 13.7, 16.1, 16.6_
|
||||
|
||||
- [x] 12.4 Write integration tests for worker orchestration
|
||||
- Test full tick evaluation with mocked DB/Redis
|
||||
- Test pipeline failure isolation (one fails, other completes)
|
||||
- Test hard filter short-circuit
|
||||
- Test shadow mode behavior
|
||||
- File: `tests/test_signal_engine_worker.py`
|
||||
- _Requirements: 11.3, 16.6_
|
||||
|
||||
- [x] 13. Checkpoint — Ensure all tests pass
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [x] 14. Database migration and infrastructure
|
||||
- [x] 14.1 Create database migration `infra/migrations/039_signal_engine_outputs.sql`
|
||||
- Create `signal_engine_outputs` table per the design schema
|
||||
- Create index on `(ticker, evaluated_at)` for per-ticker time-range queries
|
||||
- Create index on `evaluated_at` for global time-range queries
|
||||
- Create index on `(heuristic_verdict, probabilistic_verdict)` for verdict filtering
|
||||
- _Requirements: 15.1, 15.2, 15.3_
|
||||
|
||||
- [x] 14.2 Add signal engine service to Helm chart
|
||||
- Add `signalEngine` entry to `infra/helm/stonks-oracle/values.yaml`
|
||||
- Configure: replicas=1, command=`python -m services.signal_engine.main`, tier=processing
|
||||
- Set resource requests/limits per design (100m/128Mi → 500m/256Mi)
|
||||
- Reference existing secrets: `stonks-core-secrets`, `stonks-market-secrets`
|
||||
- _Requirements: 11.1, 13.1_
|
||||
|
||||
- [x] 15. Trading engine integration and backward compatibility
|
||||
- [x] 15.1 Wire signal engine output to trading engine queue
|
||||
- Publish `SignalOutput` (mapped to `Recommendation`) to `stonks:queue:trading_decisions`
|
||||
- Only publish when at least one pipeline produces BUY verdict
|
||||
- WATCH/SKIP verdicts persisted for analysis but not forwarded
|
||||
- Ensure trading engine can consume without modification via `signal_output_to_recommendation()`
|
||||
- _Requirements: 12.1, 12.2, 12.5, 16.2_
|
||||
|
||||
- [x] 15.2 Ensure backward compatibility with existing pipeline
|
||||
- Verify `dual_pipeline_enabled=false` means signal engine does not run
|
||||
- Verify existing aggregation pipeline operates unchanged when flag is off
|
||||
- Reuse existing `WeightedSignal`, `BayesianPosterior`, `RegimeClassification` (import, don't duplicate)
|
||||
- Reuse existing `compute_signal_weight`, `compute_bayesian_posterior`, `classify_regime` functions
|
||||
- No modifications to existing tables (new migration only adds new table)
|
||||
- _Requirements: 16.1, 16.2, 16.3, 16.4, 16.5_
|
||||
|
||||
- [x] 16. Final checkpoint — Ensure all tests pass
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
## Notes
|
||||
|
||||
- Tasks marked with `*` are optional and can be skipped for faster MVP
|
||||
- Each task references specific requirements for traceability
|
||||
- Checkpoints ensure incremental validation between major phases
|
||||
- Property-based tests use Hypothesis with `@settings(max_examples=100)` per project conventions
|
||||
- PBT test files are prefixed `test_pbt_*` per project conventions
|
||||
- The service reuses existing math functions from `services/aggregation/` — no reimplementation
|
||||
- All configuration is loaded from `risk_configs` table with fail-safe defaults
|
||||
- Shadow mode allows running alongside existing pipeline without affecting trading decisions
|
||||
@@ -0,0 +1 @@
|
||||
{"specId": "d2fe9091-6423-482c-a4ce-3cd72e62eb23", "workflowType": "requirements-first", "specType": "feature"}
|
||||
@@ -0,0 +1,153 @@
|
||||
# Design Document: Intelligence Pipeline Deep Dive
|
||||
|
||||
## Overview
|
||||
|
||||
This design specifies the structure, content, and creation process for a 6-page narrative deep-dive document covering the full intelligence-to-decision pipeline in Stonks Oracle. The deliverable consists of Markdown narrative pages, an index file, and standalone Mermaid diagram files — all stored under `docs/intelligence-pipeline-deep-dive/`.
|
||||
|
||||
The document targets technical readers who want to understand how raw data enters the system, gets processed by AI agents, produces structured signals, accumulates into trend summaries, and ultimately drives autonomous trading decisions. Unlike the existing reference docs (`docs/services.md`, `docs/architecture-data-pipeline.md`), this deliverable is narrative and explanatory — it tells the story of data flowing through the platform end-to-end.
|
||||
|
||||
**Key design decision**: This is a documentation-only deliverable. No application code, database schemas, or infrastructure changes are involved. The output is purely Markdown files and Mermaid diagram files.
|
||||
|
||||
### Existing Documentation Landscape
|
||||
|
||||
The codebase already has several reference documents that this deep-dive complements:
|
||||
|
||||
| Document | Purpose | Style |
|
||||
|----------|---------|-------|
|
||||
| `docs/architecture-data-pipeline.md` | Queue topology, data store summary, Mermaid flow diagrams | Reference diagrams + tables |
|
||||
| `docs/llm-to-trade-pipeline.md` | End-to-end data flow from model output to trade | Narrative + tables + code blocks |
|
||||
| `docs/services.md` | Per-service configuration, tables, queues, behaviors | Reference manual |
|
||||
| `docs/ai-agents.md` | AI agent configuration, variants, A/B testing, API | Guide + reference |
|
||||
|
||||
The deep-dive document will reference these existing docs for readers who want deeper detail, while providing a cohesive narrative that connects all pipeline stages into a single story.
|
||||
|
||||
## Architecture
|
||||
|
||||
### File Organization
|
||||
|
||||
```
|
||||
docs/intelligence-pipeline-deep-dive/
|
||||
├── index.md
|
||||
├── 01-data-ingestion-and-preparation.md
|
||||
├── 02-ai-agent-processing-and-extraction.md
|
||||
├── 03-signal-scoring-and-weighted-signals.md
|
||||
├── 04-trend-aggregation-and-accumulating-signals.md
|
||||
├── 05-recommendation-generation.md
|
||||
├── 06-trading-decisions-and-execution.md
|
||||
└── diagrams/
|
||||
├── ingestion-to-extraction-flow.md
|
||||
├── three-layer-signal-merging.md
|
||||
├── recommendation-generation-flow.md
|
||||
├── trading-engine-decision-loop.md
|
||||
├── weighted-signal-computation.md
|
||||
└── trend-accumulation-escalation.md
|
||||
```
|
||||
|
||||
### Content Flow
|
||||
|
||||
Each page covers one pipeline stage and ends with a transitional paragraph previewing the next page. Cross-references between pages use relative Markdown links. Diagrams are stored as standalone Mermaid files in the `diagrams/` subdirectory and linked from the narrative pages (not embedded inline).
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
P1["Page 1\nData Ingestion"] --> P2["Page 2\nAI Extraction"]
|
||||
P2 --> P3["Page 3\nSignal Scoring"]
|
||||
P3 --> P4["Page 4\nTrend Aggregation"]
|
||||
P4 --> P5["Page 5\nRecommendations"]
|
||||
P5 --> P6["Page 6\nTrading Execution"]
|
||||
```
|
||||
|
||||
## Components and Interfaces
|
||||
|
||||
### Index File (`index.md`)
|
||||
|
||||
The index provides:
|
||||
- A brief introduction to the deep-dive document series
|
||||
- A numbered table of contents linking to all 6 pages
|
||||
- A diagrams section linking to all Mermaid diagram files
|
||||
- References to existing documentation for additional context
|
||||
|
||||
### Narrative Pages (01 through 06)
|
||||
|
||||
Each page follows a consistent structure:
|
||||
1. **Title and introduction** — what this stage does and why it matters
|
||||
2. **Narrative body** — explanatory prose describing the pipeline stage, referencing actual code modules (`services/extractor/main.py`), database tables (`document_impact_records`), Redis queues (`stonks:queue:extraction`), and Pydantic schemas (`ExtractionResult`)
|
||||
3. **Diagram references** — links to relevant Mermaid diagram files in `diagrams/`
|
||||
4. **Transition** — a closing paragraph that previews the next page
|
||||
|
||||
### Mermaid Diagram Files
|
||||
|
||||
Each diagram file contains:
|
||||
1. A brief title comment
|
||||
2. A single Mermaid code block
|
||||
3. Service labels include both human-readable names and Python module paths
|
||||
4. Queue labels use full Redis key patterns
|
||||
5. Database references use exact PostgreSQL table names
|
||||
|
||||
Minimum 6 diagrams covering:
|
||||
- **Ingestion-to-extraction flow**: Scheduler → Ingestion → Parser → Extractor, with queues and storage
|
||||
- **Three-layer signal merging**: Company, Macro, and Competitive layers converging into aggregation
|
||||
- **Recommendation generation flow**: Suppression → Eligibility → Thesis → Risk classification
|
||||
- **Trading engine decision loop**: Pre-trade checks → Position sizing → Order submission
|
||||
- **Weighted signal computation**: Component breakdown of the composite weight formula
|
||||
- **Trend accumulation and escalation**: How consecutive signals strengthen trends and escalate actions
|
||||
|
||||
### Page Content Mapping
|
||||
|
||||
| Page | Primary Code Modules | Key Database Tables | Key Queues |
|
||||
|------|---------------------|---------------------|------------|
|
||||
| 01 - Ingestion | `services/scheduler/app.py`, `services/ingestion/worker.py`, `services/parser/worker.py` | `documents`, `ingestion_runs`, `document_company_mentions` | `stonks:queue:ingestion`, `stonks:queue:parsing` |
|
||||
| 02 - AI Extraction | `services/extractor/main.py`, `services/extractor/client.py`, `services/extractor/prompts.py`, `services/extractor/schemas.py`, `services/extractor/event_classifier.py`, `services/shared/agent_config.py` | `document_intelligence`, `document_impact_records`, `global_events`, `macro_impact_records`, `ai_agents`, `agent_variants` | `stonks:queue:extraction`, `stonks:queue:macro_classification`, `stonks:queue:aggregation` |
|
||||
| 03 - Signal Scoring | `services/aggregation/scoring.py` | `document_impact_records`, `macro_impact_records`, `competitive_signal_records`, `risk_configs` | — |
|
||||
| 04 - Trend Aggregation | `services/aggregation/worker.py`, `services/aggregation/contradiction.py`, `services/aggregation/projection.py`, `services/aggregation/pattern_matcher.py`, `services/aggregation/signal_propagation.py` | `trend_windows`, `trend_history`, `trend_evidence`, `trend_projections` | `stonks:queue:aggregation`, `stonks:queue:recommendation` |
|
||||
| 05 - Recommendations | `services/recommendation/main.py`, `services/recommendation/suppression.py`, `services/recommendation/eligibility.py`, `services/recommendation/thesis_llm.py` | `recommendations`, `recommendation_evidence`, `risk_evaluations` | `stonks:queue:recommendation` |
|
||||
| 06 - Trading | `services/trading/engine.py`, `services/trading/position_sizer.py`, `services/trading/circuit_breaker.py`, `services/trading/reserve_pool.py`, `services/trading/risk_tier_controller.py`, `services/trading/stop_loss_manager.py` | `trading_decisions`, `orders`, `positions`, `portfolio_snapshots`, `reserve_pool_ledger`, `risk_tier_history`, `circuit_breaker_events` | `stonks:queue:broker_orders` |
|
||||
|
||||
## Data Models
|
||||
|
||||
This feature produces only documentation files. There are no new data models, database tables, or schema changes.
|
||||
|
||||
The narrative pages will reference existing data models from the codebase:
|
||||
|
||||
- **`WeightedSignal`** (`services/aggregation/scoring.py`) — document reference + composite weight + sentiment + impact
|
||||
- **`SignalWeight`** (`services/aggregation/scoring.py`) — breakdown of recency, credibility, novelty, confidence gate, market context multiplier
|
||||
- **`ScoringConfig`** (`services/aggregation/scoring.py`) — tunable parameters for signal scoring
|
||||
- **`ExtractionResult`** / **`CompanyImpact`** (`services/extractor/schemas.py`) — structured JSON output from document extraction
|
||||
- **`GlobalEventSchema`** (`services/extractor/event_classifier.py`) — macro event classification output
|
||||
- **`TrendSummary`** (`services/shared/schemas.py`) — rolling trend for a ticker across a time window
|
||||
- **`Recommendation`** (`services/shared/schemas.py`) — actionable trade recommendation
|
||||
- **`TradingDecision`** (`services/trading/engine.py`) — audit record of every trading evaluation
|
||||
|
||||
## Error Handling
|
||||
|
||||
Since this is a documentation-only deliverable, there is no runtime error handling to design. The primary quality concern is **accuracy** — ensuring that all code module paths, database table names, Redis queue keys, schema field names, and configuration values referenced in the narrative match the actual codebase.
|
||||
|
||||
### Accuracy Verification Strategy
|
||||
|
||||
1. **Code module paths**: Every module path referenced in the narrative (e.g., `services/aggregation/scoring.py`) must correspond to an existing file in the repository.
|
||||
2. **Database table names**: Table names must match those defined in `infra/migrations/` SQL files.
|
||||
3. **Redis queue keys**: Queue names must match constants in `services/shared/redis_keys.py`.
|
||||
4. **Schema class names**: Pydantic model names must match their definitions in `services/shared/schemas.py` and service-specific schema files.
|
||||
5. **Configuration values**: Environment variable names and default values must match `services/shared/config.py` and service-specific configuration.
|
||||
|
||||
### Cross-Reference Integrity
|
||||
|
||||
All inter-page links (e.g., `[Page 3](03-signal-scoring-and-weighted-signals.md)`) and diagram links (e.g., `[diagram](diagrams/ingestion-to-extraction-flow.md)`) must resolve to files that exist in the deliverable.
|
||||
|
||||
## Testing Strategy
|
||||
|
||||
**Property-based testing does not apply to this feature.** The deliverable is purely documentation — Markdown narrative pages and Mermaid diagram files. There are no functions, data transformations, or code logic to test.
|
||||
|
||||
### Why PBT Does Not Apply
|
||||
|
||||
- The output is static Markdown text, not executable code
|
||||
- There are no input/output functions to verify properties against
|
||||
- There is no data transformation logic that varies with input
|
||||
- The quality criteria (narrative coherence, codebase accuracy, cross-reference integrity) are best verified through manual review
|
||||
|
||||
### Verification Approach
|
||||
|
||||
1. **File existence check**: Verify all 6 page files, the index file, and all diagram files exist at the expected paths
|
||||
2. **Link integrity**: Verify all inter-page and diagram links resolve to existing files
|
||||
3. **Mermaid syntax**: Verify each diagram file contains valid Mermaid syntax by checking for proper `flowchart` or `graph` declarations
|
||||
4. **Codebase reference spot-checks**: Verify a sample of referenced module paths, table names, and queue keys against the actual codebase
|
||||
5. **Narrative flow**: Manual review to confirm each page ends with a transition to the next and the overall story is coherent
|
||||
@@ -0,0 +1,155 @@
|
||||
# Requirements Document
|
||||
|
||||
## Introduction
|
||||
|
||||
This specification defines a 6-page narrative deep-dive document (plus separate Mermaid diagram files) that explains the full intelligence-to-decision pipeline in Stonks Oracle. The document targets a technical reader who wants to understand how raw data enters the system, gets processed by AI agents, produces structured signals, accumulates into trend summaries, and ultimately drives autonomous trading decisions. Unlike the existing service reference and API docs, this deliverable is narrative and explanatory — it tells the story of data flowing through the platform end-to-end, referencing actual code modules, database tables, queue names, and schemas from the codebase.
|
||||
|
||||
## Glossary
|
||||
|
||||
- **Deep_Dive_Document**: The 6-page Markdown document delivered under `docs/intelligence-pipeline-deep-dive/`, consisting of pages 01 through 06 covering the full intelligence-to-decision pipeline.
|
||||
- **Mermaid_Diagram_File**: A standalone Markdown file containing a single Mermaid diagram block, stored alongside the narrative pages in `docs/intelligence-pipeline-deep-dive/diagrams/`.
|
||||
- **Pipeline**: The end-to-end data flow from external source ingestion through AI extraction, signal aggregation, recommendation generation, and autonomous trading execution.
|
||||
- **Signal_Layer**: One of three independent signal sources (Company, Macro, Competitive) that produce `WeightedSignal` objects merged by the Aggregation_Engine.
|
||||
- **Aggregation_Engine**: The `services/aggregation/` module that merges weighted signals from all three layers into `TrendSummary` objects across five time windows.
|
||||
- **Trading_Engine**: The `services/trading/engine.py` module that polls recommendations and executes autonomous paper trades through a multi-check decision loop.
|
||||
- **Extractor**: The `services/extractor/` module that uses Ollama LLM inference to produce structured JSON intelligence from documents.
|
||||
- **WeightedSignal**: The `services.aggregation.scoring.WeightedSignal` dataclass that pairs a document reference with a composite aggregation weight.
|
||||
- **TrendSummary**: The `services.shared.schemas.TrendSummary` Pydantic model representing a rolling trend for a ticker across a specific time window.
|
||||
- **Recommendation**: The `services.shared.schemas.Recommendation` Pydantic model representing an actionable trade recommendation with action, mode, confidence, thesis, and position sizing.
|
||||
- **Circuit_Breaker**: The `services/trading/circuit_breaker.py` safety mechanism that halts trading when risk thresholds (daily loss, single-position loss, volatility clustering) are breached.
|
||||
|
||||
## Requirements
|
||||
|
||||
### Requirement 1: Document Structure and File Organization
|
||||
|
||||
**User Story:** As a technical reader, I want the deep-dive organized into clearly separated pages with a consistent structure, so that I can navigate to specific pipeline stages without reading the entire document.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Deep_Dive_Document SHALL consist of exactly 6 Markdown page files named `01-data-ingestion-and-preparation.md` through `06-trading-decisions-and-execution.md`, stored under `docs/intelligence-pipeline-deep-dive/`.
|
||||
2. THE Deep_Dive_Document SHALL include an `index.md` file that provides a table of contents linking to all 6 pages and all Mermaid_Diagram_Files.
|
||||
3. WHEN a page references a Mermaid diagram, THE Deep_Dive_Document SHALL link to the corresponding Mermaid_Diagram_File stored in `docs/intelligence-pipeline-deep-dive/diagrams/` rather than embedding the diagram inline.
|
||||
4. THE Deep_Dive_Document SHALL include a minimum of 4 separate Mermaid_Diagram_Files covering: (a) the ingestion-to-extraction flow, (b) the three signal layers merging into aggregation, (c) the recommendation generation pipeline, and (d) the trading engine decision loop.
|
||||
5. WHEN a page references a code module, THE Deep_Dive_Document SHALL use the full Python module path (e.g., `services/extractor/prompts.py`) rather than abbreviated names.
|
||||
6. WHEN a page references a database table, THE Deep_Dive_Document SHALL use the exact table name as defined in the PostgreSQL schema (e.g., `document_impact_records`, `trend_windows`).
|
||||
7. WHEN a page references a Redis queue, THE Deep_Dive_Document SHALL use the full key pattern as defined in `services/shared/redis_keys.py` (e.g., `stonks:queue:extraction`).
|
||||
|
||||
### Requirement 2: Page 1 — Data Ingestion and Preparation
|
||||
|
||||
**User Story:** As a technical reader, I want to understand how raw data enters Stonks Oracle and gets prepared for AI processing, so that I can trace the origin of any signal back to its external source.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Deep_Dive_Document page 01 SHALL explain the four categories of input data: news articles (Polygon.io), SEC filings (EDGAR), market data (Polygon.io grouped daily and intraday bars), and macro/geopolitical events (macro news APIs).
|
||||
2. THE Deep_Dive_Document page 01 SHALL describe the Scheduler's role in orchestrating ingestion cycles, including cadence polling intervals per source type (`market_api`: 300s, `news_api`: 300s, `filings_api`: 3600s, `macro_news`: 600s), rate limiting, and exponential backoff.
|
||||
3. THE Deep_Dive_Document page 01 SHALL describe the Ingestion worker's adapter dispatch pattern, referencing the adapter classes (`PolygonMarketAdapter`, `PolygonNewsAdapter`, `SECEdgarAdapter`, `MacroNewsAdapter`) in `services/ingestion/`.
|
||||
4. THE Deep_Dive_Document page 01 SHALL explain content deduplication via Redis content-hash markers (`stonks:dedupe:*` with 24-hour TTL) and raw artifact storage in MinIO buckets (`stonks-raw-market`, `stonks-raw-news`, `stonks-raw-filings`).
|
||||
5. THE Deep_Dive_Document page 01 SHALL describe the Parser's role in converting raw HTML/text into normalized documents, including quality scoring with confidence levels (`high`, `medium`, `low`), company mention detection via alias matching, and the routing decision that sends `macro_event` documents to `stonks:queue:macro_classification` instead of `stonks:queue:extraction`.
|
||||
6. THE Deep_Dive_Document page 01 SHALL be written in narrative prose style with explanatory paragraphs, not as a reference table or bullet-point list.
|
||||
|
||||
### Requirement 3: Page 2 — AI Agent Processing and Structured Extraction
|
||||
|
||||
**User Story:** As a technical reader, I want to understand how the AI agents process documents and produce structured JSON output, so that I can evaluate the extraction quality and understand the schema contract.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Deep_Dive_Document page 02 SHALL explain the Document Intelligence Extractor agent (`document-extractor` slug), including its entry point (`services/extractor/main.py` → `services/extractor/client.py`), the system prompt, and the user prompt template built by `build_extraction_prompt()` in `services/extractor/prompts.py`.
|
||||
2. THE Deep_Dive_Document page 02 SHALL describe the `ExtractionResult` JSON schema with all fields (summary, companies array with ticker/sentiment/impact_score/impact_horizon/catalyst_type/key_facts/risks/evidence_spans, macro_themes, novelty_score, confidence, extraction_warnings), referencing `services/extractor/schemas.py`.
|
||||
3. THE Deep_Dive_Document page 02 SHALL explain the Global Event Classifier agent (`event-classifier` slug), including its entry point (`services/extractor/event_classifier.py`), the `GlobalEvent` output schema with event_types/severity/affected_regions/affected_sectors/affected_commodities/estimated_duration/confidence, and the anti-hallucination rules that prevent classifying company-specific news as macro events.
|
||||
4. THE Deep_Dive_Document page 02 SHALL describe the JSON repair pipeline (direct parse → markdown fence stripping → `json-repair` library fallback) and the structural plus semantic validation in `services/extractor/schemas.py`, including retry logic with exponential backoff.
|
||||
5. THE Deep_Dive_Document page 02 SHALL explain the `AgentConfigResolver` mechanism (`services/shared/agent_config.py`) that enables hot-swapping models and prompts via the `ai_agents` and `agent_variants` database tables with a 60-second TTL cache.
|
||||
6. THE Deep_Dive_Document page 02 SHALL describe how extraction results are persisted to `document_intelligence` (one row per document) and `document_impact_records` (one row per company mention), and how the extractor enqueues aggregation jobs to `stonks:queue:aggregation`.
|
||||
7. THE Deep_Dive_Document page 02 SHALL be written in narrative prose style with explanatory paragraphs, not as a reference table or bullet-point list.
|
||||
|
||||
### Requirement 4: Page 3 — Signal Scoring and the WeightedSignal Abstraction
|
||||
|
||||
**User Story:** As a technical reader, I want to understand how raw extraction output gets transformed into weighted signals for decision making, so that I can reason about why certain documents influence trends more than others.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Deep_Dive_Document page 03 SHALL explain the `WeightedSignal` dataclass (`services/aggregation/scoring.py`) and the composite weight formula: `combined = gate × recency × credibility × (1 + novelty_bonus) × market_context_multiplier`.
|
||||
2. THE Deep_Dive_Document page 03 SHALL describe each weight component in detail: confidence gate (threshold 0.2), recency decay (exponential half-life per window: intraday=2h, 1d=12h, 7d=72h, 30d=240h, 90d=720h), source credibility weighting (clamped [0.1, 1.0] with configurable exponent), novelty bonus (up to 25%), and market context multiplier (volatility boost up to 30%, volume surge boost 15%).
|
||||
3. THE Deep_Dive_Document page 03 SHALL explain how sentiment labels are mapped to numeric values (+1.0 positive, -1.0 negative, 0.0 neutral/mixed) via `sentiment_to_numeric()` and how the weighted sentiment average is computed across all signals.
|
||||
4. THE Deep_Dive_Document page 03 SHALL describe the three signal layers (Company, Macro, Competitive) and how each produces `WeightedSignal` objects that are concatenated into a single list before trend computation, with relative influence controlled by `MACRO_SIGNAL_WEIGHT` (0.3) and `COMPETITIVE_SIGNAL_WEIGHT` (0.2).
|
||||
5. THE Deep_Dive_Document page 03 SHALL explain the runtime toggle mechanism for macro and competitive layers via the `risk_configs` database table, including graceful degradation when a layer is disabled or fails.
|
||||
6. THE Deep_Dive_Document page 03 SHALL be written in narrative prose style with explanatory paragraphs, not as a reference table or bullet-point list.
|
||||
|
||||
### Requirement 5: Page 4 — Trend Aggregation and Accumulating Signals
|
||||
|
||||
**User Story:** As a technical reader, I want to understand how the aggregation engine merges multiple signals — including consecutive signals suggesting the same direction — to produce trend summaries that drive grander decisions, so that I can see how accumulating bearish or bullish evidence escalates the system's response.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Deep_Dive_Document page 04 SHALL explain how the Aggregation_Engine (`services/aggregation/worker.py`) computes `TrendSummary` objects across five time windows (intraday, 1d, 7d, 30d, 90d) by fetching impact records, macro impacts, and competitive signals for a ticker.
|
||||
2. THE Deep_Dive_Document page 04 SHALL describe the trend direction derivation rules: bullish (avg_sentiment ≥ 0.15), bearish (avg_sentiment ≤ -0.15), mixed (contradiction > 0.10 and |avg_sentiment| < 0.30), neutral (otherwise), referencing `derive_trend_direction()` in `services/aggregation/worker.py`.
|
||||
3. THE Deep_Dive_Document page 04 SHALL explain contradiction detection (`services/aggregation/contradiction.py`), including sentiment disagreement analysis and catalyst-level disagreement, and how the contradiction score (minority_weight / total_weight) penalizes trend confidence.
|
||||
4. THE Deep_Dive_Document page 04 SHALL describe how consecutive signals in the same direction accumulate to strengthen trend_strength and confidence, explaining the evidence ranking mechanism (`rank_evidence()`) that uses composite scoring (weight, impact, recency, confidence) and the confidence computation that rewards unique source count (caps at 15 sources for 0.8 contribution) and signal agreement (log₂ scaling, saturates around 7 unique sources).
|
||||
5. THE Deep_Dive_Document page 04 SHALL explain how accumulating bearish signals across multiple documents and time windows escalate the system's response — from a neutral hold to a bearish sell recommendation — and conversely how accumulating bullish signals escalate from watch to buy, using the trend strength and confidence thresholds from the eligibility rules.
|
||||
6. THE Deep_Dive_Document page 04 SHALL describe trend projections (`services/aggregation/projection.py`), including macro decay, momentum, driving factors, and divergence detection.
|
||||
7. THE Deep_Dive_Document page 04 SHALL describe persistence to `trend_windows` (upserted each cycle), `trend_history` (time-series snapshots), `trend_evidence` (per-document rankings), and `trend_projections`.
|
||||
8. THE Deep_Dive_Document page 04 SHALL be written in narrative prose style with explanatory paragraphs, not as a reference table or bullet-point list.
|
||||
|
||||
### Requirement 6: Page 5 — Recommendation Generation and Signal-to-Action Translation
|
||||
|
||||
**User Story:** As a technical reader, I want to understand how trend summaries are translated into actionable recommendations with risk classification and thesis generation, so that I can see the decision logic between aggregated intelligence and trading actions.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Deep_Dive_Document page 05 SHALL explain the data quality suppression layer (`services/recommendation/suppression.py`), including the six suppression checks (extraction confidence < 0.40, evidence staleness > 168h, source diversity < 1, extraction failure rate > 50%, valid document count < 2, data quality score < 0.30) and the safety suppressions for macro-only and pattern-only trend shifts.
|
||||
2. THE Deep_Dive_Document page 05 SHALL describe the eligibility evaluation (`services/recommendation/eligibility.py`), including gate checks (confidence ≥ 0.35, strength ≥ 0.10, contradiction ≤ 0.60, evidence ≥ 2, direction ≠ neutral), action mapping (BUY/SELL for strength ≥ 0.25, HOLD for weaker directional signals, WATCH otherwise), and mode escalation (informational → paper_eligible → live_eligible based on confidence and evidence thresholds).
|
||||
3. THE Deep_Dive_Document page 05 SHALL explain position sizing computation from signal quality: base 1% + confidence × strength scaling up to 10%, with contradiction penalty, evidence count penalty, and max loss percentage scaling.
|
||||
4. THE Deep_Dive_Document page 05 SHALL describe the two-layer thesis generation: deterministic thesis assembly from trend data, and optional LLM rewrite via the `thesis-rewriter` agent (`services/recommendation/thesis_llm.py`) for trading-eligible recommendations.
|
||||
5. THE Deep_Dive_Document page 05 SHALL explain risk classification (low/moderate/high/very_high) based on contradiction score, confidence, evidence count, and mode.
|
||||
6. THE Deep_Dive_Document page 05 SHALL describe persistence to `recommendations`, `recommendation_evidence`, and `risk_evaluations` tables.
|
||||
7. THE Deep_Dive_Document page 05 SHALL be written in narrative prose style with explanatory paragraphs, not as a reference table or bullet-point list.
|
||||
|
||||
### Requirement 7: Page 6 — Trading Engine Decisions and Execution
|
||||
|
||||
**User Story:** As a technical reader, I want to understand how the trading engine uses aggregated trend data to make buy/sell/hold decisions, including position sizing, risk evaluation, and circuit breakers, so that I can trace any trade back to its intelligence origin.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Deep_Dive_Document page 06 SHALL explain the Trading_Engine decision loop (`services/trading/engine.py`), including the five concurrent async tasks: decision loop (60s polling), stop-loss monitor, performance loop, risk tier scheduler, and rebalance scheduler.
|
||||
2. THE Deep_Dive_Document page 06 SHALL describe the pre-trade check sequence in order: circuit breaker check, trading window check, confidence gate (risk-tier minimum), deduplication, declining positions check, and max open positions check, explaining that the first failure short-circuits the evaluation.
|
||||
3. THE Deep_Dive_Document page 06 SHALL explain position sizing (`services/trading/position_sizer.py`), including confidence-based scaling with sample-size-dampened agreement scoring, risk tier adjustment (conservative/moderate/aggressive with specific parameter differences), correlation-aware diversification, sector exposure reduction, earnings proximity adjustment, and the absolute position cap.
|
||||
4. THE Deep_Dive_Document page 06 SHALL describe the Circuit_Breaker mechanism (`services/trading/circuit_breaker.py`), including the three trigger types (daily_loss with emergency drawdown threshold, single_position loss with ticker cooldown, volatility with stop-loss clustering detection), cooldown computation, and Redis state tracking (`stonks:trading:circuit_breaker:*`).
|
||||
5. THE Deep_Dive_Document page 06 SHALL explain the reserve pool mechanism (`services/trading/reserve_pool.py`): profit siphoning (default 20%), high-water mark rebalancing (30% threshold), emergency liquidation, and ledger tracking in `reserve_pool_ledger`.
|
||||
6. THE Deep_Dive_Document page 06 SHALL describe risk tier auto-adjustment (`services/trading/risk_tier_controller.py`), including the evaluation criteria (Sharpe ratio, drawdown, win rate) and the three tier configurations with their parameter differences (min confidence, max position %, stop-loss ATR multiplier, reward/risk ratio, max sector %, max portfolio heat).
|
||||
7. THE Deep_Dive_Document page 06 SHALL explain the order submission flow: `TradingDecision` persistence to `trading_decisions`, order job enqueue to `stonks:queue:broker_orders`, broker adapter risk evaluation, Alpaca paper trading submission, and the full audit trail from signal to broker response.
|
||||
8. THE Deep_Dive_Document page 06 SHALL be written in narrative prose style with explanatory paragraphs, not as a reference table or bullet-point list.
|
||||
|
||||
### Requirement 8: Mermaid Diagram Quality and Separation
|
||||
|
||||
**User Story:** As a technical reader, I want architecture diagrams in separate files that I can render independently, so that I can use them in presentations or embed them in other documents.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN a Mermaid_Diagram_File is created, THE Deep_Dive_Document SHALL store the diagram in a standalone Markdown file under `docs/intelligence-pipeline-deep-dive/diagrams/` with a descriptive filename (e.g., `ingestion-to-extraction-flow.md`).
|
||||
2. THE Deep_Dive_Document SHALL include at least 4 Mermaid_Diagram_Files: one for the ingestion-to-extraction pipeline, one for the three-layer signal merging, one for the recommendation generation flow, and one for the trading engine decision loop.
|
||||
3. WHEN a Mermaid diagram references a service, THE Mermaid_Diagram_File SHALL label the service with both its human-readable name and its Python module path (e.g., `Extractor\nservices/extractor/main.py`).
|
||||
4. WHEN a Mermaid diagram references a queue, THE Mermaid_Diagram_File SHALL use the full Redis key pattern (e.g., `stonks:queue:extraction`).
|
||||
5. WHEN a Mermaid diagram references a database table, THE Mermaid_Diagram_File SHALL use the exact PostgreSQL table name.
|
||||
|
||||
### Requirement 9: Narrative Style and Cross-Referencing
|
||||
|
||||
**User Story:** As a technical reader, I want the document to read as a coherent narrative rather than a reference manual, so that I can build a mental model of the full pipeline without jumping between disconnected sections.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Deep_Dive_Document SHALL use narrative prose with explanatory paragraphs as the primary writing style, reserving tables and bullet lists for structured data summaries only.
|
||||
2. WHEN a page references content covered in a different page, THE Deep_Dive_Document SHALL include a Markdown link to the relevant page and section.
|
||||
3. THE Deep_Dive_Document SHALL include transitional paragraphs at the end of each page that preview what the next page covers, creating a continuous narrative flow.
|
||||
4. THE Deep_Dive_Document SHALL reference the existing documentation where appropriate (e.g., `docs/services.md`, `docs/ai-agents.md`, `docs/architecture-data-pipeline.md`, `docs/llm-to-trade-pipeline.md`) for readers who want deeper reference-level detail.
|
||||
5. IF a concept is introduced for the first time, THEN THE Deep_Dive_Document SHALL provide a brief inline explanation before using the concept in subsequent discussion.
|
||||
|
||||
### Requirement 10: Codebase Accuracy
|
||||
|
||||
**User Story:** As a developer, I want the document to reference actual code modules, database tables, and queue names from the codebase, so that I can use the document as a reliable guide when navigating the source code.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Deep_Dive_Document SHALL reference code modules using paths that exist in the repository (e.g., `services/aggregation/scoring.py`, `services/trading/circuit_breaker.py`, `services/shared/schemas.py`).
|
||||
2. THE Deep_Dive_Document SHALL reference database tables using names that match the PostgreSQL schema as defined in `infra/migrations/`.
|
||||
3. THE Deep_Dive_Document SHALL reference Redis queue names using the constants defined in `services/shared/redis_keys.py` (e.g., `QUEUE_EXTRACTION`, `QUEUE_AGGREGATION`, `QUEUE_RECOMMENDATION`, `QUEUE_BROKER`).
|
||||
4. THE Deep_Dive_Document SHALL reference Pydantic schema classes using their actual class names from `services/shared/schemas.py` (e.g., `DocumentIntelligence`, `TrendSummary`, `Recommendation`, `GlobalEventSchema`, `CompanyImpact`).
|
||||
5. THE Deep_Dive_Document SHALL reference configuration environment variables using the exact names defined in `services/shared/config.py` and the service-specific configuration sections.
|
||||
@@ -0,0 +1,35 @@
|
||||
# Tasks — Intelligence Pipeline Deep Dive
|
||||
|
||||
## Task 1: Create directory structure and index file
|
||||
- [x] 1.1 Create `docs/intelligence-pipeline-deep-dive/` directory and `docs/intelligence-pipeline-deep-dive/diagrams/` subdirectory
|
||||
- [x] 1.2 Create `docs/intelligence-pipeline-deep-dive/index.md` with table of contents linking to all 6 pages and all diagram files, plus references to existing docs (`docs/services.md`, `docs/ai-agents.md`, `docs/architecture-data-pipeline.md`, `docs/llm-to-trade-pipeline.md`)
|
||||
|
||||
## Task 2: Create Mermaid diagram files
|
||||
- [x] 2.1 Create `docs/intelligence-pipeline-deep-dive/diagrams/ingestion-to-extraction-flow.md` — flowchart from Scheduler through Ingestion, Parser, to Extractor with all queues (`stonks:queue:ingestion`, `stonks:queue:parsing`, `stonks:queue:extraction`, `stonks:queue:macro_classification`), storage (MinIO buckets, PostgreSQL tables), and service module paths
|
||||
- [x] 2.2 Create `docs/intelligence-pipeline-deep-dive/diagrams/three-layer-signal-merging.md` — flowchart showing Company signals (`document_impact_records`), Macro signals (`macro_impact_records`), and Competitive signals (`competitive_signal_records`) each producing `WeightedSignal` objects that merge into the Aggregation engine (`services/aggregation/worker.py`)
|
||||
- [x] 2.3 Create `docs/intelligence-pipeline-deep-dive/diagrams/weighted-signal-computation.md` — diagram showing the composite weight formula components: confidence gate, recency decay, source credibility, novelty bonus, and market context multiplier
|
||||
- [x] 2.4 Create `docs/intelligence-pipeline-deep-dive/diagrams/trend-accumulation-escalation.md` — diagram showing how consecutive signals accumulate across time windows to escalate from neutral → watch → hold → buy/sell decisions
|
||||
- [x] 2.5 Create `docs/intelligence-pipeline-deep-dive/diagrams/recommendation-generation-flow.md` — flowchart from TrendSummary through data quality suppression, eligibility evaluation, thesis generation, risk classification, to recommendation persistence
|
||||
- [x] 2.6 Create `docs/intelligence-pipeline-deep-dive/diagrams/trading-engine-decision-loop.md` — flowchart showing the pre-trade check sequence (circuit breaker → trading window → confidence gate → dedup → declining positions → max positions), position sizing, and order submission to `stonks:queue:broker_orders`
|
||||
|
||||
## Task 3: Write Page 1 — Data Ingestion and Preparation
|
||||
- [x] 3.1 Write `docs/intelligence-pipeline-deep-dive/01-data-ingestion-and-preparation.md` covering: four input data categories (Polygon news, SEC EDGAR filings, Polygon market data, macro news APIs), Scheduler cadence polling (market_api: 300s, news_api: 300s, filings_api: 3600s, macro_news: 600s) with rate limiting and backoff, Ingestion worker adapter dispatch (`PolygonMarketAdapter`, `PolygonNewsAdapter`, `SECEdgarAdapter`, `MacroNewsAdapter`), content deduplication via Redis (`stonks:dedupe:*` with 24h TTL), raw artifact storage in MinIO (`stonks-raw-market`, `stonks-raw-news`, `stonks-raw-filings`), Parser role (HTML normalization, quality scoring, company mention detection, routing `macro_event` docs to `stonks:queue:macro_classification`). Written in narrative prose with links to diagrams and transition to Page 2.
|
||||
|
||||
## Task 4: Write Page 2 — AI Agent Processing and Structured Extraction
|
||||
- [x] 4.1 Write `docs/intelligence-pipeline-deep-dive/02-ai-agent-processing-and-extraction.md` covering: Document Intelligence Extractor agent (`document-extractor` slug, `services/extractor/main.py` → `services/extractor/client.py`, system prompt, `build_extraction_prompt()` in `services/extractor/prompts.py`), `ExtractionResult` JSON schema with all fields, Global Event Classifier agent (`event-classifier` slug, `services/extractor/event_classifier.py`, `GlobalEvent` schema, anti-hallucination rules), JSON repair pipeline (direct parse → fence stripping → `json-repair` fallback), structural + semantic validation in `services/extractor/schemas.py`, `AgentConfigResolver` mechanism (`services/shared/agent_config.py`, `ai_agents`/`agent_variants` tables, 60s TTL cache), persistence to `document_intelligence` and `document_impact_records`, aggregation job enqueue. Written in narrative prose with links to diagrams and transition to Page 3.
|
||||
|
||||
## Task 5: Write Page 3 — Signal Scoring and the WeightedSignal Abstraction
|
||||
- [x] 5.1 Write `docs/intelligence-pipeline-deep-dive/03-signal-scoring-and-weighted-signals.md` covering: `WeightedSignal` dataclass (`services/aggregation/scoring.py`), composite weight formula (`combined = gate × recency × credibility × (1 + novelty_bonus) × market_context_multiplier`), each component in detail (confidence gate threshold 0.2, recency decay half-lives per window, source credibility clamped [0.1, 1.0], novelty bonus up to 25%, market context volatility boost up to 30% and volume surge boost 15%), sentiment mapping via `sentiment_to_numeric()`, weighted sentiment average computation, three signal layers (Company, Macro weight 0.3, Competitive weight 0.2), runtime toggle via `risk_configs` table. Written in narrative prose with links to diagrams and transition to Page 4.
|
||||
|
||||
## Task 6: Write Page 4 — Trend Aggregation and Accumulating Signals
|
||||
- [x] 6.1 Write `docs/intelligence-pipeline-deep-dive/04-trend-aggregation-and-accumulating-signals.md` covering: Aggregation engine computing TrendSummary across 5 windows (intraday, 1d, 7d, 30d, 90d), trend direction rules (bullish ≥ 0.15, bearish ≤ -0.15, mixed, neutral), contradiction detection (`services/aggregation/contradiction.py`, minority_weight/total_weight), evidence ranking (`rank_evidence()` composite scoring), confidence computation (unique source count caps at 15, log₂ scaling saturates at 7 sources), how consecutive same-direction signals accumulate to escalate decisions (neutral → watch → hold → buy/sell), trend projections (`services/aggregation/projection.py`, macro decay, momentum, divergence detection), persistence to `trend_windows`, `trend_history`, `trend_evidence`, `trend_projections`. Written in narrative prose with links to diagrams and transition to Page 5.
|
||||
|
||||
## Task 7: Write Page 5 — Recommendation Generation and Signal-to-Action Translation
|
||||
- [x] 7.1 Write `docs/intelligence-pipeline-deep-dive/05-recommendation-generation.md` covering: data quality suppression (`services/recommendation/suppression.py`, 6 checks: extraction confidence < 0.40, staleness > 168h, source diversity < 1, failure rate > 50%, valid docs < 2, quality score < 0.30, plus macro-only and pattern-only safety), eligibility evaluation (`services/recommendation/eligibility.py`, gate checks, action mapping BUY/SELL/HOLD/WATCH, mode escalation informational/paper_eligible/live_eligible), position sizing (base 1% + confidence × strength up to 10%, contradiction and evidence penalties), thesis generation (deterministic + optional LLM rewrite via `thesis-rewriter` agent), risk classification (low/moderate/high/very_high), persistence to `recommendations`, `recommendation_evidence`, `risk_evaluations`. Written in narrative prose with links to diagrams and transition to Page 6.
|
||||
|
||||
## Task 8: Write Page 6 — Trading Engine Decisions and Execution
|
||||
- [x] 8.1 Write `docs/intelligence-pipeline-deep-dive/06-trading-decisions-and-execution.md` covering: Trading engine decision loop (`services/trading/engine.py`, 5 concurrent tasks: decision loop 60s, stop-loss monitor, performance loop, risk tier scheduler, rebalance scheduler), pre-trade check sequence (circuit breaker → trading window → confidence gate → dedup → declining positions → max positions), position sizing (`services/trading/position_sizer.py`, confidence scaling, risk tier adjustment, correlation diversification, sector exposure, earnings proximity, absolute cap), circuit breaker (`services/trading/circuit_breaker.py`, daily_loss, single_position, volatility triggers, cooldown, Redis state), reserve pool (`services/trading/reserve_pool.py`, profit siphoning 20%, high-water mark 30%, emergency liquidation), risk tier auto-adjustment (`services/trading/risk_tier_controller.py`, Sharpe/drawdown/win-rate evaluation, conservative/moderate/aggressive tiers), order submission flow (TradingDecision → `stonks:queue:broker_orders` → broker adapter → Alpaca). Written in narrative prose with links to diagrams.
|
||||
|
||||
## Task 9: Update index and verify cross-references
|
||||
- [x] 9.1 Update `docs/intelligence-pipeline-deep-dive/index.md` to ensure all page links and diagram links are correct and all files exist
|
||||
- [x] 9.2 Verify all inter-page links within narrative pages resolve correctly and all diagram references point to existing files
|
||||
@@ -0,0 +1 @@
|
||||
{"specId": "ce34e647-8d91-4295-a3c0-7b001abccdee", "workflowType": "requirements-first", "specType": "feature"}
|
||||
@@ -0,0 +1,69 @@
|
||||
# Stonks Oracle Intelligence Architecture Review
|
||||
|
||||
## Recommendation
|
||||
|
||||
Add generic OpenAI-compatible support, but implement it as a protocol/capability layer rather than a third vendor-specific branch. Keep Ollama native support. Convert the existing vLLM path into an OpenAI-compatible endpoint profile.
|
||||
|
||||
For the RTX 4070 Ti SUPER cluster, do not replace the current 9B model with one smaller all-purpose model. Retain the 9B model as a focused adjudicator and split routine work into CPU-first specialist stages:
|
||||
|
||||
1. Deterministic parsing and symbol-registry resolution.
|
||||
2. GLiNER2 Large for entities, event classes, relations, and evidence spans.
|
||||
3. FinBERT for company-specific financial sentiment probabilities.
|
||||
4. Retrieval-based novelty and duplicate detection.
|
||||
5. Calibrated confidence from observed field correctness.
|
||||
6. A stock-specific tabular model trained on realized abnormal returns for impact and horizon.
|
||||
7. The existing 9B Qwen-class model for ambiguous, causal, multi-company, or implied reasoning.
|
||||
|
||||
This preserves the current reasoning ceiling, reduces average GPU inference, improves evidence fidelity, and adds stock-specific intelligence that a general language model cannot obtain from article text alone.
|
||||
|
||||
## Option Review
|
||||
|
||||
| Option | Best use | Weakness | Production role |
|
||||
|---|---|---|---|
|
||||
| Current Qwen3.5-class 9B monolith | Broad zero-shot semantics and hard reasoning | Expensive per document; stochastic; self-scores confidence/novelty/impact; weak calibration | Keep as adjudicator, not universal extractor |
|
||||
| Qwen3.5 4B | Smaller generalist | Lower reasoning ceiling with same architectural weaknesses | Benchmark only; not preferred |
|
||||
| NuExtract 1.5 3.8B | Literal schema filling | Limited implicit market reasoning; adds another generative runtime | Optional benchmark/fallback |
|
||||
| NuExtract 1.5 Smol 1.7B | Compact long-form extraction | Still autoregressive and not a sentiment/impact model | Optional CPU/on-demand filing stage |
|
||||
| NuExtract Tiny 0.5B | Very small extraction experiments | Accuracy ceiling too low for authoritative trading inputs without task tuning | Research/fine-tuning baseline |
|
||||
| GLiNER2 Large 340M | CPU-first entities, classes, relations, spans | Needs calibration and task-specific tuning for best results | Primary fast-path specialist |
|
||||
| FinBERT | Financial positive/negative/neutral probabilities | Not an extractor or reasoner | Per-company evidence sentiment |
|
||||
| Hybrid specialist + 9B | Routine precision plus retained hard-case intelligence | More engineering and observability work | Recommended architecture |
|
||||
| Hybrid + historical impact model | Text intelligence plus actual market-response learning | Requires leakage-safe dataset and monitoring | Best end-state |
|
||||
|
||||
## Highest-Priority Existing Problems
|
||||
|
||||
1. `services/extractor/vllm_client.py` ignores the supplied schema and requests only a generic JSON object.
|
||||
2. The vLLM default extraction temperature is `0.7`.
|
||||
3. Unknown provider values silently route to Ollama.
|
||||
4. Documents are truncated to 8,000 characters.
|
||||
5. The model is asked to invent authoritative novelty, confidence, impact, and horizon values.
|
||||
6. Those self-scores directly affect aggregation weighting.
|
||||
7. Provider attribution is hardcoded to Ollama.
|
||||
8. The extractor processes one job at a time at the application layer.
|
||||
9. Endpoint/model defaults conflict across code, migrations, Helm, and the vLLM deployment.
|
||||
10. A tracked Helm override contains plaintext production-like credentials and requires immediate rotation.
|
||||
|
||||
## Expected Performance Shape
|
||||
|
||||
The following are design targets to validate, not promises:
|
||||
|
||||
- 60-80% of representative documents accepted through the CPU fast path after calibration.
|
||||
- 2x or greater reduction in GPU-seconds per accepted document.
|
||||
- Peak GPU memory near the current 9B deployment because no second generative model is permanently GPU-resident.
|
||||
- p50 latency substantially lower for routine documents.
|
||||
- p95 latency near the current model path for adjudicated documents.
|
||||
- Better exact-field and evidence accuracy from deterministic/specialist stages.
|
||||
- Same broad semantic ceiling because the 9B model remains available.
|
||||
- Better impact/horizon calibration once the historical outcome model is approved.
|
||||
|
||||
## Immediate Next Decision
|
||||
|
||||
The first implementation milestone should not be GLiNER integration. It should be:
|
||||
|
||||
1. Rotate exposed credentials.
|
||||
2. Establish the real runtime model/configuration.
|
||||
3. Fix strict JSON Schema output and temperature on the current 9B endpoint.
|
||||
4. Build the gold corpus and replay harness.
|
||||
5. Then implement the gateway and specialist shadow path.
|
||||
|
||||
That order creates a fair baseline and prevents the project from attributing simple request fixes to the new architecture.
|
||||
@@ -0,0 +1,792 @@
|
||||
# Design Document
|
||||
|
||||
## Overview
|
||||
|
||||
Intelligence Pipeline v3 replaces a monolithic "article to final trading-oriented JSON" request with a staged evidence and prediction architecture. The existing 9B model remains available, but its role changes from universal extractor and self-scorer to **semantic adjudicator** for the minority of documents that need broad language understanding.
|
||||
|
||||
The design intentionally chooses the best long-term architecture rather than the minimum code change:
|
||||
|
||||
- Generic OpenAI-compatible support is implemented as a capability-aware gateway, not another provider branch.
|
||||
- Explicit facts, entities, numbers, and sentiment are produced by CPU-first specialist components.
|
||||
- Novelty comes from retrieval and similarity.
|
||||
- Confidence comes from empirical calibration.
|
||||
- Impact and horizon come from a stock-specific model trained against realized outcomes.
|
||||
- The existing 9B vLLM model handles ambiguity, causality, implication, and conflicts.
|
||||
- Every field retains source evidence and model lineage.
|
||||
|
||||
## Repository Review Findings
|
||||
|
||||
The following findings materially shaped this design:
|
||||
|
||||
| Finding | Repository location | Consequence |
|
||||
|---|---|---|
|
||||
| The vLLM client receives a JSON Schema but sends only `response_format: {"type": "json_object"}`. | `services/extractor/vllm_client.py:63-91` | The server is not constraining generation to the actual schema. |
|
||||
| vLLM extraction defaults to temperature `0.7`. | `services/shared/config.py:64-71`, `services/shared/config.py:284-291` | Routine extraction is needlessly stochastic. |
|
||||
| Unknown provider values silently fall back to Ollama. | `services/extractor/llm_factory.py:1-6`, `services/extractor/llm_factory.py:47-67` | Configuration mistakes can invoke the wrong endpoint without failing. |
|
||||
| Long documents are truncated to the first 8,000 characters. | `services/extractor/prompts.py:102-105` | Filings, transcripts, and long articles can lose material facts. |
|
||||
| The prompt asks one model for summary, entities, relevance, sentiment, impact, horizon, novelty, confidence, and evidence. | `services/extractor/prompts.py:107-126` | Extraction, reasoning, prediction, and self-evaluation are coupled. |
|
||||
| The prompt supplies tracked tickers and invites inferred sector/theme exposure. | `services/extractor/prompts.py:85-98` | Explicit mentions and inferred exposure are mixed before evidence validation. |
|
||||
| Persisted provider attribution is hardcoded to `ollama`, including failures. | `services/extractor/worker.py:166-184`, `services/extractor/worker.py:227-244` | Audit and model-performance attribution are incorrect for vLLM. |
|
||||
| A single worker loop pops and processes one job at a time. | `services/extractor/main.py:438-468`, invocation near `services/extractor/main.py:628` | Application-level parallelism is constrained even if vLLM supports batching. |
|
||||
| Runtime refresh mutates a client's private `_config`. | `services/extractor/main.py:496-531` | The protocol does not expose lifecycle or reconfiguration cleanly. |
|
||||
| The thesis rewriter reimplements Ollama/vLLM branching. | `services/recommendation/thesis_llm.py:87-200` | Provider support is duplicated and will continue drifting. |
|
||||
| Model defaults conflict across Python config, database migrations, Helm values, and the standalone vLLM deployment. | `services/shared/config.py`, `infra/migrations`, `infra/helm/stonks-oracle/values.yaml`, `infra/kube-vllm/deployment.yaml` | The repository cannot prove which model is canonical at runtime. |
|
||||
| Model-produced novelty and confidence directly affect aggregation weight; model-produced impact is reused as sentiment strength and impact. | `services/aggregation/scoring.py:436-529`, `services/aggregation/worker.py:430-463` | Uncalibrated model self-scores can materially influence downstream signals. |
|
||||
| A tracked Helm override contains plaintext production-like credentials. | `infra/helm/stonks-oracle/values-live-math.yaml` | Immediate rotation and history remediation are required before feature work ships. |
|
||||
|
||||
The existing test suite around the LLM clients is useful. The focused provider tests passed after installing the declared dependencies plus the missing property-test dependency, but they encode current behavior and do not test true schema-constrained vLLM output.
|
||||
|
||||
## Decision Summary
|
||||
|
||||
### 1. Add generic OpenAI-compatible support
|
||||
|
||||
Yes, but do not add an `OpenAIClient` beside `VLLMClient` and `OllamaClient`. Rename the concept:
|
||||
|
||||
- `OllamaNativeClient` for `/api/chat` and Ollama-specific controls.
|
||||
- `OpenAICompatibleClient` for `/v1/chat/completions` and optionally `/v1/responses` after a separate compatibility gate.
|
||||
- `SpecialistHttpClient` for typed non-generative endpoints.
|
||||
|
||||
`vllm` becomes a profile alias whose protocol is `openai_chat`. Hosted OpenAI, LM Studio, SGLang, LocalAI, or another compatible server can be represented by endpoint capabilities rather than new `if provider == ...` branches.
|
||||
|
||||
Use direct `httpx` requests in the generic layer. This keeps the wire payload explicit, permits provider-specific `extra_body`, simplifies redacted request auditing, and avoids binding every compatible server to one SDK's assumptions.
|
||||
|
||||
### 2. Retain the 9B model, but stop using it for every stage
|
||||
|
||||
The RTX 4070 Ti SUPER remains dedicated to one 9B-class vLLM deployment. This preserves the current semantic ceiling and current peak VRAM class. The 9B model is invoked only for ambiguous cases and receives a compressed evidence packet rather than the raw entire document and ticker universe.
|
||||
|
||||
### 3. Use a CPU-first fast path
|
||||
|
||||
Run the following on CPU nodes:
|
||||
|
||||
- GLiNER2 Large candidate for entities, event classes, relations, and schema-oriented extraction.
|
||||
- FinBERT candidate for company-specific positive/negative/neutral probabilities.
|
||||
- Deterministic parsers and the existing symbol registry for numeric facts and ticker identity.
|
||||
- Compact embeddings plus fingerprints for novelty and deduplication.
|
||||
- A calibrated tabular impact model for market direction, magnitude, and horizon.
|
||||
|
||||
NuExtract 1.5 Smol is retained as an evaluated optional stage for long-form or hierarchical extraction. It is not made a second always-resident GPU model because the intended deployment should preserve the 9B model's GPU footprint and because GLiNER2 plus deterministic parsing may already cover most literal extraction.
|
||||
|
||||
## Architecture
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
A[Normalized document] --> B[Segmenter and offset map]
|
||||
B --> C[Deterministic candidates\ncompany aliases, tickers, numbers, dates]
|
||||
B --> D[GLiNER2 specialist\nentities, events, relations, facts]
|
||||
C --> E[Symbol resolver]
|
||||
D --> E
|
||||
E --> F[Evidence linker and verifier]
|
||||
F --> G[FinBERT per-company sentiment]
|
||||
F --> H[Novelty and dedup retrieval]
|
||||
G --> I[Confidence calibrator]
|
||||
H --> I
|
||||
I --> J{Fast-path acceptance?}
|
||||
J -->|yes| K[Approved evidence graph]
|
||||
J -->|no| L[9B Qwen adjudicator on vLLM]
|
||||
L --> M[Post-adjudication verifier]
|
||||
M --> K
|
||||
K --> N[Stock-specific impact model]
|
||||
N --> O[v3 intelligence records]
|
||||
O --> P[Compatibility adapter]
|
||||
P --> Q[Current aggregation and recommendation consumers]
|
||||
```
|
||||
|
||||
### Why this can be more intelligent without a larger footprint
|
||||
|
||||
A monolithic 9B model is broadly intelligent but is not necessarily the best estimator for every subproblem. The v3 design keeps that model for tasks requiring broad semantics while giving narrower jobs to components whose output can be calibrated and verified. The impact model adds information the language model does not have: observed historical market response. The result is not merely a smaller extractor; it is a system that combines textual reasoning with market-specific learned behavior.
|
||||
|
||||
## Component Design
|
||||
|
||||
### A. Inference Gateway
|
||||
|
||||
#### Package layout
|
||||
|
||||
```text
|
||||
services/shared/inference/
|
||||
├── protocol.py
|
||||
├── models.py
|
||||
├── registry.py
|
||||
├── router.py
|
||||
├── capabilities.py
|
||||
├── errors.py
|
||||
├── redaction.py
|
||||
└── clients/
|
||||
├── ollama_native.py
|
||||
├── openai_compatible.py
|
||||
└── specialist_http.py
|
||||
```
|
||||
|
||||
#### Core types
|
||||
|
||||
```python
|
||||
@dataclass(frozen=True)
|
||||
class ProviderCapabilities:
|
||||
chat_completions: bool
|
||||
responses_api: bool
|
||||
json_schema: bool
|
||||
json_object: bool
|
||||
seed: bool
|
||||
usage: bool
|
||||
max_completion_tokens: bool
|
||||
reasoning_toggle: bool
|
||||
model_listing: bool
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class InferenceTarget:
|
||||
endpoint_id: UUID
|
||||
deployment_id: UUID
|
||||
protocol: Literal["ollama_native", "openai_chat", "specialist_http"]
|
||||
base_url: str
|
||||
model: str
|
||||
capabilities: ProviderCapabilities
|
||||
auth_secret_ref: str | None
|
||||
extra_headers: Mapping[str, str]
|
||||
extra_body: Mapping[str, Any]
|
||||
|
||||
@dataclass
|
||||
class StructuredGenerationRequest:
|
||||
messages: list[ChatMessage]
|
||||
json_schema: dict[str, Any] | None
|
||||
max_output_tokens: int
|
||||
temperature: float = 0.0
|
||||
seed: int | None = 0
|
||||
timeout_seconds: float = 120.0
|
||||
trace_id: str = ""
|
||||
|
||||
@dataclass
|
||||
class InferenceResult:
|
||||
content: str
|
||||
parsed: dict[str, Any] | None
|
||||
target: InferenceTarget
|
||||
structured_mode: Literal["json_schema", "json_object", "prompt_only", "none"]
|
||||
latency_ms: int
|
||||
input_tokens: int | None
|
||||
output_tokens: int | None
|
||||
request_id: str | None
|
||||
repaired: bool
|
||||
retries: int
|
||||
```
|
||||
|
||||
#### OpenAI-compatible structured output
|
||||
|
||||
The client chooses the strongest declared mode:
|
||||
|
||||
1. `json_schema`: send the actual schema and strict mode.
|
||||
2. `json_object`: allow only if the deployment profile explicitly permits it.
|
||||
3. `prompt_only`: allow only for experiments or legacy fallback.
|
||||
|
||||
For current vLLM versions, the gateway should support both standard `response_format` JSON Schema and a configurable vLLM `structured_outputs` extra body because deployed versions may differ. The endpoint profile records which wire form passed its capability probe.
|
||||
|
||||
Example standard payload:
|
||||
|
||||
```json
|
||||
{
|
||||
"model": "AxionML/Qwen3.5-9B-NVFP4",
|
||||
"messages": [{"role": "system", "content": "..."}, {"role": "user", "content": "..."}],
|
||||
"temperature": 0,
|
||||
"max_tokens": 1536,
|
||||
"response_format": {
|
||||
"type": "json_schema",
|
||||
"json_schema": {
|
||||
"name": "adjudication_response",
|
||||
"strict": true,
|
||||
"schema": {}
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
The gateway validates the parsed response again locally. Wire constraints reduce malformed output; they do not replace semantic validation.
|
||||
|
||||
### B. Endpoint Registry
|
||||
|
||||
#### New tables
|
||||
|
||||
```sql
|
||||
CREATE TABLE inference_endpoints (
|
||||
id UUID PRIMARY KEY,
|
||||
name TEXT NOT NULL UNIQUE,
|
||||
protocol TEXT NOT NULL CHECK (protocol IN ('ollama_native','openai_chat','specialist_http')),
|
||||
base_url TEXT NOT NULL,
|
||||
auth_secret_ref TEXT,
|
||||
auth_scheme TEXT NOT NULL DEFAULT 'bearer',
|
||||
default_headers JSONB NOT NULL DEFAULT '{}',
|
||||
health_path TEXT,
|
||||
enabled BOOLEAN NOT NULL DEFAULT TRUE,
|
||||
revision INTEGER NOT NULL DEFAULT 1,
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
||||
updated_at TIMESTAMPTZ NOT NULL DEFAULT now()
|
||||
);
|
||||
|
||||
CREATE TABLE model_deployments (
|
||||
id UUID PRIMARY KEY,
|
||||
endpoint_id UUID NOT NULL REFERENCES inference_endpoints(id),
|
||||
served_model_name TEXT NOT NULL,
|
||||
display_name TEXT NOT NULL,
|
||||
capabilities JSONB NOT NULL,
|
||||
context_window INTEGER,
|
||||
max_output_tokens INTEGER,
|
||||
quantization TEXT,
|
||||
runtime_metadata JSONB NOT NULL DEFAULT '{}',
|
||||
enabled BOOLEAN NOT NULL DEFAULT TRUE,
|
||||
revision INTEGER NOT NULL DEFAULT 1,
|
||||
UNIQUE(endpoint_id, served_model_name)
|
||||
);
|
||||
|
||||
CREATE TABLE agent_stage_bindings (
|
||||
id UUID PRIMARY KEY,
|
||||
agent_id UUID NOT NULL REFERENCES ai_agents(id),
|
||||
stage TEXT NOT NULL,
|
||||
model_deployment_id UUID REFERENCES model_deployments(id),
|
||||
route_order INTEGER NOT NULL DEFAULT 0,
|
||||
routing_config JSONB NOT NULL DEFAULT '{}',
|
||||
is_active BOOLEAN NOT NULL DEFAULT TRUE,
|
||||
revision INTEGER NOT NULL DEFAULT 1,
|
||||
UNIQUE(agent_id, stage, route_order)
|
||||
);
|
||||
```
|
||||
|
||||
Authentication values are not stored in these tables. `auth_secret_ref` identifies a mounted secret or environment key understood by the deployment.
|
||||
|
||||
### C. Document Segmenter
|
||||
|
||||
The segmenter replaces the 8,000-character prefix truncation.
|
||||
|
||||
#### Output
|
||||
|
||||
```python
|
||||
class DocumentChunk(BaseModel):
|
||||
chunk_id: str
|
||||
document_id: UUID
|
||||
document_type: str
|
||||
section_path: list[str]
|
||||
speaker: str | None
|
||||
start_char: int
|
||||
end_char: int
|
||||
text: str
|
||||
overlap_left: int
|
||||
overlap_right: int
|
||||
boilerplate_score: float
|
||||
```
|
||||
|
||||
Suggested initial limits:
|
||||
|
||||
| Document type | Target chunk size | Overlap | Notes |
|
||||
|---|---:|---:|---|
|
||||
| News / press release | 700-1,000 tokens | 100 tokens | Preserve paragraph boundaries. |
|
||||
| Filing | 900-1,300 tokens | 150 tokens | Preserve headings and item sections. |
|
||||
| Transcript | 700-1,000 tokens | 100 tokens | Preserve speaker turns. |
|
||||
| Macro event | 500-800 tokens | 80 tokens | Favor compact event context. |
|
||||
|
||||
The final values are benchmark parameters, not hard-coded assumptions.
|
||||
|
||||
### D. Candidate Extraction and Symbol Resolution
|
||||
|
||||
Deterministic parsers generate high-precision candidates before specialist inference:
|
||||
|
||||
- Ticker tokens and exchange-qualified symbols.
|
||||
- Currency and number expressions, including `million`, `billion`, ranges, percentages, basis points, and per-share amounts.
|
||||
- Calendar and fiscal periods.
|
||||
- Comparison cues such as `up`, `down`, `beat`, `miss`, `raised`, `cut`, `above`, and `below`.
|
||||
- Company aliases from the symbol registry.
|
||||
|
||||
GLiNER2 receives focused schemas and returns spans for entities, event classes, relations, and structured facts. The resolver merges deterministic and specialist candidates using source offsets, aliases, and local context.
|
||||
|
||||
Explicit mentions and inferred exposures are different edge types:
|
||||
|
||||
```text
|
||||
Document --explicitly_mentions--> Company
|
||||
Event --directly_affects--> Company
|
||||
Event --inferred_exposure--> Company
|
||||
Company --competes_with--> Company
|
||||
Company --supplies--> Company
|
||||
```
|
||||
|
||||
Only explicit and verified direct effects enter the primary company extraction. Inferred exposure continues through the existing interpolation/propagation architecture with separate confidence and provenance.
|
||||
|
||||
### E. Sentiment Stage
|
||||
|
||||
FinBERT is run on company-linked evidence groups rather than the entire article. Each record contains:
|
||||
|
||||
```python
|
||||
class CompanySentiment(BaseModel):
|
||||
company_id: UUID
|
||||
evidence_ids: list[UUID]
|
||||
positive_probability: float
|
||||
negative_probability: float
|
||||
neutral_probability: float
|
||||
calibration_version: str
|
||||
model_deployment_id: UUID
|
||||
```
|
||||
|
||||
Mixed sentiment is computed from multiple evidence groups and disagreement. It is not an unconstrained fourth softmax label.
|
||||
|
||||
### F. Novelty Stage
|
||||
|
||||
Novelty becomes retrieval-based:
|
||||
|
||||
1. Compute an exact/near-duplicate fingerprint of normalized content.
|
||||
2. Embed document chunks and canonical company-event representations.
|
||||
3. Search a recent window in the vector index.
|
||||
4. Calculate document novelty and event novelty from nearest-neighbor similarity, duplicate count, source timing, and event identity.
|
||||
5. Store nearest matches for explainability.
|
||||
|
||||
A compact embedding model will be selected in the evaluation harness. The implementation must keep the embedding backend replaceable and must not entangle novelty scoring with the generative endpoint.
|
||||
|
||||
### G. Confidence and Routing
|
||||
|
||||
#### Confidence features
|
||||
|
||||
- Entity span score.
|
||||
- Alias-resolution margin between first and second candidate.
|
||||
- Numeric parser validity.
|
||||
- Evidence coverage.
|
||||
- Relation score.
|
||||
- Sentiment calibration confidence.
|
||||
- Cross-stage agreement.
|
||||
- Duplicate/novelty certainty.
|
||||
- Document completeness.
|
||||
- Document type and known hard-case patterns.
|
||||
|
||||
A calibration artifact maps these features to field-level correctness probabilities. Routing uses both calibrated confidence and hard rules.
|
||||
|
||||
#### Example adjudication triggers
|
||||
|
||||
```text
|
||||
UNRESOLVED_ALIAS
|
||||
MULTIPLE_PRIMARY_COMPANIES
|
||||
CONTRADICTORY_NUMERIC_FACTS
|
||||
CONFLICTING_SENTIMENT
|
||||
IMPLIED_CAUSAL_IMPACT
|
||||
GUIDANCE_VS_CONSENSUS_REQUIRES_REASONING
|
||||
MATERIAL_FIELD_MISSING
|
||||
EVIDENCE_COVERAGE_BELOW_THRESHOLD
|
||||
CALIBRATED_CONFIDENCE_BELOW_THRESHOLD
|
||||
LONG_DOCUMENT_CROSS_CHUNK_RELATION
|
||||
```
|
||||
|
||||
The desired initial target is 60-80 percent Fast_Path coverage after calibration. This is an evaluation target, not an assumed result.
|
||||
|
||||
### H. Adjudication Packet
|
||||
|
||||
The 9B model receives only the information necessary to resolve a specific ambiguity:
|
||||
|
||||
```python
|
||||
class AdjudicationPacket(BaseModel):
|
||||
document_id: UUID
|
||||
document_type: str
|
||||
question_codes: list[str]
|
||||
candidate_companies: list[CompanyCandidate]
|
||||
candidate_events: list[EventCandidate]
|
||||
candidate_facts: list[FactCandidate]
|
||||
candidate_sentiments: list[CompanySentiment]
|
||||
evidence_spans: list[EvidenceSpan]
|
||||
relevant_chunks: list[DocumentChunk]
|
||||
required_decisions: list[str]
|
||||
```
|
||||
|
||||
The adjudicator output does not contain final novelty, confidence, impact, or horizon. It resolves candidate identity, relationship, event interpretation, and supported qualitative direction. All output is evidence-linked and revalidated.
|
||||
|
||||
### I. Impact and Horizon Model
|
||||
|
||||
This is the highest-impact stock-specific change.
|
||||
|
||||
#### Problem decomposition
|
||||
|
||||
- **Text sentiment**: What tone or directional implication is supported by the document?
|
||||
- **Event identity**: What happened?
|
||||
- **Market impact**: How has this kind of event historically affected this kind of security in this market regime?
|
||||
- **Horizon**: Over what time window did the response usually manifest or decay?
|
||||
|
||||
The current model conflates these. v3 separates them.
|
||||
|
||||
#### Features
|
||||
|
||||
- Event class probability vector.
|
||||
- Company-specific sentiment probability vector.
|
||||
- Numeric magnitude and normalized surprise when consensus or prior value exists.
|
||||
- Source credibility and historical source accuracy.
|
||||
- Novelty and duplicate count.
|
||||
- Evidence coverage and extraction uncertainty.
|
||||
- Document type.
|
||||
- Company sector, industry, market-cap bucket, liquidity, and beta.
|
||||
- Pre-event volatility, volume regime, and broad market regime.
|
||||
- Whether the event is direct, second-order, confirmed, quoted, or speculative.
|
||||
|
||||
#### Labels
|
||||
|
||||
Generate leakage-safe targets at defined event timestamps:
|
||||
|
||||
- Signed abnormal return relative to an approved benchmark.
|
||||
- Absolute abnormal move.
|
||||
- Abnormal volume.
|
||||
- Direction labels for intraday, 1d, 7d, 30d, and 90d windows.
|
||||
- Time to peak response and decay where data quality supports it.
|
||||
|
||||
#### Model family
|
||||
|
||||
Start with a deterministic event-weight baseline plus a CPU tabular learner such as gradient-boosted trees. Calibrate class probabilities out-of-time. The model artifact and feature pipeline are versioned independently.
|
||||
|
||||
The compatibility adapter may initially map expected signed magnitude to current `impact_score` and the most probable horizon to current `impact_horizon`, but richer distributions remain available to new consumers.
|
||||
|
||||
### J. V3 Storage Schema
|
||||
|
||||
Suggested logical records:
|
||||
|
||||
```python
|
||||
class EvidenceSpan(BaseModel):
|
||||
id: UUID
|
||||
document_id: UUID
|
||||
chunk_id: str
|
||||
start_char: int
|
||||
end_char: int
|
||||
text: str
|
||||
checksum: str
|
||||
|
||||
class ExtractedEntity(BaseModel):
|
||||
id: UUID
|
||||
entity_type: str
|
||||
literal_text: str
|
||||
canonical_id: UUID | None
|
||||
evidence_id: UUID
|
||||
confidence: float
|
||||
derivation: str
|
||||
|
||||
class ExtractedFact(BaseModel):
|
||||
id: UUID
|
||||
fact_type: str
|
||||
subject_entity_id: UUID | None
|
||||
predicate: str
|
||||
literal_value: str
|
||||
normalized_value: dict | None
|
||||
period: dict | None
|
||||
evidence_ids: list[UUID]
|
||||
confidence: float
|
||||
derivation: str
|
||||
|
||||
class CompanySignalCandidate(BaseModel):
|
||||
company_id: UUID
|
||||
relevance_probability: float
|
||||
event_probabilities: dict[str, float]
|
||||
sentiment_probabilities: dict[str, float]
|
||||
direction_probabilities: dict[str, float]
|
||||
horizon_probabilities: dict[str, float]
|
||||
expected_magnitude: float | None
|
||||
evidence_ids: list[UUID]
|
||||
routing_reasons: list[str]
|
||||
adjudicated: bool
|
||||
|
||||
class StageLineage(BaseModel):
|
||||
stage: str
|
||||
endpoint_id: UUID | None
|
||||
deployment_id: UUID | None
|
||||
model_version: str | None
|
||||
schema_version: str
|
||||
calibration_version: str | None
|
||||
started_at: datetime
|
||||
duration_ms: int
|
||||
status: str
|
||||
```
|
||||
|
||||
### K. Compatibility Adapter
|
||||
|
||||
The adapter creates current records without discarding v3 provenance:
|
||||
|
||||
| Current field | V3 source |
|
||||
|---|---|
|
||||
| `summary` | Deterministic template or optional 9B narrative generated from approved facts. |
|
||||
| `macro_themes` | Approved event/theme classes. |
|
||||
| `novelty_score` | Retrieval-derived event/document novelty. |
|
||||
| `confidence` | Calibrated record correctness probability. |
|
||||
| `ticker` | Canonical symbol registry resolution. |
|
||||
| `relevance` | Calibrated direct-relevance probability. |
|
||||
| `sentiment` | Company-specific calibrated distribution mapped to legacy enum. |
|
||||
| `impact_score` | Approved impact-model magnitude mapped to legacy range. |
|
||||
| `impact_horizon` | Most probable approved horizon. |
|
||||
| `catalyst_type` | Versioned event taxonomy mapping. |
|
||||
| `evidence_spans` | Exact source spans. |
|
||||
|
||||
The adapter marks `model_provider = 'hybrid'` and stores complete stage lineage separately. No provider identity is hardcoded.
|
||||
|
||||
## Deployment Design for RTX 4070 Ti SUPER Cluster
|
||||
|
||||
### GPU deployment
|
||||
|
||||
One vLLM pod remains on the RTX 4070 Ti SUPER:
|
||||
|
||||
```yaml
|
||||
resources:
|
||||
limits:
|
||||
nvidia.com/gpu: 1
|
||||
nodeSelector:
|
||||
accelerator: rtx-4070-ti-super
|
||||
args:
|
||||
- --model
|
||||
- AxionML/Qwen3.5-9B-NVFP4
|
||||
- --served-model-name
|
||||
- stonks-adjudicator-9b
|
||||
- --max-model-len
|
||||
- "8192"
|
||||
- --max-num-seqs
|
||||
- "8"
|
||||
- --gpu-memory-utilization
|
||||
- "0.80"
|
||||
- --structured-outputs-config.backend
|
||||
- auto
|
||||
```
|
||||
|
||||
Exact flags must match the pinned vLLM version. The deployment test must verify strict schema output before promotion.
|
||||
|
||||
### CPU specialist deployment
|
||||
|
||||
```yaml
|
||||
replicas: 2
|
||||
resources:
|
||||
requests:
|
||||
cpu: "2"
|
||||
memory: 4Gi
|
||||
limits:
|
||||
cpu: "6"
|
||||
memory: 10Gi
|
||||
```
|
||||
|
||||
Initial pod contents:
|
||||
|
||||
- GLiNER2 Large.
|
||||
- FinBERT.
|
||||
- Tokenizers and deterministic parsers.
|
||||
- Optional embedding model.
|
||||
|
||||
NuExtract 1.5 Smol should run as a separate benchmark or on-demand CPU deployment so its value can be measured independently.
|
||||
|
||||
### Queue topology
|
||||
|
||||
```text
|
||||
extraction.incoming
|
||||
-> intelligence.router
|
||||
-> extraction.fast
|
||||
-> extraction.adjudication
|
||||
-> extraction.persist
|
||||
-> extraction.review
|
||||
```
|
||||
|
||||
The router owns document state transitions. Workers use leases and idempotency keys so a retry cannot create duplicate intelligence records.
|
||||
|
||||
### Concurrency
|
||||
|
||||
- Fast path: configurable worker pool, initially 4-8 concurrent documents per pod.
|
||||
- Specialist API: micro-batching bounded by maximum wait time.
|
||||
- Adjudicator: application semaphore aligned with vLLM `max-num-seqs` and measured KV-cache behavior.
|
||||
- Persistence: independent bounded pool.
|
||||
|
||||
## OpenAI-Compatible Support Details
|
||||
|
||||
### Profiles
|
||||
|
||||
| Profile | Protocol | Typical use |
|
||||
|---|---|---|
|
||||
| `ollama` | `ollama_native` | Existing Ollama endpoint. |
|
||||
| `vllm` | `openai_chat` | Backward-compatible alias using vLLM capability profile. |
|
||||
| `openai` | `openai_chat` | Hosted OpenAI endpoint with secret reference and egress policy. |
|
||||
| `openai_compatible` | `openai_chat` | Any explicitly configured compatible server. |
|
||||
| `specialist` | `specialist_http` | Typed GLiNER/FinBERT service. |
|
||||
|
||||
### Capability probes
|
||||
|
||||
A deployment activation test performs:
|
||||
|
||||
1. Health request.
|
||||
2. Model listing if supported.
|
||||
3. Minimal chat request.
|
||||
4. Strict JSON Schema request.
|
||||
5. Usage metadata check.
|
||||
6. Seed/determinism check if declared.
|
||||
7. Maximum output field compatibility check.
|
||||
|
||||
Probe results are stored with timestamp and software version. A failed capability cannot be enabled merely by selecting it in the UI.
|
||||
|
||||
### Egress and data policy
|
||||
|
||||
Endpoint profiles include data-handling classification:
|
||||
|
||||
```text
|
||||
local_private
|
||||
cluster_private
|
||||
approved_external
|
||||
forbidden_for_sensitive_docs
|
||||
```
|
||||
|
||||
External endpoints are disabled by default. Routing to an approved external endpoint requires both an active binding and a document policy permitting egress.
|
||||
|
||||
## Evaluation Strategy
|
||||
|
||||
### Gold corpus
|
||||
|
||||
Build a minimum initial corpus of 1,000 human-reviewed documents, stratified across:
|
||||
|
||||
- News, filings, transcripts, and press releases.
|
||||
- Single-company and multi-company stories.
|
||||
- Earnings beats/misses and guidance changes.
|
||||
- M&A, legal, regulatory, product, supply-chain, rating, management, and macro events.
|
||||
- Explicit facts versus implied consequences.
|
||||
- Short and long documents.
|
||||
- Duplicate and recycled stories.
|
||||
|
||||
### Compared systems
|
||||
|
||||
1. Current production path with current model and current prompt.
|
||||
2. Current 9B model with corrected temperature and strict JSON Schema.
|
||||
3. GLiNER2 + deterministic extraction.
|
||||
4. GLiNER2 + deterministic extraction + FinBERT.
|
||||
5. Optional NuExtract 1.5 Smol extraction path.
|
||||
6. Full v3 fast path.
|
||||
7. Full v3 with 9B adjudication.
|
||||
|
||||
This separation prevents architecture gains from being confused with a simple fix to the current vLLM request.
|
||||
|
||||
### Metrics
|
||||
|
||||
- Company/ticker precision, recall, F1.
|
||||
- Event macro-F1 and per-class F1.
|
||||
- Numeric exact match with tolerance-aware normalization.
|
||||
- Relation F1.
|
||||
- Evidence support and offset validity.
|
||||
- Company-specific sentiment macro-F1.
|
||||
- Confidence ECE and Brier score.
|
||||
- Unsupported-claim rate.
|
||||
- JSON/schema failure rate.
|
||||
- Fast-path coverage.
|
||||
- Adjudication reason distribution.
|
||||
- p50/p95 latency and documents per minute.
|
||||
- CPU-seconds, GPU-seconds, tokens, and peak GPU memory per document.
|
||||
- Impact-model direction accuracy, calibration, rank correlation, and out-of-time error by horizon.
|
||||
|
||||
### Promotion sequence
|
||||
|
||||
1. Correct current vLLM schema constraints and temperature; establish baseline.
|
||||
2. Run v3 offline replay.
|
||||
3. Run v3 shadow mode.
|
||||
4. Enable v3 for audit-only UI.
|
||||
5. Canary compatibility outputs for a small percentage of non-trading downstream traffic.
|
||||
6. Canary signal influence with automatic rollback.
|
||||
7. Promote by document type and confidence tier.
|
||||
8. Retire v2 only after a separately reviewed milestone.
|
||||
|
||||
## Security Design
|
||||
|
||||
### Immediate blocker
|
||||
|
||||
The repository contains plaintext production-like credentials in a tracked Helm values file. Treat them as compromised:
|
||||
|
||||
1. Rotate the database, object-store, Redis, broker, and market-data credentials.
|
||||
2. Disable or replace old keys.
|
||||
3. Remove secret values from tracked files.
|
||||
4. Purge historical values using an approved Git history rewrite process.
|
||||
5. Migrate to an external secret manager.
|
||||
6. Add secret scanning to local hooks and CI.
|
||||
7. Audit access logs for the affected credentials.
|
||||
|
||||
No values are reproduced in this specification.
|
||||
|
||||
### Inference security
|
||||
|
||||
- Secrets are resolved only at runtime.
|
||||
- Request logging records hashes and metadata, not authorization values.
|
||||
- Raw source text is not logged at INFO level.
|
||||
- External endpoint use is policy-gated.
|
||||
- Stored raw prompts and responses use restricted object-store buckets and retention policies.
|
||||
- Provider errors are normalized to avoid echoing secret-bearing response headers.
|
||||
|
||||
## Testing Strategy
|
||||
|
||||
### Unit tests
|
||||
|
||||
- Endpoint capability selection.
|
||||
- Strict schema payload construction.
|
||||
- Header and error redaction.
|
||||
- Segment offset round trips.
|
||||
- Numeric normalization.
|
||||
- Alias resolution and ambiguity margins.
|
||||
- Evidence linkage.
|
||||
- Compatibility mappings.
|
||||
- Confidence feature construction.
|
||||
- Routing rules.
|
||||
|
||||
### Property-based tests
|
||||
|
||||
- Every Evidence_Span round-trips to identical source text.
|
||||
- Normalization never changes the literal stored value.
|
||||
- Unknown providers always fail closed.
|
||||
- Credentials never appear in serialized errors or logs.
|
||||
- Compatibility mappings remain bounded in legacy field ranges.
|
||||
- Reprocessing the same document and model versions is idempotent.
|
||||
- Route decisions are deterministic for identical calibrated inputs.
|
||||
|
||||
### Contract tests
|
||||
|
||||
- Ollama native endpoint.
|
||||
- vLLM OpenAI-compatible endpoint.
|
||||
- Mock hosted OpenAI-compatible endpoint.
|
||||
- Specialist service schemas.
|
||||
- Capability probe behavior across supported structured-output modes.
|
||||
|
||||
### Integration tests
|
||||
|
||||
- Full document through fast path.
|
||||
- Full document through adjudication path.
|
||||
- Adjudicator outage with safe fast-path handling.
|
||||
- Long filing crossing multiple chunks.
|
||||
- Multi-company article with opposing sentiment.
|
||||
- Duplicate story and novelty calculation.
|
||||
- Rollback from v3 to v2.
|
||||
|
||||
### Load tests
|
||||
|
||||
- CPU specialist batching.
|
||||
- Queue backpressure.
|
||||
- vLLM concurrent adjudication.
|
||||
- Peak 4070 Ti SUPER memory.
|
||||
- End-to-end throughput at representative article arrival rates.
|
||||
|
||||
## Migration Plan
|
||||
|
||||
### Phase 0: Security and baseline
|
||||
|
||||
Rotate secrets, add scanning, pin current runtime configuration, and benchmark the existing path.
|
||||
|
||||
### Phase 1: Provider foundation
|
||||
|
||||
Add the Inference Gateway and registry. Convert current extractor, classifier, and thesis rewriter. Keep behavior otherwise equivalent.
|
||||
|
||||
### Phase 2: Correct current generative extraction
|
||||
|
||||
Use strict schema-constrained vLLM output, temperature zero, accurate provider lineage, and bounded output sizes. This produces a fair current baseline.
|
||||
|
||||
### Phase 3: V3 data and specialist shadow path
|
||||
|
||||
Add segmentation, v3 storage, deterministic extraction, GLiNER2, FinBERT, novelty, confidence, and audit UI. Persist shadow outputs only.
|
||||
|
||||
### Phase 4: Adjudication and impact model
|
||||
|
||||
Add confidence routing, focused 9B adjudication, historical outcome features, deterministic impact baseline, and trained impact model.
|
||||
|
||||
### Phase 5: Canary and promotion
|
||||
|
||||
Enable compatibility outputs by percentage and document type, then gradually allow v3 signals into aggregation.
|
||||
|
||||
### Phase 6: Fine-tuning and cleanup
|
||||
|
||||
Fine-tune specialist models from reviewed cases, increase fast-path coverage, then remove deprecated v2 code in a separate change.
|
||||
|
||||
## Risks and Mitigations
|
||||
|
||||
| Risk | Mitigation |
|
||||
|---|---|
|
||||
| Specialist model misses implicit meaning. | Retain 9B adjudication with calibrated routing. |
|
||||
| Complexity creates more failure modes. | Typed stage contracts, idempotent queues, stage-level metrics, and safe fallback. |
|
||||
| Fast-path confidence is overestimated. | Held-out calibration, conservative thresholds, and shadow review. |
|
||||
| Impact model learns leakage or regime artifacts. | Event-time feature snapshots, out-of-time validation, per-regime monitoring, and immutable predictions. |
|
||||
| OpenAI-compatible servers differ subtly. | Capability probes and profile-specific wire settings, not optimistic assumptions. |
|
||||
| A second model increases memory. | Keep specialists on CPU and make NuExtract optional/on-demand. |
|
||||
| Existing downstream code assumes one model output. | Compatibility adapter and additive migrations. |
|
||||
| Reviewer labels become inconsistent. | Annotation guide, double review for hard cases, and inter-annotator agreement tracking. |
|
||||
@@ -0,0 +1,314 @@
|
||||
# Requirements Document
|
||||
|
||||
## Introduction
|
||||
|
||||
Stonks Oracle currently asks a general-purpose generative model to perform entity discovery, ticker attribution, fact extraction, event classification, sentiment analysis, novelty estimation, confidence estimation, impact scoring, horizon selection, evidence quoting, and summarization in one response. That design is convenient, but it couples factual extraction to generative sampling and allows uncalibrated model self-assessments to influence signal weighting.
|
||||
|
||||
This specification introduces **Intelligence Pipeline v3**, a multi-stage, evidence-grounded inference system that preserves the current 9B model's reasoning ability for genuinely ambiguous documents while moving routine extraction, sentiment, novelty, confidence, and impact estimation into specialized and calibratable components. The target deployment retains the existing RTX 4070 Ti SUPER vLLM footprint and uses CPU-first specialist services for the fast path.
|
||||
|
||||
The specification also replaces provider-specific branching with a capability-aware inference gateway supporting Ollama native endpoints and generic OpenAI-compatible endpoints, including vLLM and hosted OpenAI-compatible services.
|
||||
|
||||
## Goals
|
||||
|
||||
1. Match or exceed the current 9B pipeline's field-level accuracy and reasoning ceiling.
|
||||
2. Reduce average GPU work per document without increasing peak GPU memory materially.
|
||||
3. Make every extracted fact traceable to evidence in the source document.
|
||||
4. Replace model-generated confidence, novelty, impact, and horizon values with calibrated or deterministic values.
|
||||
5. Support generic OpenAI-compatible inference without adding another duplicated provider branch.
|
||||
6. Establish a measurable benchmark, shadow rollout, and promotion process.
|
||||
7. Preserve downstream compatibility while the v2 schema and database consumers are migrated.
|
||||
|
||||
## Non-Goals
|
||||
|
||||
1. Replacing the existing recommendation, risk, or trading engines in one release.
|
||||
2. Removing the current 9B model before the v3 pipeline passes promotion gates.
|
||||
3. Treating backtest profit alone as proof of extraction correctness.
|
||||
4. Sending credentials or proprietary documents to external providers by default.
|
||||
5. Requiring a second GPU-resident generative model.
|
||||
|
||||
## Glossary
|
||||
|
||||
- **Inference_Gateway**: Shared client and routing layer that invokes Ollama-native, OpenAI-compatible, and specialist inference endpoints through one typed interface.
|
||||
- **Endpoint_Profile**: Persisted endpoint configuration containing protocol, URL, authentication reference, capabilities, and health settings.
|
||||
- **Model_Deployment**: A model served by an Endpoint_Profile with declared capabilities and limits.
|
||||
- **Pipeline_Stage**: One step in Intelligence Pipeline v3, such as segmentation, extraction, sentiment, verification, novelty, adjudication, or impact prediction.
|
||||
- **Fast_Path**: CPU-first processing that completes without invoking the 9B generative model.
|
||||
- **Adjudication_Path**: Processing that invokes the 9B model because evidence is ambiguous, contradictory, incomplete, or semantically complex.
|
||||
- **Evidence_Span**: Exact source text plus stable character offsets and a chunk identifier.
|
||||
- **Candidate**: A proposed entity, fact, event, sentiment, or relation before validation and calibration.
|
||||
- **Calibrated_Confidence**: Probability-like confidence derived from validation data, not a number supplied by a generative model.
|
||||
- **Impact_Model**: A lightweight supervised model that estimates signed market impact and horizon from extracted features and historical outcomes.
|
||||
- **Compatibility_Adapter**: Mapper from v3 records to the current v2 `document_intelligence` and `document_impact_records` structures.
|
||||
- **Gold_Corpus**: Human-reviewed documents and field-level labels used for acceptance testing.
|
||||
- **Shadow_Mode**: Running v3 alongside the current pipeline without allowing v3 outputs to affect production decisions.
|
||||
|
||||
## Requirements
|
||||
|
||||
### Requirement 1: Secure Baseline and Credential Remediation
|
||||
|
||||
**User Story:** As an operator, I want repository and deployment credentials handled through secret stores, so that model-pipeline improvements do not ship on top of exposed credentials.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Team SHALL rotate every credential currently stored as plaintext in tracked repository files before deploying Intelligence Pipeline v3.
|
||||
2. THE Repository SHALL remove plaintext database, object-store, Redis, broker, and market-data credentials from tracked Helm values and Git history.
|
||||
3. THE Deployment SHALL reference credentials through Kubernetes Secrets populated by External Secrets, SOPS, Sealed Secrets, or an equivalent approved mechanism.
|
||||
4. THE CI_Pipeline SHALL run secret scanning on pull requests and protected branches.
|
||||
5. IF secret scanning detects a high-confidence credential, THEN THE CI_Pipeline SHALL fail before packaging or deployment.
|
||||
6. THE Documentation SHALL record the rotation date and affected secret names without recording secret values.
|
||||
|
||||
### Requirement 2: Capability-Aware Generic Inference Gateway
|
||||
|
||||
**User Story:** As a developer, I want one inference abstraction that supports Ollama and generic OpenAI-compatible services, so that endpoints can be changed without duplicating business logic.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Inference_Gateway SHALL support the protocols `ollama_native`, `openai_chat`, and `specialist_http`.
|
||||
2. THE Inference_Gateway SHALL treat `vllm` as a backward-compatible profile alias for `openai_chat`, not as a separate client implementation.
|
||||
3. WHEN an `openai_chat` request requires structured output and the endpoint declares `json_schema` support, THE Inference_Gateway SHALL send the complete supplied JSON Schema in strict structured-output mode.
|
||||
4. WHEN an endpoint supports only JSON-object mode, THE Inference_Gateway SHALL use JSON-object mode only when that fallback is explicitly enabled for the Model_Deployment.
|
||||
5. WHEN neither schema nor JSON-object constraints are supported, THE Inference_Gateway SHALL use prompt-only JSON generation only when explicitly enabled and SHALL mark the response as unconstrained.
|
||||
6. IF a provider or protocol value is unknown, THEN THE Inference_Gateway SHALL fail closed with a configuration error and SHALL NOT silently route to Ollama.
|
||||
7. THE Inference_Gateway SHALL support configurable base URL, request path, API-key secret reference, authorization scheme, additional headers, timeouts, retries, concurrency limit, and provider-specific extra request fields.
|
||||
8. THE Inference_Gateway SHALL redact authentication values and configured sensitive headers from logs, traces, and stored request snapshots.
|
||||
9. THE Inference_Gateway SHALL expose typed response metadata including endpoint ID, deployment ID, model name, protocol, request ID, latency, token usage, structured-output mode, retry count, and error category.
|
||||
10. THE Inference_Gateway SHALL provide health and capability probes and cache their results with a bounded TTL.
|
||||
11. WHEN endpoint capabilities are changed or a probe fails, THE Router SHALL invalidate the cached capability record before the next invocation.
|
||||
12. THE Existing thesis rewriter, event classifier, and document extractor SHALL use the same Inference_Gateway rather than implementing separate Ollama/vLLM branches.
|
||||
|
||||
### Requirement 3: Canonical Endpoint and Model Registry
|
||||
|
||||
**User Story:** As an operator, I want the database and UI to identify exactly which endpoint and model serve each stage, so that environment, migration, Helm, and runtime defaults cannot drift silently.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Database SHALL store `inference_endpoints`, `model_deployments`, and `agent_stage_bindings` as canonical runtime records.
|
||||
2. EACH Inference_Endpoint SHALL include name, protocol, base URL, authentication secret reference, health path, default headers, enabled state, and timestamps.
|
||||
3. EACH Model_Deployment SHALL include endpoint ID, served model name, display name, capabilities, context limit, output limit, quantization, structured-output modes, and enabled state.
|
||||
4. EACH Agent_Stage_Binding SHALL map an agent and pipeline stage to one or more ordered Model_Deployments plus routing configuration.
|
||||
5. WHEN runtime configuration is resolved, THE Service SHALL record the exact endpoint, deployment, and binding revision used.
|
||||
6. THE API SHALL validate endpoint URLs, protocol values, capability declarations, and model names before activation.
|
||||
7. THE UI SHALL use controlled protocol and endpoint selections rather than an unrestricted provider text field.
|
||||
8. THE Migration SHALL translate existing `ollama` and `vllm` agent settings into Endpoint_Profile and Model_Deployment records without breaking active agents.
|
||||
9. THE Application SHALL have one documented fallback configuration source; conflicting model defaults in code, migrations, and Helm SHALL be removed.
|
||||
|
||||
### Requirement 4: Document Segmentation and Source Preservation
|
||||
|
||||
**User Story:** As an analyst, I want long articles, filings, and transcripts processed without destructive truncation, so that material facts near the end of a document are not lost.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Pipeline SHALL preserve the full normalized source document and SHALL NOT truncate it to a fixed character prefix for extraction.
|
||||
2. THE Segmenter SHALL create sentence-aware chunks with stable chunk IDs, source character offsets, and configurable overlap.
|
||||
3. THE Segmenter SHALL use document-type-specific chunk sizes for articles, filings, transcripts, and press releases.
|
||||
4. THE Segmenter SHALL preserve headings, speaker labels, table-derived text markers, and section boundaries when present.
|
||||
5. WHEN duplicate or boilerplate sections are detected, THE Segmenter SHALL mark them without deleting the only occurrence of a fact.
|
||||
6. THE Pipeline SHALL retain a mapping from every downstream Evidence_Span to the original document offsets.
|
||||
7. IF a document cannot be decoded or segmented, THEN THE Pipeline SHALL mark the document as a typed preprocessing failure and SHALL NOT fabricate an empty extraction.
|
||||
|
||||
### Requirement 5: Deterministic Candidate Generation and Ticker Resolution
|
||||
|
||||
**User Story:** As a signal consumer, I want explicit companies and numeric facts resolved deterministically where possible, so that a language model is not asked to invent identifiers or parse trivial values.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Candidate_Generator SHALL detect explicit ticker symbols, company names, aliases, executives, products, currencies, percentages, dates, ranges, EPS values, revenue values, guidance values, and common financial ratios.
|
||||
2. THE Symbol_Resolver SHALL use the existing company and symbol registry as the source of truth for ticker identity.
|
||||
3. THE Pipeline SHALL distinguish explicit company mentions from inferred exposure relationships.
|
||||
4. THE Pipeline SHALL NOT pass the entire tracked-ticker universe to a generative prompt.
|
||||
5. WHEN multiple companies match an alias, THE Symbol_Resolver SHALL return ranked candidates and SHALL require contextual disambiguation or adjudication.
|
||||
6. WHEN a ticker is not present in the symbol registry, THE Pipeline SHALL preserve the literal mention as unresolved rather than inventing a registered ticker.
|
||||
7. THE Numeric_Normalizer SHALL retain both literal source text and normalized values, currencies, units, periods, and ranges.
|
||||
8. THE Pipeline SHALL reject normalized numeric facts whose value cannot be traced to an Evidence_Span.
|
||||
|
||||
### Requirement 6: Specialist Extraction Service
|
||||
|
||||
**User Story:** As an operator, I want routine entity, event, relation, and fact extraction to run on CPU-first specialist models, so that GPU capacity is reserved for difficult reasoning.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Specialist_Service SHALL expose batched APIs for entity extraction, schema extraction, relation extraction, and text classification.
|
||||
2. THE Initial specialist extractor SHALL support company, person, product, event, financial metric, date, percentage, currency, and relationship schemas.
|
||||
3. THE Specialist_Service SHALL return character spans and per-candidate scores for every extracted item.
|
||||
4. THE Specialist_Service SHALL run without requiring the RTX 4070 Ti SUPER.
|
||||
5. THE Initial deployment SHALL evaluate GLiNER2 Large as the primary unified extraction and classification model.
|
||||
6. THE Benchmark SHALL evaluate NuExtract 1.5 Smol as an optional long-form or hierarchical fact-extraction stage, but it SHALL NOT become an always-resident GPU model without passing incremental-value and resource gates.
|
||||
7. THE Specialist_Service SHALL support model version pinning, warm-up, health checks, bounded batching, and graceful degradation.
|
||||
8. WHEN specialist inference fails, THE Router SHALL either retry according to policy or route to adjudication; it SHALL record the failure and SHALL NOT silently substitute default facts.
|
||||
9. THE Specialist_Service SHALL expose model and schema versions in every response.
|
||||
|
||||
### Requirement 7: Company-Specific Financial Sentiment
|
||||
|
||||
**User Story:** As an analyst, I want sentiment tied to each company and supporting evidence, so that a positive statement about one firm is not applied to every company in the article.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Sentiment_Stage SHALL score evidence sentences or evidence groups associated with each resolved company.
|
||||
2. THE Initial sentiment classifier SHALL evaluate FinBERT as the baseline financial-domain model.
|
||||
3. THE Sentiment_Stage SHALL return positive, negative, and neutral probabilities rather than only a discrete label.
|
||||
4. THE Pipeline SHALL derive mixed sentiment from conflicting supported evidence, not from an unconstrained model label.
|
||||
5. WHEN an article mentions competitors with opposing effects, THE Pipeline SHALL produce separate company-specific sentiment records.
|
||||
6. THE Sentiment_Stage SHALL preserve the evidence IDs used for each probability distribution.
|
||||
7. THE Production model SHALL be calibrated on the Gold_Corpus before its probabilities are treated as confidence values.
|
||||
|
||||
### Requirement 8: Evidence Verification and Grounding
|
||||
|
||||
**User Story:** As an auditor, I want every material claim verified against source evidence, so that generated summaries and signals cannot rely on unsupported assertions.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. EVERY material company fact, event, amount, direction, and relationship SHALL reference one or more Evidence_Spans.
|
||||
2. THE Verifier SHALL check span validity, source offsets, entity association, and schema compatibility.
|
||||
3. THE Benchmark SHALL evaluate a compact entailment verifier for claims that require semantic validation beyond exact matching.
|
||||
4. IF a candidate conflicts with its evidence, THEN THE Pipeline SHALL reject it or route the conflict to adjudication.
|
||||
5. THE Pipeline SHALL calculate evidence coverage as the proportion of required fields supported by valid spans.
|
||||
6. THE Pipeline SHALL store rejected candidates and rejection reasons for audit and active learning.
|
||||
7. JSON repair SHALL NOT transform an unsupported or truncated generative answer into a valid production extraction without marking it as repaired and revalidating every material field.
|
||||
|
||||
### Requirement 9: Deterministic Novelty and Duplicate Detection
|
||||
|
||||
**User Story:** As a signal consumer, I want novelty based on comparison with recent information, so that a model's subjective novelty guess does not amplify repeated news.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Novelty_Stage SHALL compare each document and material event against a configurable recent-history window.
|
||||
2. THE Novelty_Stage SHALL combine exact/near-duplicate fingerprints with compact semantic embeddings.
|
||||
3. THE Pipeline SHALL calculate novelty separately for document-level content and company-event content.
|
||||
4. THE Novelty_Stage SHALL return nearest matching document or event IDs plus similarity scores.
|
||||
5. THE Pipeline SHALL derive `novelty_score` from the similarity distribution and duplicate count using a versioned deterministic formula or calibrated model.
|
||||
6. A generative model SHALL NOT provide the authoritative novelty value used by aggregation.
|
||||
7. WHEN novelty cannot be calculated because history is unavailable, THE Pipeline SHALL use a conservative versioned default and mark the reason.
|
||||
|
||||
### Requirement 10: Calibrated Extraction Confidence
|
||||
|
||||
**User Story:** As a downstream scorer, I want confidence to reflect observed correctness, so that the system does not trust a model merely because it reports confidence in itself.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Pipeline SHALL calculate field-level and record-level confidence from specialist scores, symbol resolution, evidence validation, schema completeness, model agreement, and historical calibration.
|
||||
2. A generative model's self-reported confidence SHALL NOT be used as authoritative extraction confidence.
|
||||
3. THE Calibration_Process SHALL evaluate isotonic, Platt, or equivalent calibration methods on held-out Gold_Corpus data.
|
||||
4. THE Pipeline SHALL report Expected Calibration Error and Brier score for probability-bearing stages.
|
||||
5. THE Router SHALL use calibrated uncertainty and explicit conflict rules to choose Fast_Path or Adjudication_Path.
|
||||
6. THE Pipeline SHALL retain stage-level confidence components for explainability.
|
||||
7. WHEN calibration data is insufficient for a class, THE Pipeline SHALL use conservative thresholds and mark the class as under-calibrated.
|
||||
|
||||
### Requirement 11: 9B Generative Adjudicator
|
||||
|
||||
**User Story:** As an analyst, I want the current reasoning capability retained for hard documents, so that specialization does not reduce intelligence on nuanced cases.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Adjudicator SHALL initially use the existing 9B-class model served by vLLM on the RTX 4070 Ti SUPER.
|
||||
2. THE Adjudicator SHALL receive selected source chunks, Evidence_Spans, candidate facts, candidate probabilities, conflicts, and a precise adjudication question rather than the entire tracked ticker list.
|
||||
3. THE Adjudicator SHALL use strict JSON Schema constrained output when supported by the endpoint.
|
||||
4. THE Adjudicator SHALL use deterministic generation settings appropriate for extraction, including a production default temperature of zero unless a benchmark proves a different value superior.
|
||||
5. THE Adjudicator SHALL NOT be asked to provide authoritative novelty, confidence, or impact values.
|
||||
6. THE Router SHALL invoke adjudication for unresolved entity aliases, contradictory evidence, multi-company causal relationships, implied consequences, complex guidance, materially incomplete fast-path results, or low calibrated confidence.
|
||||
7. THE Adjudicator SHALL return field-level decisions, evidence references, and decision reasons.
|
||||
8. IF adjudication output references evidence not supplied to it, THEN THE Verifier SHALL reject the unsupported field.
|
||||
9. THE Adjudicator SHALL remain optional for thesis prose; deterministic signal records SHALL not depend on prose generation succeeding.
|
||||
10. THE Peak GPU memory budget SHALL not exceed the measured current 9B deployment baseline by more than 5 percent unless explicitly approved.
|
||||
|
||||
### Requirement 12: Stock-Specific Impact and Horizon Model
|
||||
|
||||
**User Story:** As a trader, I want impact and horizon estimated from historical market behavior rather than language-model intuition, so that signals are tied to observed outcomes.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Pipeline SHALL separate textual sentiment from expected market impact.
|
||||
2. THE Impact_Model SHALL consume versioned features including event type probabilities, sentiment probabilities, magnitude, surprise where available, source history, novelty, company attributes, market regime, pre-event volatility, and evidence quality.
|
||||
3. THE Training_Pipeline SHALL create leakage-safe labels from abnormal returns and volume responses over configured horizons.
|
||||
4. THE Initial model family SHALL be a CPU-efficient calibrated tabular model and SHALL include a transparent deterministic baseline.
|
||||
5. THE Impact_Model SHALL output signed direction probabilities, expected magnitude, horizon probabilities, and model uncertainty.
|
||||
6. THE Production model SHALL be evaluated out-of-time and by event type, sector, market-cap bucket, and source.
|
||||
7. THE Pipeline SHALL preserve existing downstream fields through a Compatibility_Adapter while storing richer probability distributions in v3 tables.
|
||||
8. IF no trained Impact_Model is approved, THEN THE Pipeline SHALL use the deterministic baseline and SHALL NOT fall back to a generative model's impact score.
|
||||
9. THE Outcome_Evaluator SHALL feed realized outcomes back into model monitoring and retraining datasets without mutating historical predictions.
|
||||
10. THE Pipeline SHALL version feature definitions, training data ranges, model artifacts, thresholds, and calibration artifacts.
|
||||
|
||||
### Requirement 13: Versioned Intelligence Schema and Provenance
|
||||
|
||||
**User Story:** As a developer, I want a richer schema with field-level provenance, so that downstream consumers can distinguish facts, probabilities, decisions, and generated prose.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Database SHALL store v3 entities, facts, evidence spans, company signal candidates, stage runs, adjudication decisions, and model lineage in normalized or well-defined JSONB-backed tables.
|
||||
2. EVERY v3 field SHALL identify whether it is deterministic, specialist-derived, adjudicated, calibrated, or compatibility-derived.
|
||||
3. EVERY stage run SHALL record input references, output references, model versions, endpoint identity, duration, error state, and trace ID.
|
||||
4. THE Compatibility_Adapter SHALL map approved v3 outputs to existing v2 persistence records during migration.
|
||||
5. THE Compatibility_Adapter SHALL identify its own version and SHALL not overwrite original v3 probabilities.
|
||||
6. THE persisted `model_provider` and model lineage SHALL reflect the actual route used and SHALL not be hardcoded to Ollama.
|
||||
7. THE Pipeline SHALL retain raw model output only in approved object storage with configured retention and access controls.
|
||||
|
||||
### Requirement 14: Parallelism, Queues, and Resource Isolation
|
||||
|
||||
**User Story:** As an operator, I want parallel throughput without saturating the GPU or blocking unrelated stages, so that the cluster remains responsive.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Extractor SHALL support multiple in-flight documents using bounded asynchronous workers rather than a single unbounded sequential loop.
|
||||
2. THE Fast_Path and Adjudication_Path SHALL have separate queue or concurrency controls.
|
||||
3. THE Specialist_Service SHALL support dynamic batching within configured latency limits.
|
||||
4. THE Adjudicator SHALL enforce a GPU-safe concurrency semaphore coordinated with vLLM limits.
|
||||
5. THE Router SHALL apply backpressure when either path exceeds its queue-depth or latency thresholds.
|
||||
6. THE Deployment SHALL assign specialist workloads to CPU nodes and the 9B vLLM workload to the RTX 4070 Ti SUPER node by default.
|
||||
7. THE System SHALL expose queue depth, service time, batch size, GPU memory, GPU utilization, fast-path rate, and adjudication rate.
|
||||
8. WHEN the adjudicator is unavailable, THE Pipeline SHALL continue only for documents meeting a conservative fast-path acceptance threshold; all others SHALL remain queued or fail safely.
|
||||
|
||||
### Requirement 15: Observability, Audit, and Explainability
|
||||
|
||||
**User Story:** As an operator and analyst, I want to understand why a document produced a signal and which component made each decision.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Pipeline SHALL emit one distributed trace covering preprocessing, specialist stages, routing, adjudication, impact prediction, and persistence.
|
||||
2. THE Metrics SHALL include field validity, evidence coverage, entity resolution rate, sentiment agreement, calibration metrics, fast-path coverage, adjudication causes, schema failure rate, latency percentiles, token usage, and GPU-seconds per document.
|
||||
3. THE Audit API SHALL return model lineage and evidence for a document, company, and generated signal.
|
||||
4. THE UI SHALL distinguish observed facts, inferred exposure, sentiment, predicted impact, and generated narrative.
|
||||
5. THE Pipeline SHALL store routing reasons as structured codes rather than log-only text.
|
||||
6. THE Pipeline SHALL allow a reviewer to mark a field correct, incorrect, unsupported, or ambiguous and add a corrected value.
|
||||
7. Reviewer corrections SHALL be immutable audit events and SHALL feed the active-learning dataset only through an approved export process.
|
||||
|
||||
### Requirement 16: Benchmark, Shadow Mode, and Promotion Gates
|
||||
|
||||
**User Story:** As an owner, I want the new architecture proven against the current system before it affects trades, so that complexity is justified by measured improvement.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Team SHALL create a versioned Gold_Corpus covering articles, filings, press releases, transcripts, macro news, multi-company stories, contradictory reports, and long documents.
|
||||
2. THE Evaluation_Harness SHALL run the current pipeline and every proposed v3 configuration on identical inputs.
|
||||
3. THE Evaluation SHALL report field precision, recall, F1, exact-match accuracy, evidence support rate, ticker-resolution accuracy, event macro-F1, sentiment macro-F1, calibration, latency, throughput, CPU use, GPU use, and cost.
|
||||
4. THE Evaluation SHALL report results by document type, event class, source, sector, and difficulty bucket.
|
||||
5. THE Initial promotion gate SHALL require no statistically meaningful regression in any safety-critical field and measurable improvement in at least one of evidence support, calibration, schema validity, or resource efficiency.
|
||||
6. THE Initial production target SHALL achieve at least 60 percent Fast_Path coverage on the representative corpus while meeting accuracy gates.
|
||||
7. THE GPU-seconds per accepted document SHALL improve by at least 2x relative to the current 9B-every-document baseline before full promotion.
|
||||
8. THE v3 pipeline SHALL run in Shadow_Mode for a configurable period and minimum document count before it may influence aggregation.
|
||||
9. THE Promotion process SHALL support canary percentages, automatic rollback thresholds, and one-click reversion to the current pipeline.
|
||||
10. Backtest or paper-trading performance SHALL be reported separately from extraction correctness and SHALL not override failed correctness gates.
|
||||
|
||||
### Requirement 17: Active Learning and Specialist Fine-Tuning
|
||||
|
||||
**User Story:** As a model owner, I want difficult and corrected examples to improve the specialist path over time, so that fewer documents require the 9B adjudicator.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Active_Learning_Exporter SHALL select low-confidence, conflicting, adjudicated, and reviewer-corrected examples without exporting secrets or unauthorized content.
|
||||
2. THE Export format SHALL retain source text, spans, schema labels, relations, adjudicator decisions, reviewer corrections, and provenance.
|
||||
3. THE Training_Pipeline SHALL support fine-tuning the selected specialist extractor on the Stonks Oracle schema.
|
||||
4. EACH trained artifact SHALL be evaluated against a frozen holdout and the current production artifact.
|
||||
5. A specialist model SHALL not be promoted solely because it reduces adjudication rate; it SHALL also pass field-level correctness and calibration gates.
|
||||
6. THE Registry SHALL retain model cards containing training range, dataset version, intended use, limitations, and evaluation results.
|
||||
|
||||
### Requirement 18: Backward-Compatible Rollout
|
||||
|
||||
**User Story:** As a maintainer, I want to ship the new pipeline incrementally, so that existing APIs and downstream services continue operating during migration.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Current v2 extractor SHALL remain available behind a feature flag until v3 completes shadow and canary promotion.
|
||||
2. THE Compatibility_Adapter SHALL produce the fields required by aggregation, recommendation, validation, reporting, and API consumers.
|
||||
3. THE Database migration SHALL be additive before any destructive column or table change.
|
||||
4. THE Deployment SHALL permit per-agent, per-document-type, and percentage-based routing between v2 and v3.
|
||||
5. WHEN rollback is triggered, THE System SHALL route new work to v2 without deleting v3 audit data.
|
||||
6. THE Team SHALL remove deprecated provider branches, v2 prompt logic, and compatibility mappings only in a separately approved cleanup milestone.
|
||||
@@ -0,0 +1,427 @@
|
||||
# Implementation Plan: Intelligence Pipeline v3
|
||||
|
||||
## Overview
|
||||
|
||||
Replace the monolithic 9B model extraction pipeline with a staged, evidence-grounded multi-component architecture. CPU-first specialist services handle routine extraction, sentiment, novelty, and calibration while the existing 9B vLLM model is preserved for semantic adjudication of ambiguous cases. A capability-aware inference gateway replaces duplicated provider branches, a stock-specific impact model replaces generative self-scores, and a full shadow/canary promotion process ensures measured improvement before production influence.
|
||||
|
||||
## Tasks
|
||||
|
||||
- [x] 1. Rotate exposed credentials
|
||||
- [x] 1.1 Identify every live or reusable credential in `infra/helm/stonks-oracle/values-live-math.yaml` and any other tracked files
|
||||
- [x] 1.2 Rotate database, MinIO/object-store, Redis, broker, and market-data credentials
|
||||
- [x] 1.3 Disable the replaced keys and review relevant access logs
|
||||
- [x] 1.4 Remove plaintext values from the working tree without copying them into issues, PRs, logs, or spec comments
|
||||
- [x] 1.5 Purge the values from Git history using an approved coordinated history rewrite
|
||||
- [x] 1.6 Verify that old credentials no longer authenticate
|
||||
- _Requirements: 1.1, 1.2, 1.6_
|
||||
|
||||
- [x] 2. Add managed secret delivery
|
||||
- [x] 2.1 Select External Secrets, SOPS, Sealed Secrets, or the cluster-standard mechanism
|
||||
- [x] 2.2 Replace Helm secret values with secret references
|
||||
- [x] 2.3 Document bootstrap and rotation procedures
|
||||
- [x] 2.4 Add a deployment test proving pods receive required keys without values appearing in rendered manifests
|
||||
- _Requirements: 1.3, 1.6_
|
||||
|
||||
- [x] 3. Add repository secret scanning
|
||||
- [x] 3.1 Add a secret scanner to pre-commit or Kiro hooks
|
||||
- [x] 3.2 Add the scanner to pull-request and protected-branch CI
|
||||
- [x] 3.3 Add tests/fixtures that prove real-looking secrets fail and explicit safe fixtures pass
|
||||
- _Requirements: 1.4, 1.5_
|
||||
|
||||
- [x] 4. Establish the current runtime source of truth
|
||||
- [x] 4.1 Inventory active cluster deployments, agent database records, Helm releases, and environment variables
|
||||
- [x] 4.2 Record the actual model, quantization, vLLM version, max model length, max sequences, GPU utilization limit, and current provider for every agent
|
||||
- [x] 4.3 Resolve the conflicting Qwen/NuExtract defaults in code and infrastructure for the baseline branch
|
||||
- [x] 4.4 Produce `docs/intelligence-pipeline-v3/current-runtime-baseline.md` without credentials
|
||||
- _Requirements: 3.9_
|
||||
|
||||
- [x] 5. Build a baseline replay command
|
||||
- [x] 5.1 Add a CLI that replays a fixed document set through the current pipeline without writing trading outputs
|
||||
- [x] 5.2 Capture structured output, schema validity, retries, duration, token usage, GPU metrics, provider/model lineage, and current downstream mappings
|
||||
- [x] 5.3 Pin all baseline configuration and random seeds that the provider supports
|
||||
- [x] 5.4 Store baseline reports under a versioned artifact path
|
||||
- _Requirements: 16.2, 16.3_
|
||||
|
||||
- [x] 6. Define the v3 annotation schema
|
||||
- [x] 6.1 Define labels for entities, canonical companies, events, relations, numeric facts, periods, sentiment, evidence spans, direct effects, inferred exposure, and ambiguity
|
||||
- [x] 6.2 Define evidence and adjudication guidelines with positive and negative examples
|
||||
- [x] 6.3 Define which fields are safety-critical for promotion gates
|
||||
- [x] 6.4 Add schema validators and sample annotations
|
||||
- _Requirements: 16.1, 16.4_
|
||||
|
||||
- [x] 7. Create the first Gold Corpus
|
||||
- [x] 7.1 Sample at least 1,000 documents stratified by type, event class, length, source, company count, and difficulty
|
||||
- [x] 7.2 Include duplicate stories, long filings, transcripts, contradictory reports, macro events, and opposing multi-company effects
|
||||
- [x] 7.3 Double-review a hard-case subset and calculate inter-annotator agreement
|
||||
- [x] 7.4 Freeze a holdout split that cannot be used for prompt or model tuning
|
||||
- _Requirements: 16.1_
|
||||
|
||||
- [x] 8. Implement evaluation metrics
|
||||
- [x] 8.1 Implement entity/ticker precision, recall, F1, and ambiguity accuracy
|
||||
- [x] 8.2 Implement event and relation macro/micro F1
|
||||
- [x] 8.3 Implement numeric exact/tolerance-aware matching
|
||||
- [x] 8.4 Implement evidence offset validity and support rate
|
||||
- [x] 8.5 Implement sentiment macro-F1 and probability calibration metrics
|
||||
- [x] 8.6 Implement latency, throughput, token, CPU, GPU, and memory metrics
|
||||
- [x] 8.7 Generate per-document-type and per-difficulty reports
|
||||
- _Requirements: 16.3, 16.4_
|
||||
|
||||
- [x] 9. Benchmark corrected current-model extraction
|
||||
- [x] 9.1 Run the current request unchanged
|
||||
- [x] 9.2 Run the same 9B model with temperature zero
|
||||
- [x] 9.3 Run the same 9B model with strict JSON Schema output and temperature zero
|
||||
- [x] 9.4 Quantify how much of the apparent architecture gain comes from fixing the current request alone
|
||||
- _Requirements: 16.2, 16.3, 16.5_
|
||||
|
||||
- [x] 10. Add shared inference domain models
|
||||
- [x] 10.1 Create `services/shared/inference/models.py` with capabilities, target, request, result, usage, and lineage types
|
||||
- [x] 10.2 Create normalized error categories for timeout, authentication, rate limit, server, invalid response, schema, capability, and policy failures
|
||||
- [x] 10.3 Add serialization tests proving credentials and sensitive headers are excluded
|
||||
- _Requirements: 2.1, 2.8, 2.9_
|
||||
|
||||
- [x] 11. Implement `OpenAICompatibleClient`
|
||||
- [x] 11.1 Implement `/v1/chat/completions` using `httpx.AsyncClient`
|
||||
- [x] 11.2 Implement Bearer and configurable authentication headers via runtime secret resolution
|
||||
- [x] 11.3 Implement standard `response_format.json_schema` payloads
|
||||
- [x] 11.4 Implement configurable vLLM `structured_outputs` extra-body payloads
|
||||
- [x] 11.5 Implement explicit JSON-object and prompt-only fallback policies
|
||||
- [x] 11.6 Capture request ID, usage, finish reason, structured mode, retries, and provider error category
|
||||
- [x] 11.7 Revalidate parsed JSON locally against the supplied schema
|
||||
- [x] 11.8 Add contract tests against a mocked compatible server and the cluster vLLM deployment
|
||||
- _Requirements: 2.1, 2.2, 2.3, 2.4, 2.5, 2.7, 2.9_
|
||||
|
||||
- [x] 12. Refactor Ollama support into `OllamaNativeClient`
|
||||
- [x] 12.1 Move current Ollama request logic behind the shared request/result types
|
||||
- [x] 12.2 Honor configured max output tokens and context settings consistently
|
||||
- [x] 12.3 Preserve native schema formatting when supported and explicitly report prompt-only mode otherwise
|
||||
- [x] 12.4 Retain stall/loop detection as Ollama-specific policy without leaking it into the generic protocol
|
||||
- _Requirements: 2.1, 2.6_
|
||||
|
||||
- [x] 13. Implement capability probing
|
||||
- [x] 13.1 Probe health and model listing
|
||||
- [x] 13.2 Probe strict JSON Schema with a minimal schema
|
||||
- [x] 13.3 Probe usage metadata, seed behavior, and output-token field compatibility
|
||||
- [x] 13.4 Store probe results and software/version metadata with TTL
|
||||
- [x] 13.5 Refuse activation when declared required capabilities fail
|
||||
- _Requirements: 2.10, 2.11_
|
||||
|
||||
- [x] 14. Replace provider fallback behavior
|
||||
- [x] 14.1 Replace `VLLMClient` with an alias/profile using `OpenAICompatibleClient`
|
||||
- [x] 14.2 Make unknown providers a typed configuration error
|
||||
- [x] 14.3 Add migration warnings for `vllm` provider records
|
||||
- [x] 14.4 Add property tests proving unknown providers never invoke Ollama
|
||||
- _Requirements: 2.2, 2.6_
|
||||
|
||||
- [x] 15. Migrate all LLM consumers
|
||||
- [x] 15.1 Migrate document extraction
|
||||
- [x] 15.2 Migrate global event classification
|
||||
- [x] 15.3 Migrate thesis rewriting and remove duplicate provider branching
|
||||
- [x] 15.4 Replace direct/private `_config` mutation with an explicit target refresh or client-pool lifecycle
|
||||
- [x] 15.5 Fix persistence so actual endpoint, model, and route lineage are recorded
|
||||
- _Requirements: 2.12, 13.6_
|
||||
|
||||
- [x] 16. Add registry migrations
|
||||
- [x] 16.1 Create `inference_endpoints`
|
||||
- [x] 16.2 Create `model_deployments`
|
||||
- [x] 16.3 Create `agent_stage_bindings`
|
||||
- [x] 16.4 Add revision, audit, uniqueness, and enabled-state constraints
|
||||
- [x] 16.5 Add additive lineage columns/tables for existing performance logs
|
||||
- _Requirements: 3.1, 3.2, 3.3, 3.4_
|
||||
|
||||
- [x] 17. Implement registry resolver
|
||||
- [x] 17.1 Resolve active stage bindings with TTL caching
|
||||
- [x] 17.2 Invalidate cache on revisions and failed probes
|
||||
- [x] 17.3 Resolve authentication only at invocation time
|
||||
- [x] 17.4 Add deterministic resolution and fail-closed tests
|
||||
- _Requirements: 3.5, 3.9_
|
||||
|
||||
- [x] 18. Migrate existing provider records
|
||||
- [x] 18.1 Create the current Ollama endpoint profile if in use
|
||||
- [x] 18.2 Create the current vLLM OpenAI-compatible endpoint profile
|
||||
- [x] 18.3 Create model deployments matching actual runtime state
|
||||
- [x] 18.4 Convert agent and variant provider/model fields to stage bindings while retaining compatibility reads
|
||||
- [x] 18.5 Remove conflicting runtime model defaults after migration verification
|
||||
- _Requirements: 3.8, 3.9_
|
||||
|
||||
- [x] 19. Add endpoint API and UI
|
||||
- [x] 19.1 Add CRUD endpoints that never return secret values
|
||||
- [x] 19.2 Add probe, enable, disable, and test-structured-output actions
|
||||
- [x] 19.3 Replace free-text provider inputs with protocol, endpoint, and model-deployment selectors
|
||||
- [x] 19.4 Display last probe, capabilities, model limits, and active stage bindings
|
||||
- [x] 19.5 Require confirmation for external endpoint egress enablement
|
||||
- _Requirements: 3.6, 3.7_
|
||||
|
||||
- [x] 20. Add v3 persistence tables
|
||||
- [x] 20.1 Add pipeline runs and stage runs
|
||||
- [x] 20.2 Add document chunks and evidence spans
|
||||
- [x] 20.3 Add extracted entities, facts, relations, and rejected candidates
|
||||
- [x] 20.4 Add company signal candidates and probability distributions
|
||||
- [x] 20.5 Add adjudication decisions, routing reasons, calibration references, and model lineage
|
||||
- [x] 20.6 Add idempotency and immutable-revision constraints
|
||||
- _Requirements: 13.1, 13.2, 13.3_
|
||||
|
||||
- [x] 21. Implement sentence-aware segmenter
|
||||
- [x] 21.1 Preserve source offsets and checksums
|
||||
- [x] 21.2 Add document-type-specific chunk strategies
|
||||
- [x] 21.3 Preserve filing sections and transcript speakers
|
||||
- [x] 21.4 Mark boilerplate and duplicate chunks
|
||||
- [x] 21.5 Remove the 8,000-character truncation from v3
|
||||
- [x] 21.6 Add property tests proving every chunk/evidence span maps exactly to source text
|
||||
- _Requirements: 4.1, 4.2, 4.3, 4.4, 4.5, 4.6_
|
||||
|
||||
- [x] 22. Implement compatibility adapter skeleton
|
||||
- [x] 22.1 Map approved v3 records to current intelligence and impact data classes
|
||||
- [x] 22.2 Persist `hybrid` lineage plus stage details
|
||||
- [x] 22.3 Add golden mapping tests for every legacy enum and field range
|
||||
- [x] 22.4 Keep adapter output disabled outside replay/shadow mode
|
||||
- _Requirements: 13.4, 13.5, 18.2_
|
||||
|
||||
- [x] 23. Implement deterministic financial parsing
|
||||
- [x] 23.1 Parse tickers, currencies, money, percentages, basis points, ranges, EPS, revenue, dates, and fiscal periods
|
||||
- [x] 23.2 Store literal and normalized representations
|
||||
- [x] 23.3 Link each candidate to exact offsets
|
||||
- [x] 23.4 Add broad property tests for numeric formatting and unit conversions
|
||||
- _Requirements: 5.1, 5.7, 5.8_
|
||||
|
||||
- [x] 24. Integrate symbol registry resolution
|
||||
- [x] 24.1 Build canonical alias indexes from existing companies and symbol registry data
|
||||
- [x] 24.2 Return ranked candidates and ambiguity margins
|
||||
- [x] 24.3 Separate explicit mentions from inferred exposures
|
||||
- [x] 24.4 Preserve unresolved literal entities without invented tickers
|
||||
- [x] 24.5 Add tests for aliases shared by multiple companies
|
||||
- _Requirements: 5.2, 5.3, 5.4, 5.5, 5.6_
|
||||
|
||||
- [x] 25. Create specialist inference service
|
||||
- [x] 25.1 Add typed batch endpoints for entities, classification, relations, and structured extraction
|
||||
- [x] 25.2 Integrate pinned GLiNER2 Large as the initial candidate
|
||||
- [x] 25.3 Return spans, scores, model version, and schema version
|
||||
- [x] 25.4 Add bounded dynamic batching and warm-up
|
||||
- [x] 25.5 Add Kubernetes CPU deployment, health probes, and metrics
|
||||
- [x] 25.6 Add contract and load tests
|
||||
- _Requirements: 6.1, 6.2, 6.3, 6.4, 6.5, 6.7, 6.9_
|
||||
|
||||
- [x] 26. Integrate company-specific sentiment
|
||||
- [x] 26.1 Build company-linked evidence groups
|
||||
- [x] 26.2 Integrate pinned FinBERT baseline
|
||||
- [x] 26.3 Store full probability distributions and evidence IDs
|
||||
- [x] 26.4 Implement mixed sentiment from evidence-group disagreement
|
||||
- [x] 26.5 Benchmark and calibrate on the Gold_Corpus
|
||||
- _Requirements: 7.1, 7.2, 7.3, 7.4, 7.5, 7.6, 7.7_
|
||||
|
||||
- [x] 27. Benchmark NuExtract 1.5 Smol
|
||||
- [x] 27.1 Add an isolated adapter and CPU/on-demand deployment
|
||||
- [x] 27.2 Test hierarchical extraction on long filings and transcripts
|
||||
- [x] 27.3 Measure incremental correctness over GLiNER2 plus deterministic parsing
|
||||
- [x] 27.4 Measure CPU latency and memory
|
||||
- [x] 27.5 Promote it only for document classes where incremental value passes a predefined gate
|
||||
- _Requirements: 6.6_
|
||||
|
||||
- [x] 28. Add evidence verification
|
||||
- [x] 28.1 Validate offsets, source text, entity association, and numeric consistency
|
||||
- [x] 28.2 Add rejected-candidate storage and reason codes
|
||||
- [x] 28.3 Benchmark a compact entailment verifier on claims exact matching cannot validate
|
||||
- [x] 28.4 Add unsupported-claim and evidence-coverage metrics
|
||||
- _Requirements: 8.1, 8.2, 8.3, 8.4, 8.5, 8.6, 8.7_
|
||||
|
||||
- [x] 29. Implement retrieval-based novelty
|
||||
- [x] 29.1 Add exact and near-duplicate fingerprints
|
||||
- [x] 29.2 Add replaceable compact embedding backend
|
||||
- [x] 29.3 Index document and canonical company-event embeddings
|
||||
- [x] 29.4 Return nearest matches and similarity scores
|
||||
- [x] 29.5 Implement and version the novelty formula
|
||||
- [x] 29.6 Compare novelty values against human duplicate/novelty labels
|
||||
- _Requirements: 9.1, 9.2, 9.3, 9.4, 9.5, 9.6, 9.7_
|
||||
|
||||
- [x] 30. Build confidence feature pipeline
|
||||
- [x] 30.1 Compute field-level features from extraction, resolution, evidence, sentiment, and agreement
|
||||
- [x] 30.2 Train and compare calibration methods on training folds
|
||||
- [x] 30.3 Evaluate ECE and Brier score on held-out data
|
||||
- [x] 30.4 Version and load calibration artifacts
|
||||
- [x] 30.5 Define conservative defaults for underrepresented classes
|
||||
- _Requirements: 10.1, 10.2, 10.3, 10.4, 10.6, 10.7_
|
||||
|
||||
- [x] 31. Implement deterministic routing engine
|
||||
- [x] 31.1 Define routing reason enums
|
||||
- [x] 31.2 Implement hard ambiguity/conflict rules
|
||||
- [x] 31.3 Implement calibrated fast-path thresholds by document and event type
|
||||
- [x] 31.4 Store every route decision and feature snapshot
|
||||
- [x] 31.5 Add property tests for determinism and threshold boundaries
|
||||
- _Requirements: 10.5, 11.6_
|
||||
|
||||
- [x] 32. Define adjudication schemas
|
||||
- [x] 32.1 Define candidate, conflict, question, evidence, and decision models
|
||||
- [x] 32.2 Exclude authoritative confidence, novelty, impact, and horizon from the model output
|
||||
- [x] 32.3 Require evidence IDs for every material decision
|
||||
- _Requirements: 11.2, 11.5, 11.7_
|
||||
|
||||
- [x] 33. Build focused adjudication prompts
|
||||
- [x] 33.1 Build packets from only relevant chunks and candidates
|
||||
- [x] 33.2 Use strict JSON Schema and temperature zero
|
||||
- [x] 33.3 Set a bounded output budget appropriate to decisions rather than long summaries
|
||||
- [x] 33.4 Add prompt/version metadata and exact provider lineage
|
||||
- _Requirements: 11.2, 11.3, 11.4_
|
||||
|
||||
- [x] 34. Deploy and validate the 9B adjudicator
|
||||
- [x] 34.1 Pin the approved 9B model and vLLM version
|
||||
- [x] 34.2 Verify strict structured output with the deployment's actual vLLM version
|
||||
- [x] 34.3 Measure peak VRAM against the current baseline and enforce the +5 percent gate
|
||||
- [x] 34.4 Load-test concurrency and select a safe application semaphore
|
||||
- [x] 34.5 Add availability and queue-depth alerts
|
||||
- _Requirements: 11.1, 11.10_
|
||||
|
||||
- [x] 35. Add post-adjudication verification
|
||||
- [x] 35.1 Verify every referenced evidence ID was included in the packet
|
||||
- [x] 35.2 Reject unsupported or schema-incompatible decisions
|
||||
- [x] 35.3 Preserve both pre-adjudication candidates and final decisions
|
||||
- [x] 35.4 Route repeated failures to review rather than accepting repaired defaults
|
||||
- _Requirements: 11.8, 8.7_
|
||||
|
||||
- [x] 36. Define event-time feature snapshots
|
||||
- [x] 36.1 Define feature names, types, timing rules, and missing-value policy
|
||||
- [x] 36.2 Include event, sentiment, magnitude, surprise, source, novelty, evidence, company, volatility, volume, and regime features
|
||||
- [x] 36.3 Persist immutable feature snapshots at prediction time
|
||||
- [x] 36.4 Add leakage tests preventing post-event data from entering features
|
||||
- _Requirements: 12.2, 12.10_
|
||||
|
||||
- [x] 37. Build outcome labels
|
||||
- [x] 37.1 Define approved market benchmarks and abnormal-return calculations
|
||||
- [x] 37.2 Generate signed and absolute response labels for intraday, 1d, 7d, 30d, and 90d horizons
|
||||
- [x] 37.3 Generate abnormal-volume and time-to-peak labels where data quality permits
|
||||
- [x] 37.4 Version label-generation code and market-data snapshots
|
||||
- _Requirements: 12.3_
|
||||
|
||||
- [x] 38. Implement deterministic impact baseline
|
||||
- [x] 38.1 Map event classes, sentiment, magnitude, evidence, novelty, and source credibility to conservative outputs
|
||||
- [x] 38.2 Unit-test every event type and boundary
|
||||
- [x] 38.3 Use this baseline whenever no approved trained model exists
|
||||
- _Requirements: 12.4, 12.8_
|
||||
|
||||
- [x] 39. Train calibrated tabular impact models
|
||||
- [x] 39.1 Train CPU-efficient gradient-boosted candidates for direction, magnitude, and horizon
|
||||
- [x] 39.2 Use walk-forward/out-of-time splits
|
||||
- [x] 39.3 Calibrate probabilities on a separate calibration fold
|
||||
- [x] 39.4 Report metrics by event, sector, market cap, source, and regime
|
||||
- [x] 39.5 Register artifacts, feature versions, training ranges, and model cards
|
||||
- _Requirements: 12.4, 12.5, 12.6, 12.10_
|
||||
|
||||
- [x] 40. Integrate impact outputs
|
||||
- [x] 40.1 Store full direction, magnitude, horizon, and uncertainty outputs
|
||||
- [x] 40.2 Map approved outputs to legacy `impact_score` and `impact_horizon` through the compatibility adapter
|
||||
- [x] 40.3 Remove generative impact/novelty/confidence from aggregation inputs in v3 mode
|
||||
- [x] 40.4 Add comparison dashboards against realized outcomes
|
||||
- _Requirements: 12.1, 12.7, 12.9_
|
||||
|
||||
- [x] 41. Implement v3 pipeline orchestrator
|
||||
- [x] 41.1 Create explicit stage state transitions and idempotency keys
|
||||
- [x] 41.2 Add fast-path, adjudication, persistence, and review queues
|
||||
- [x] 41.3 Implement leases, retry policies, dead-letter handling, and resumable stages
|
||||
- [x] 41.4 Keep v2 and v3 routing behind independent feature flags
|
||||
- _Requirements: 14.1, 14.2, 14.5, 14.8_
|
||||
|
||||
- [x] 42. Add bounded application parallelism
|
||||
- [x] 42.1 Replace the single sequential extraction loop for v3 with configurable async workers
|
||||
- [x] 42.2 Add specialist micro-batching
|
||||
- [x] 42.3 Add adjudicator semaphore and queue backpressure
|
||||
- [x] 42.4 Add load shedding rules that never drop safety-critical documents silently
|
||||
- _Requirements: 14.1, 14.3, 14.4, 14.5_
|
||||
|
||||
- [x] 43. Add traces and metrics
|
||||
- [x] 43.1 Trace every stage under one document trace ID
|
||||
- [x] 43.2 Add stage latency, errors, batch size, queue depth, and route metrics
|
||||
- [x] 43.3 Add field accuracy, evidence coverage, calibration, fast-path rate, and adjudication reason dashboards
|
||||
- [x] 43.4 Add GPU memory, utilization, and GPU-seconds per document
|
||||
- [x] 43.5 Add alerts for schema failures, unsupported claims, calibration drift, queue saturation, and provider probe failures
|
||||
- _Requirements: 14.7, 15.1, 15.2, 15.5_
|
||||
|
||||
- [x] 44. Add audit/review API and UI
|
||||
- [x] 44.1 Display source evidence and offsets for each fact
|
||||
- [x] 44.2 Display specialist probabilities, routing reasons, adjudicator decisions, and impact-model outputs separately
|
||||
- [x] 44.3 Allow immutable reviewer correction events
|
||||
- [x] 44.4 Add filters for low confidence, unsupported claims, and adjudicated documents
|
||||
- _Requirements: 15.3, 15.4, 15.6, 15.7_
|
||||
|
||||
- [x] 45. Run offline replay
|
||||
- [x] 45.1 Compare every required system configuration on the Gold_Corpus
|
||||
- [x] 45.2 Publish field-level, calibration, resource, and difficulty-bucket reports
|
||||
- [x] 45.3 Confirm corrected current-9B baseline versus full v3 incremental gain
|
||||
- [x] 45.4 Reject or retune any stage failing safety-critical gates
|
||||
- _Requirements: 16.2, 16.3, 16.4, 16.5_
|
||||
|
||||
- [x] 46. Enable production shadow mode
|
||||
- [x] 46.1 Run v3 for live documents without affecting aggregation or trading
|
||||
- [x] 46.2 Compare v2/v3 disagreements and sample reviews by risk
|
||||
- [x] 46.3 Measure fast-path coverage, GPU reduction, and operational stability
|
||||
- [x] 46.4 Require the configured minimum shadow duration and document count
|
||||
- _Requirements: 16.6, 16.7, 16.8_
|
||||
|
||||
- [x] 47. Canary compatibility outputs
|
||||
- [x] 47.1 Enable v3 adapter outputs for non-trading consumers first
|
||||
- [x] 47.2 Add percentage- and document-type-based routing
|
||||
- [x] 47.3 Configure automatic rollback on correctness, latency, queue, or availability thresholds
|
||||
- [x] 47.4 Verify rollback leaves v3 audit records intact
|
||||
- _Requirements: 16.9, 18.4, 18.5_
|
||||
|
||||
- [x] 48. Canary signal influence
|
||||
- [x] 48.1 Enable v3 signals in paper trading at a small percentage
|
||||
- [x] 48.2 Report extraction correctness separately from trading outcomes
|
||||
- [x] 48.3 Review material recommendation divergences
|
||||
- [x] 48.4 Promote only after explicit owner approval and all gates pass
|
||||
- _Requirements: 16.9, 16.10_
|
||||
|
||||
- [x] 49. Build active-learning export
|
||||
- [x] 49.1 Select low-confidence, conflicting, adjudicated, and corrected cases
|
||||
- [x] 49.2 Remove or policy-filter sensitive content
|
||||
- [x] 49.3 Export source spans, labels, relations, decisions, and provenance in a versioned format
|
||||
- _Requirements: 17.1, 17.2_
|
||||
|
||||
- [x] 50. Fine-tune specialist extractor
|
||||
- [x] 50.1 Train GLiNER2 candidate artifacts on the Stonks Oracle schema
|
||||
- [x] 50.2 Evaluate against frozen holdout and production artifact
|
||||
- [x] 50.3 Calibrate new scores and update routing thresholds
|
||||
- [x] 50.4 Promote only when correctness gates pass, not merely when adjudication rate falls
|
||||
- _Requirements: 17.3, 17.4, 17.5, 17.6_
|
||||
|
||||
- [x] 51. Deprecate legacy paths
|
||||
- [x] 51.1 Remove duplicated `VLLMClient`/provider branching after all consumers use the gateway
|
||||
- [x] 51.2 Remove v2 8,000-character truncation and monolithic extraction prompt after v2 retirement
|
||||
- [x] 51.3 Remove obsolete environment/model defaults and provider free-text fields
|
||||
- [x] 51.4 Remove compatibility adapter only after every downstream consumer reads v3 natively
|
||||
- [x] 51.5 Archive final migration and benchmark reports
|
||||
- _Requirements: 18.6_
|
||||
|
||||
- [x] 52. Checkpoint — Ensure all tests pass
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
## Task Dependency Graph
|
||||
|
||||
```json
|
||||
{
|
||||
"waves": [
|
||||
{ "id": 0, "tasks": ["1.1", "1.2", "1.3", "1.4", "1.5", "1.6", "2.1", "2.2", "2.3", "2.4", "3.1", "3.2", "3.3", "4.1", "4.2", "4.3", "4.4", "5.1", "5.2", "5.3", "5.4"] },
|
||||
{ "id": 1, "tasks": ["6.1", "6.2", "6.3", "6.4", "7.1", "7.2", "7.3", "7.4", "8.1", "8.2", "8.3", "8.4", "8.5", "8.6", "8.7", "9.1", "9.2", "9.3", "9.4"] },
|
||||
{ "id": 2, "tasks": ["10.1", "10.2", "10.3", "11.1", "11.2", "11.3", "11.4", "11.5", "11.6", "11.7", "11.8", "12.1", "12.2", "12.3", "12.4"] },
|
||||
{ "id": 3, "tasks": ["13.1", "13.2", "13.3", "13.4", "13.5", "14.1", "14.2", "14.3", "14.4", "15.1", "15.2", "15.3", "15.4", "15.5", "16.1", "16.2", "16.3", "16.4", "16.5"] },
|
||||
{ "id": 4, "tasks": ["17.1", "17.2", "17.3", "17.4", "18.1", "18.2", "18.3", "18.4", "18.5", "19.1", "19.2", "19.3", "19.4", "19.5", "20.1", "20.2", "20.3", "20.4", "20.5", "20.6", "21.1", "21.2", "21.3", "21.4", "21.5", "21.6", "22.1", "22.2", "22.3", "22.4"] },
|
||||
{ "id": 5, "tasks": ["23.1", "23.2", "23.3", "23.4", "24.1", "24.2", "24.3", "24.4", "24.5", "25.1", "25.2", "25.3", "25.4", "25.5", "25.6", "26.1", "26.2", "26.3", "26.4", "26.5", "27.1", "27.2", "27.3", "27.4", "27.5", "28.1", "28.2", "28.3", "28.4"] },
|
||||
{ "id": 6, "tasks": ["29.1", "29.2", "29.3", "29.4", "29.5", "29.6", "30.1", "30.2", "30.3", "30.4", "30.5", "31.1", "31.2", "31.3", "31.4", "31.5"] },
|
||||
{ "id": 7, "tasks": ["32.1", "32.2", "32.3", "33.1", "33.2", "33.3", "33.4", "34.1", "34.2", "34.3", "34.4", "34.5", "35.1", "35.2", "35.3", "35.4"] },
|
||||
{ "id": 8, "tasks": ["36.1", "36.2", "36.3", "36.4", "37.1", "37.2", "37.3", "37.4", "38.1", "38.2", "38.3", "39.1", "39.2", "39.3", "39.4", "39.5", "40.1", "40.2", "40.3", "40.4"] },
|
||||
{ "id": 9, "tasks": ["41.1", "41.2", "41.3", "41.4", "42.1", "42.2", "42.3", "42.4", "43.1", "43.2", "43.3", "43.4", "43.5", "44.1", "44.2", "44.3", "44.4"] },
|
||||
{ "id": 10, "tasks": ["45.1", "45.2", "45.3", "45.4", "46.1", "46.2", "46.3", "46.4", "47.1", "47.2", "47.3", "47.4", "48.1", "48.2", "48.3", "48.4"] },
|
||||
{ "id": 11, "tasks": ["49.1", "49.2", "49.3", "50.1", "50.2", "50.3", "50.4", "51.1", "51.2", "51.3", "51.4", "51.5"] }
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
## Notes
|
||||
|
||||
- This plan is intentionally staged so the current system remains available until the replacement is measured and promoted
|
||||
- Task 1 (credential rotation) is a **BLOCKER** — must complete before any feature deployment
|
||||
- Tasks within the same wave may run in parallel; tasks in later waves depend on earlier waves completing
|
||||
- Each phase ends with an explicit evidence artifact: test output, benchmark report, migration result, or deployment probe
|
||||
- The current v2 extractor remains available behind a feature flag until v3 completes shadow and canary promotion
|
||||
- Peak GPU memory must not exceed the measured current 9B deployment baseline by more than 5 percent
|
||||
- The definition of done requires: credentials rotated, all consumers on shared gateway, evidence-linked outputs, calibrated scores replacing generative self-scores, 9B preserved for adjudication, and shadow/canary gates passed
|
||||
- Rollback to the current production path must be exercised successfully before full promotion
|
||||
- Property tests validate deterministic behaviors (unknown providers fail closed, chunk offsets map to source, routing thresholds are deterministic)
|
||||
- Deprecated legacy paths (task 51) require separate approval and must not proceed until all downstream consumers read v3 natively
|
||||
@@ -0,0 +1 @@
|
||||
{"specId": "ff0d03d7-3469-4551-bf05-15295b735c83", "workflowType": "requirements-first", "specType": "feature"}
|
||||
@@ -0,0 +1,851 @@
|
||||
# Design Document: Math Core v3 Engine
|
||||
|
||||
## Overview
|
||||
|
||||
The v3 Calibrated Evidence Engine replaces the current dual-mode pipeline (heuristic + probabilistic) with a principled Bayesian evidence accumulation system. The upgrade transforms the signal processing core from weighted-sentiment averaging to:
|
||||
|
||||
```
|
||||
EvidenceUnit → calibrated reliability (q_i) → log-likelihood ratio (LLR_i)
|
||||
→ correlation-adjusted cluster LLR → posterior P_up → return distribution → EV/risk decision
|
||||
```
|
||||
|
||||
**Key design goals:**
|
||||
- Replace arbitrary weight products with calibrated probabilistic evidence
|
||||
- Prevent correlated articles from inflating evidence counts
|
||||
- Make confidence multiplicative so one bad dimension suppresses the trade
|
||||
- Size positions using fractional Kelly criterion under hard risk caps
|
||||
- Preserve the existing three-layer architecture and service boundaries
|
||||
- Gate behind `v3_engine_enabled` feature flag with heuristic fallback
|
||||
|
||||
**What changes vs. what stays:**
|
||||
- The `WeightedSignal` abstraction remains as an intermediate before LLR conversion
|
||||
- All output goes into existing JSONB metadata columns (no new migrations)
|
||||
- Service file boundaries are preserved; internals are upgraded
|
||||
- The heuristic pipeline stays as fallback, controlled by a single flag read per aggregation cycle
|
||||
|
||||
## Architecture
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
subgraph Input Layer
|
||||
CS[Company Signals]
|
||||
MS[Macro Signals]
|
||||
XS[Competitive Signals]
|
||||
end
|
||||
|
||||
subgraph Normalization
|
||||
EU[EvidenceUnit Normalization]
|
||||
end
|
||||
|
||||
subgraph Scoring Pipeline
|
||||
QI[Calibrated Reliability q_i]
|
||||
LLR[LLR Conversion]
|
||||
end
|
||||
|
||||
subgraph Clustering
|
||||
CL[Correlation-Aware Clustering]
|
||||
NEFF[n_eff Computation]
|
||||
CLLR[Cluster LLR]
|
||||
end
|
||||
|
||||
subgraph Posterior Assembly
|
||||
REG[Regime Detection v3]
|
||||
POST[Log-Odds Posterior P_up]
|
||||
CON[LLR Entropy Contradiction]
|
||||
CONF[Multiplicative Confidence]
|
||||
end
|
||||
|
||||
subgraph Decision Layer
|
||||
PROJ[Posterior State Projection]
|
||||
RET[Return Distribution / EV Gate]
|
||||
ELIG[Regime-Aware Eligibility]
|
||||
KELLY[Fractional Kelly Sizing]
|
||||
STOP[Regime-Aware Stops]
|
||||
HEAT[Stop-Defined Portfolio Heat]
|
||||
end
|
||||
|
||||
subgraph Quality & Control
|
||||
DQ[Data Quality v3]
|
||||
FF[Feature Flag Router]
|
||||
TIER[Risk Tier Auto-Adjustment]
|
||||
end
|
||||
|
||||
CS --> EU
|
||||
MS --> EU
|
||||
XS --> EU
|
||||
EU --> QI
|
||||
QI --> LLR
|
||||
LLR --> CL
|
||||
CL --> NEFF
|
||||
NEFF --> CLLR
|
||||
REG --> POST
|
||||
CLLR --> POST
|
||||
POST --> CON
|
||||
POST --> CONF
|
||||
CONF --> PROJ
|
||||
PROJ --> RET
|
||||
RET --> ELIG
|
||||
ELIG --> KELLY
|
||||
KELLY --> STOP
|
||||
KELLY --> HEAT
|
||||
DQ --> CONF
|
||||
FF --> EU
|
||||
TIER --> KELLY
|
||||
```
|
||||
|
||||
### Service File Mapping
|
||||
|
||||
| Service File | v3 Responsibility |
|
||||
|---|---|
|
||||
| `services/aggregation/scoring.py` | EvidenceUnit normalization, q_i pipeline, LLR conversion |
|
||||
| `services/aggregation/bayesian.py` | Posterior assembly via log-odds, P_up, strength |
|
||||
| `services/aggregation/contradiction.py` | LLR entropy contradiction score |
|
||||
| `services/aggregation/regime.py` | Regime detection v3 (ATR-normalized trend_z) |
|
||||
| `services/aggregation/interpolation.py` | Noisy-OR macro exposure, LLR emission |
|
||||
| `services/aggregation/signal_propagation.py` | Correlation-shrunk competitive propagation |
|
||||
| `services/aggregation/projection.py` | Posterior state A_t, regime-aware decay |
|
||||
| `services/aggregation/worker.py` | Orchestration, clustering, n_eff, confidence assembly |
|
||||
| `services/recommendation/eligibility.py` | EV gate, regime-aware eligibility, mode escalation |
|
||||
| `services/trading/position_sizer.py` | Fractional Kelly sizing under caps |
|
||||
| `services/trading/stop_loss_manager.py` | Regime-aware stop/TP, trailing activation |
|
||||
| `services/risk/engine.py` | Stop-defined heat, tier auto-adjustment |
|
||||
|
||||
### Feature Flag Flow
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant Worker as aggregation/worker.py
|
||||
participant DB as risk_configs table
|
||||
participant V3 as v3 Pipeline
|
||||
participant Heuristic as Heuristic Pipeline
|
||||
|
||||
Worker->>DB: SELECT v3_engine_enabled
|
||||
alt v3_engine_enabled = True
|
||||
Worker->>V3: Run v3 pipeline
|
||||
V3-->>Worker: Posterior + confidence + EV
|
||||
alt Unhandled error
|
||||
V3-->>Worker: Exception
|
||||
Worker->>Heuristic: Fallback to heuristic
|
||||
Worker->>Worker: Log error, record fallback in metadata
|
||||
end
|
||||
else v3_engine_enabled = False or DB error
|
||||
Worker->>Heuristic: Run heuristic pipeline
|
||||
end
|
||||
```
|
||||
|
||||
## Components and Interfaces
|
||||
|
||||
### 1. EvidenceUnit Dataclass (`scoring.py`)
|
||||
|
||||
```python
|
||||
@dataclass(frozen=True)
|
||||
class EvidenceUnit:
|
||||
"""Canonical normalized signal representation for v3 pipeline."""
|
||||
symbol: str
|
||||
layer: str # "company" | "macro" | "competitive"
|
||||
event_type: str
|
||||
source_id: str
|
||||
source_group: str
|
||||
timestamp: datetime
|
||||
horizon: str # "intraday" | "1d" | "7d" | "30d" | "90d"
|
||||
direction: int # -1, 0, +1
|
||||
sentiment_strength: float # [0, 1]
|
||||
impact: float # [0, 1]
|
||||
extraction_conf: float # [0, 1]
|
||||
source_cred: float # [0, 1]
|
||||
novelty: float # [0, 1]
|
||||
event_base_rate: float # (0, 1]
|
||||
cluster_id: str
|
||||
```
|
||||
|
||||
**Normalization functions** — one per layer:
|
||||
- `normalize_company_signal(impact_row, ...) -> EvidenceUnit`
|
||||
- `normalize_macro_signal(macro_impact_record, global_event, ...) -> EvidenceUnit`
|
||||
- `normalize_competitive_signal(competitive_signal_record, ...) -> EvidenceUnit`
|
||||
|
||||
Each validates required fields (symbol, timestamp, source_id), substitutes 0.5 for missing optional numeric fields, and assigns direction from sentiment/impact_direction strings.
|
||||
|
||||
### 2. Calibrated Reliability Pipeline (`scoring.py`)
|
||||
|
||||
```python
|
||||
@dataclass(frozen=True)
|
||||
class ReliabilityComponents:
|
||||
q_ext: float
|
||||
q_source: float
|
||||
q_recency: float
|
||||
q_uniqueness: float
|
||||
q_i: float # final combined reliability
|
||||
|
||||
def compute_v3_reliability(
|
||||
unit: EvidenceUnit,
|
||||
source_stats: SourceStats,
|
||||
cluster_position: int, # duplicate_count_before
|
||||
reference_time: datetime,
|
||||
) -> ReliabilityComponents: ...
|
||||
```
|
||||
|
||||
Sub-computations:
|
||||
- `q_ext = sigmoid(8.0 * (extraction_conf - 0.55))`
|
||||
- `q_source = clamp((E[theta_s] - 0.50) / 0.35, 0, 1)` with Beta(alpha_0+hits, beta_0+misses)
|
||||
- `q_recency = 2^(-age_hours / tau_adaptive)` with adaptive half-life
|
||||
- `q_uniqueness = clamp(0.5 + 0.5 * novelty, 0.5, 1.0) * (1 / sqrt(1 + dup_count))`
|
||||
- `q_i = clamp(q_ext * q_source * source_cred * q_recency * q_uniqueness, 0, 1)`
|
||||
|
||||
### 3. LLR Conversion (`scoring.py`)
|
||||
|
||||
```python
|
||||
def compute_llr(unit: EvidenceUnit, q_i: float) -> float:
|
||||
"""Convert calibrated reliability to log-likelihood ratio."""
|
||||
p_correct = clamp(0.50 + 0.35 * q_i * unit.impact * unit.sentiment_strength, 0.501, 0.85)
|
||||
if unit.direction == 0:
|
||||
return 0.0
|
||||
return unit.direction * math.log(p_correct / (1 - p_correct))
|
||||
```
|
||||
|
||||
### 4. Correlation-Aware Clustering (`worker.py`)
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class EvidenceCluster:
|
||||
cluster_id: str
|
||||
units: list[EvidenceUnit]
|
||||
llrs: list[float]
|
||||
n_eff: float
|
||||
cluster_llr: float
|
||||
|
||||
def cluster_evidence(units: list[EvidenceUnit], llrs: list[float]) -> list[EvidenceCluster]:
|
||||
"""Group by (symbol, horizon, event_type, source_group, time_bucket)."""
|
||||
...
|
||||
|
||||
def compute_n_eff(llrs: list[float], correlations: list[list[float]]) -> float:
|
||||
"""n_eff = (sum w_i)^2 / (sum w_i^2 + 2*sum_{i<j} rho_ij*w_i*w_j)"""
|
||||
...
|
||||
|
||||
def compute_cluster_llr(llrs: list[float], n_eff: float) -> float:
|
||||
"""LLR_c = clamp(weighted_mean(LLR_i, |LLR_i|) * sqrt(n_eff), -2.5, 2.5)"""
|
||||
...
|
||||
```
|
||||
|
||||
Default pairwise correlations:
|
||||
| Relationship | rho |
|
||||
|---|---:|
|
||||
| Same wire/story/source group | 0.80 |
|
||||
| Same event, different publisher | 0.50 |
|
||||
| Same theme, different event | 0.25 |
|
||||
| Independent events | 0.00 |
|
||||
|
||||
### 5. Posterior Assembly (`bayesian.py`)
|
||||
|
||||
```python
|
||||
@dataclass(frozen=True)
|
||||
class V3Posterior:
|
||||
p_up: float # sigmoid(log_odds)
|
||||
p_down: float # 1 - p_up
|
||||
log_odds: float # logit(P_prior) + sum(gamma * LLR_c)
|
||||
strength: float # abs(2 * p_up - 1)
|
||||
direction: str # bullish | bearish | neutral
|
||||
n_eff_total: float
|
||||
regime: str
|
||||
|
||||
def compute_v3_posterior(
|
||||
clusters: list[EvidenceCluster],
|
||||
regime: RegimeClassification,
|
||||
p_prior: float = 0.50,
|
||||
) -> V3Posterior: ...
|
||||
```
|
||||
|
||||
Direction thresholds by regime:
|
||||
| Regime | Bullish if P_up >= | Bearish if P_up <= |
|
||||
|---|---:|---:|
|
||||
| panic | 0.68 | 0.32 |
|
||||
| trend_following | 0.60 | 0.40 |
|
||||
| mean_reversion | 0.63 | 0.37 |
|
||||
| uncertainty | 0.65 | 0.35 |
|
||||
|
||||
### 6. Regime Detection v3 (`regime.py`)
|
||||
|
||||
```python
|
||||
@dataclass(frozen=True)
|
||||
class V3RegimeClassification:
|
||||
regime: MarketRegime
|
||||
trend_z: float # (EMA_20 - EMA_100) / ATR_20
|
||||
vol_ratio: float # sigma_20 / sigma_100
|
||||
evidence_multiplier: float # gamma_regime
|
||||
confidence_multiplier: float
|
||||
phi_decay: float # for projection
|
||||
atr_multiplier: float # for stops
|
||||
|
||||
def classify_regime_v3(
|
||||
closing_prices: list[float],
|
||||
daily_returns: list[float],
|
||||
atr_20: float,
|
||||
) -> V3RegimeClassification: ...
|
||||
```
|
||||
|
||||
Classification rules (in priority order):
|
||||
1. **Panic**: vol_ratio > 1.5 OR abs(trend_z) > 2.5
|
||||
2. **Trend following**: abs(trend_z) >= 0.75 AND vol_ratio < 1.3
|
||||
3. **Mean reversion**: abs(trend_z) < 0.50 AND vol_ratio < 1.0
|
||||
4. **Uncertainty**: all other cases
|
||||
|
||||
### 7. LLR Entropy Contradiction (`contradiction.py`)
|
||||
|
||||
```python
|
||||
def compute_v3_contradiction(clusters: list[EvidenceCluster]) -> float:
|
||||
"""
|
||||
E_pos = sum(max(LLR_c, 0))
|
||||
E_neg = sum(max(-LLR_c, 0))
|
||||
H = -f_pos*log2(f_pos) - f_neg*log2(f_neg)
|
||||
volume_factor = 1 - exp(-E_total / 3.0)
|
||||
return H * volume_factor
|
||||
"""
|
||||
...
|
||||
```
|
||||
|
||||
### 8. Multiplicative Confidence (`worker.py`)
|
||||
|
||||
```python
|
||||
def compute_v3_confidence(
|
||||
n_eff_total: float,
|
||||
q_values: list[float],
|
||||
llrs: list[float],
|
||||
strength: float,
|
||||
regime_confidence_mult: float,
|
||||
contradiction: float,
|
||||
data_quality: float,
|
||||
) -> float:
|
||||
"""
|
||||
C_evidence = 1 - exp(-n_eff_total / 5.0)
|
||||
C_quality = weighted_mean(q_i, |LLR_i|)
|
||||
confidence = clamp(
|
||||
C_evidence * sqrt(C_quality) * sqrt(max(strength, 0.05))
|
||||
* regime_confidence_mult * (1 - contradiction) * data_quality,
|
||||
0, 1
|
||||
)
|
||||
"""
|
||||
...
|
||||
```
|
||||
|
||||
### 9. Noisy-OR Macro Exposure (`interpolation.py`)
|
||||
|
||||
```python
|
||||
def compute_normalized_macro_exposure(overlaps: dict[str, float]) -> float:
|
||||
"""
|
||||
E_raw = 1 - prod(1 - w_k * O_k)
|
||||
E_max = 1 - prod(1 - w_k)
|
||||
return E_raw / E_max
|
||||
"""
|
||||
...
|
||||
|
||||
def compute_macro_llr(
|
||||
macro_impact: float,
|
||||
event_confidence: float,
|
||||
q_recency: float,
|
||||
macro_direction: int,
|
||||
) -> float:
|
||||
"""
|
||||
p_macro = clamp(0.50 + 0.30 * macro_impact * event_confidence * q_recency, 0.501, 0.80)
|
||||
return macro_direction * log(p_macro / (1 - p_macro))
|
||||
"""
|
||||
...
|
||||
```
|
||||
|
||||
### 10. Correlation-Shrunk Competitive Propagation (`signal_propagation.py`)
|
||||
|
||||
```python
|
||||
def compute_shrunk_correlation(
|
||||
rho_rolling: float,
|
||||
n_observations: int,
|
||||
same_sector: bool,
|
||||
) -> float:
|
||||
"""
|
||||
rho_prior = 0.30 if same_sector else 0.10
|
||||
rho_shrunk = (n/(n+30)) * rho_rolling + (30/(n+30)) * rho_prior
|
||||
return max(rho_shrunk, 0)
|
||||
"""
|
||||
...
|
||||
|
||||
def compute_competitive_llr(
|
||||
llr_source: float,
|
||||
rho_effective: float,
|
||||
d_network: int,
|
||||
pattern_confidence: float,
|
||||
) -> float:
|
||||
"""
|
||||
attenuation = rho_effective * exp(-0.85 * d_network)
|
||||
return clamp(llr_source * attenuation * pattern_confidence, -1.25, 1.25)
|
||||
"""
|
||||
...
|
||||
```
|
||||
|
||||
### 11. Posterior State Projection (`projection.py`)
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class V3ProjectionState:
|
||||
a_t: float # accumulated evidence state
|
||||
p_up_projected: float # sigmoid(logit(P_prior) + phi^h * A_t)
|
||||
projected_strength: float
|
||||
diverges: bool
|
||||
phi_regime: float
|
||||
|
||||
def compute_v3_projection(
|
||||
a_prev: float,
|
||||
cluster_llrs: list[float],
|
||||
regime: V3RegimeClassification,
|
||||
p_prior: float,
|
||||
projection_horizon: int,
|
||||
known_catalyst_llr: float = 0.0,
|
||||
) -> V3ProjectionState: ...
|
||||
```
|
||||
|
||||
Regime decay factors (phi):
|
||||
| Regime | phi |
|
||||
|---|---:|
|
||||
| panic | 0.35 |
|
||||
| trend_following | 0.80 |
|
||||
| mean_reversion | 0.55 |
|
||||
| uncertainty | 0.50 |
|
||||
|
||||
### 12. Return Distribution and EV Gate (`eligibility.py`)
|
||||
|
||||
```python
|
||||
@dataclass(frozen=True)
|
||||
class ReturnDistribution:
|
||||
sigma_h: float # realized_vol_20d * sqrt(horizon_days / 252)
|
||||
mu_h: float # tanh(A_projected / 3.0) * confidence * sigma_h
|
||||
ev_long: float # mu_h - costs - 0.10 * CVaR_5
|
||||
min_edge: float # regime-specific minimum edge
|
||||
eligible: bool
|
||||
|
||||
def compute_return_distribution(
|
||||
a_projected: float,
|
||||
confidence: float,
|
||||
realized_vol_20d: float,
|
||||
horizon_days: int,
|
||||
costs: float,
|
||||
regime: str,
|
||||
) -> ReturnDistribution: ...
|
||||
```
|
||||
|
||||
### 13. Fractional Kelly Position Sizing (`position_sizer.py`)
|
||||
|
||||
```python
|
||||
def compute_kelly_sizing(
|
||||
p_win: float, # P_up from posterior
|
||||
b: float, # reward ratio: clamp(1.2 + 2*conf + str - contra, 1.2, 3.0)
|
||||
confidence: float,
|
||||
data_quality: float,
|
||||
contradiction: float,
|
||||
max_position_pct: float,
|
||||
available_caps: dict[str, float], # sector, correlation, heat capacities
|
||||
) -> float:
|
||||
"""
|
||||
f_kelly = (p_win * b - (1 - p_win)) / b
|
||||
portfolio_pct = clamp(max(0, f_kelly) * 0.25 * confidence * data_quality * (1 - contradiction), 0, max_position_pct)
|
||||
Apply min of all capacity constraints.
|
||||
"""
|
||||
...
|
||||
```
|
||||
|
||||
### 14. Stop-Defined Portfolio Heat (`risk/engine.py`)
|
||||
|
||||
```python
|
||||
def compute_portfolio_heat(
|
||||
positions: list[OpenPosition],
|
||||
stop_distances: dict[str, float],
|
||||
) -> float:
|
||||
"""risk_dollars = position_value * stop_distance_pct; heat = sum(risk_dollars)"""
|
||||
...
|
||||
|
||||
def check_heat_capacity(
|
||||
current_heat: float,
|
||||
new_risk_dollars: float,
|
||||
max_heat_pct: float,
|
||||
portfolio_value: float,
|
||||
) -> bool: ...
|
||||
```
|
||||
|
||||
### 15. Data Quality v3 (`worker.py`)
|
||||
|
||||
```python
|
||||
def compute_v3_data_quality(
|
||||
units: list[EvidenceUnit],
|
||||
extraction_failure_rate: float,
|
||||
age_newest_hours: float,
|
||||
n_source_types: int,
|
||||
) -> float:
|
||||
"""
|
||||
Q_parse = 1 - extraction_failure_rate
|
||||
Q_conf = weighted_mean(extraction_conf, impact)
|
||||
Q_fresh = exp(-age_newest_hours / 168)
|
||||
Q_coverage = 1 - exp(-N_valid / 5)
|
||||
Q_diversity = min(1, log2(1 + N_source_types) / log2(4))
|
||||
return clamp(Q_parse * sqrt(Q_conf) * Q_fresh * Q_coverage * Q_diversity, 0, 1)
|
||||
"""
|
||||
...
|
||||
```
|
||||
|
||||
### 16. Risk Tier Auto-Adjustment (`risk/engine.py`)
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class TierMetrics:
|
||||
profit_factor_30d: float
|
||||
max_drawdown_30d: float
|
||||
calibration_error: float
|
||||
realized_sharpe_30d: float
|
||||
n_trades_30d: int
|
||||
reserve_pool_pct: float
|
||||
|
||||
def evaluate_tier_adjustment(metrics: TierMetrics) -> str:
|
||||
"""Returns 'upgrade' | 'downgrade' | 'hold'"""
|
||||
...
|
||||
```
|
||||
|
||||
## Data Models
|
||||
|
||||
### EvidenceUnit (frozen dataclass)
|
||||
|
||||
| Field | Type | Range | Source |
|
||||
|---|---|---|---|
|
||||
| symbol | str | — | Required from signal |
|
||||
| layer | str | company/macro/competitive | Set during normalization |
|
||||
| event_type | str | — | catalyst_type or impact_type |
|
||||
| source_id | str | — | document_id or event_id |
|
||||
| source_group | str | — | publisher / "macro" / "competitive" |
|
||||
| timestamp | datetime | — | published_at |
|
||||
| horizon | str | intraday/1d/7d/30d/90d | window or estimated_duration mapping |
|
||||
| direction | int | -1, 0, +1 | sentiment/direction mapping |
|
||||
| sentiment_strength | float | [0, 1] | impact_score or sentiment confidence |
|
||||
| impact | float | [0, 1] | impact_score |
|
||||
| extraction_conf | float | [0, 1] | confidence field |
|
||||
| source_cred | float | [0, 1] | source_credibility |
|
||||
| novelty | float | [0, 1] | novelty_score |
|
||||
| event_base_rate | float | (0, 1] | lookup by event_type |
|
||||
| cluster_id | str | — | computed hash of clustering key |
|
||||
|
||||
### V3 Posterior Output (stored in JSONB metadata)
|
||||
|
||||
```json
|
||||
{
|
||||
"v3_posterior": {
|
||||
"p_up": 0.64,
|
||||
"p_down": 0.36,
|
||||
"log_odds": 0.58,
|
||||
"strength": 0.28,
|
||||
"confidence": 0.61,
|
||||
"contradiction": 0.18,
|
||||
"n_eff": 5.7,
|
||||
"data_quality": 0.82,
|
||||
"regime": "trend_following"
|
||||
},
|
||||
"v3_return_model": {
|
||||
"mu_h": 0.012,
|
||||
"sigma_h": 0.041,
|
||||
"ev_long": 0.007,
|
||||
"min_edge": 0.0035
|
||||
},
|
||||
"v3_explainability": {
|
||||
"top_positive_clusters": [...],
|
||||
"top_negative_clusters": [...],
|
||||
"suppression_reasons": [],
|
||||
"risk_adjustments": []
|
||||
},
|
||||
"pipeline_mode": "v3"
|
||||
}
|
||||
```
|
||||
|
||||
### Source Statistics (for q_source computation)
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class SourceStats:
|
||||
source_id: str
|
||||
hits: int = 0 # correct directional predictions
|
||||
misses: int = 0 # incorrect directional predictions
|
||||
alpha_0: int = 3 # prior
|
||||
beta_0: int = 3 # prior
|
||||
```
|
||||
|
||||
### Regime Parameters Table
|
||||
|
||||
| Regime | gamma (evidence) | confidence_mult | phi (decay) | ATR_mult (stops) | min_edge |
|
||||
|---|---:|---:|---:|---:|---:|
|
||||
| panic | 0.70 | 0.70 | 0.35 | 2.5 | 0.0100 |
|
||||
| trend_following | 1.10 | 1.00 | 0.80 | 1.8 | 0.0035 |
|
||||
| mean_reversion | 0.90 | 0.95 | 0.55 | 1.4 | 0.0050 |
|
||||
| uncertainty | 0.80 | 0.85 | 0.50 | 2.0 | 0.0075 |
|
||||
|
||||
### Eligibility Thresholds (Regime-Specific)
|
||||
|
||||
| Regime | confidence_min | contradiction_max | strength_min |
|
||||
|---|---:|---:|---:|
|
||||
| panic | 0.70 | 0.25 | 0.36 |
|
||||
| trend_following | 0.55 | 0.40 | 0.20 |
|
||||
| mean_reversion | 0.60 | 0.35 | 0.26 |
|
||||
| uncertainty | 0.65 | 0.30 | 0.30 |
|
||||
|
||||
|
||||
## Correctness Properties
|
||||
|
||||
*A property is a characteristic or behavior that should hold true across all valid executions of a system — essentially, a formal statement about what the system should do. Properties serve as the bridge between human-readable specifications and machine-verifiable correctness guarantees.*
|
||||
|
||||
### Property 1: Reliability q_i is bounded in [0, 1]
|
||||
|
||||
*For any* valid EvidenceUnit with extraction_conf in [0,1], source_cred in [0,1], novelty in [0,1], any non-negative age_hours, and any non-negative duplicate_count_before, the computed q_i SHALL be in the range [0.0, 1.0].
|
||||
|
||||
**Validates: Requirements 2.8, 21.1**
|
||||
|
||||
### Property 2: p_correct is bounded in [0.501, 0.85]
|
||||
|
||||
*For any* valid q_i in [0, 1], impact in [0, 1], and sentiment_strength in [0, 1], the computed p_correct SHALL be in the range [0.501, 0.85].
|
||||
|
||||
**Validates: Requirements 3.1, 21.2**
|
||||
|
||||
### Property 3: LLR sign matches direction and magnitude is bounded
|
||||
|
||||
*For any* valid signal with direction in {-1, +1}, the computed LLR SHALL have the same sign as direction, with absolute magnitude in [ln(0.501/0.499), ln(0.85/0.15)] ≈ [0.004, 1.735].
|
||||
|
||||
**Validates: Requirements 3.2, 3.4, 3.5, 3.6, 21.3**
|
||||
|
||||
### Property 4: Neutral signals produce zero LLR
|
||||
|
||||
*For any* valid EvidenceUnit with direction = 0, regardless of all other field values, the computed LLR SHALL be exactly 0.0.
|
||||
|
||||
**Validates: Requirements 1.5, 3.3**
|
||||
|
||||
### Property 5: Effective evidence count n_eff is bounded by cluster size
|
||||
|
||||
*For any* cluster of N signals with non-negative pairwise correlations rho_ij in [0, 1], the computed n_eff SHALL satisfy 0 < n_eff <= N.
|
||||
|
||||
**Validates: Requirements 4.2, 21.4**
|
||||
|
||||
### Property 6: Cluster LLR is clamped to [-2.5, 2.5]
|
||||
|
||||
*For any* cluster configuration with any number of signals and any LLR values, the computed cluster LLR_c SHALL be in the range [-2.5, 2.5].
|
||||
|
||||
**Validates: Requirements 4.4, 4.5**
|
||||
|
||||
### Property 7: Posterior P_up is in open interval (0, 1)
|
||||
|
||||
*For any* set of cluster LLRs (each in [-2.5, 2.5]), any regime evidence multiplier gamma in {0.70, 0.80, 0.90, 1.10}, and any prior P_prior in [0.40, 0.60], the computed P_up SHALL be in (1e-10, 1 - 1e-10).
|
||||
|
||||
**Validates: Requirements 5.3, 21.5**
|
||||
|
||||
### Property 8: Contradiction is zero when evidence is unidirectional
|
||||
|
||||
*For any* set of cluster LLRs where all clusters have the same sign (all positive or all negative), the computed contradiction score SHALL be 0.0.
|
||||
|
||||
**Validates: Requirements 7.7, 21.7**
|
||||
|
||||
### Property 9: Contradiction score is bounded in [0, 1]
|
||||
|
||||
*For any* set of cluster LLRs (including mixed positive and negative), the computed contradiction score SHALL be in the range [0.0, 1.0].
|
||||
|
||||
**Validates: Requirements 7.6**
|
||||
|
||||
### Property 10: Multiplicative confidence is bounded in [0, 1] and suppressed by weak dimensions
|
||||
|
||||
*For any* valid inputs (n_eff_total >= 0, q_values in [0,1], strength in [0,1], regime_confidence_mult in (0,1], contradiction in [0,1], data_quality in [0,1]), the computed confidence SHALL be in [0, 1]. Furthermore, if any single dimension (data_quality, 1-contradiction, or C_quality) is below 0.01, the resulting confidence SHALL be below 0.10.
|
||||
|
||||
**Validates: Requirements 8.3, 8.5, 21.6**
|
||||
|
||||
### Property 11: Fractional Kelly sizing is bounded and respects negative edge
|
||||
|
||||
*For any* valid inputs (P_up in (0,1), b in [1.2, 3.0], confidence in [0,1], data_quality in [0,1], contradiction in [0,1], max_position_pct > 0), the computed portfolio_pct SHALL be in [0, max_position_pct]. When f_kelly = (P_up * b - (1 - P_up)) / b <= 0, portfolio_pct SHALL be exactly 0.
|
||||
|
||||
**Validates: Requirements 14.4, 14.7, 21.8, 21.9**
|
||||
|
||||
### Property 12: Posterior state JSON round-trip
|
||||
|
||||
*For any* valid V3Posterior state (p_up, log_odds, strength, confidence, contradiction, n_eff, data_quality, regime), serializing to JSON and deserializing SHALL produce an equivalent state within floating-point tolerance (1e-10).
|
||||
|
||||
**Validates: Requirements 20.1, 21.10**
|
||||
|
||||
### Property 13: Noisy-OR normalized exposure is bounded in [0, 1]
|
||||
|
||||
*For any* overlap values O_k in [0, 1] for each dimension (geo, supply, commodity, sector) with fixed positive weights, the normalized macro exposure E_macro SHALL be in [0.0, 1.0], reaching exactly 1.0 when all O_k = 1.0.
|
||||
|
||||
**Validates: Requirements 9.1, 9.2**
|
||||
|
||||
### Property 14: Competitive LLR is clamped to [-1.25, 1.25]
|
||||
|
||||
*For any* source LLR, shrunk correlation (non-negative), graph distance (1-3), and pattern confidence in [0,1], the computed competitive LLR SHALL be in [-1.25, 1.25].
|
||||
|
||||
**Validates: Requirements 10.4, 10.5**
|
||||
|
||||
### Property 15: Graph attenuation is zero beyond max distance
|
||||
|
||||
*For any* inputs where graph distance > 3, the computed attenuation SHALL be 0.0, producing zero competitive LLR regardless of other parameters.
|
||||
|
||||
**Validates: Requirements 10.3**
|
||||
|
||||
### Property 16: Projection evidence state decays toward zero
|
||||
|
||||
*For any* initial evidence state A_t and regime decay phi in (0, 1), the projected state A_projected_h = phi^h * A_t SHALL have |A_projected_h| < |A_t| for all h >= 1, converging toward 0 as h increases.
|
||||
|
||||
**Validates: Requirements 11.1, 11.3**
|
||||
|
||||
### Property 17: Data quality score is bounded in [0, 1]
|
||||
|
||||
*For any* valid inputs (extraction_failure_rate in [0,1], extraction_conf_i in [0,1], impact_i in [0,1], age_newest_hours >= 0, N_valid >= 0, N_source_types >= 0), the computed data_quality_score SHALL be in [0, 1].
|
||||
|
||||
**Validates: Requirements 17.6**
|
||||
|
||||
### Property 18: Stop loss is below entry price and take profit is above
|
||||
|
||||
*For any* entry_price > 0, stop_distance_pct in [0.005, 1.0), and reward ratio b >= 1.2, the computed stop_loss SHALL be less than entry_price and take_profit SHALL be greater than entry_price.
|
||||
|
||||
**Validates: Requirements 16.2, 16.3**
|
||||
|
||||
### Property 19: Trailing stop never decreases
|
||||
|
||||
*For any* sequence of current prices and trailing stop computations, each new trailing_stop value SHALL be >= the previous trailing_stop value (monotonically non-decreasing).
|
||||
|
||||
**Validates: Requirements 16.5**
|
||||
|
||||
### Property 20: Regime classification is exhaustive and deterministic
|
||||
|
||||
*For any* valid market data inputs (closing_prices of sufficient length, daily_returns, ATR_20 > 0), the regime classification SHALL produce exactly one of {panic, trend_following, mean_reversion, uncertainty} and the same inputs SHALL always produce the same classification.
|
||||
|
||||
**Validates: Requirements 6.2, 6.3, 6.4, 6.5**
|
||||
|
||||
### Property 21: EvidenceUnit normalization preserves field ranges
|
||||
|
||||
*For any* valid company, macro, or competitive signal input, the normalized EvidenceUnit SHALL have: direction in {-1, 0, +1}, sentiment_strength in [0, 1], impact in [0, 1], extraction_conf in [0, 1], source_cred in [0, 1], novelty in [0, 1], and event_base_rate in (0, 1].
|
||||
|
||||
**Validates: Requirements 1.1, 1.2, 1.3, 1.4, 1.7**
|
||||
|
||||
### Property 22: Portfolio heat rejection is correct
|
||||
|
||||
*For any* set of open positions with stop distances, if the sum of (position_value × stop_distance_pct) exceeds max_portfolio_heat × portfolio_value, then new position entry SHALL be rejected.
|
||||
|
||||
**Validates: Requirements 15.3, 15.5**
|
||||
|
||||
### Property 23: Tier auto-adjustment obeys downgrade-any, upgrade-all logic
|
||||
|
||||
*For any* TierMetrics, if ANY single downgrade condition is met (profit_factor < 1.0 OR drawdown > 0.12 OR calibration_error > 0.20 OR sharpe < 0), the result SHALL be "downgrade". An "upgrade" SHALL only occur when ALL upgrade conditions are simultaneously met.
|
||||
|
||||
**Validates: Requirements 18.3, 18.4**
|
||||
|
||||
## Error Handling
|
||||
|
||||
### Fail-Closed Philosophy
|
||||
|
||||
The v3 engine follows a fail-closed design: when in doubt, suppress the trade rather than emit a false signal.
|
||||
|
||||
| Error Scenario | Response | Fallback |
|
||||
|---|---|---|
|
||||
| `v3_engine_enabled` flag unreadable | Default to heuristic mode | Log warning |
|
||||
| Unhandled exception in v3 pipeline | Fall back to heuristic for that cycle | Log ERROR with traceback, record in metadata |
|
||||
| Missing market data for regime | Default to "uncertainty" regime | Most conservative multipliers |
|
||||
| Missing source statistics | q_source = 0.0 (neutral prior) | Source treated as untrusted |
|
||||
| Missing realized_vol_20d | Use default 0.25 annualized | Conservative volatility estimate |
|
||||
| Division by zero in n_eff | Return n_eff = 1.0 (single signal) | Denominator guard |
|
||||
| NaN/Inf in any computation | Clamp to boundary, log warning | Never propagate NaN to output |
|
||||
| data_quality < 0.50 | Force informational mode | Suppress trade recommendation |
|
||||
| No company evidence (only macro/competitive) | Force informational | Unless macro_only_enabled |
|
||||
|
||||
### Numerical Guards
|
||||
|
||||
All mathematical functions include:
|
||||
- **Sigmoid overflow**: Guard `exp(-x)` for x > 500 or x < -500
|
||||
- **Log domain**: Guard `log(x)` with x > 0 check; `log2(0)` treated as 0 in entropy
|
||||
- **Division by zero**: All denominators checked > 0 before division
|
||||
- **NaN propagation**: All outputs validated with `math.isnan()` check before storage
|
||||
- **Clamp boundaries**: Final values clamped to documented ranges
|
||||
|
||||
### Graceful Degradation Chain
|
||||
|
||||
```
|
||||
v3 pipeline error → heuristic fallback → informational mode → no recommendation
|
||||
```
|
||||
|
||||
Each level preserves audit trail via output metadata.
|
||||
|
||||
## Testing Strategy
|
||||
|
||||
### Dual Testing Approach
|
||||
|
||||
**Property-Based Tests (Hypothesis):**
|
||||
- Library: `hypothesis` (already in use in this project)
|
||||
- Configuration: `@settings(max_examples=100)` minimum per property
|
||||
- File naming: `tests/test_pbt_v3_*.py`
|
||||
- Each property test tagged with: `# Feature: math-core-v3-engine, Property N: <title>`
|
||||
- One property-based test per correctness property (23 properties → 23 PBT tests)
|
||||
|
||||
**Unit Tests (pytest):**
|
||||
- Specific examples with known inputs/outputs for each formula
|
||||
- Edge cases: zero inputs, boundary values, NaN handling
|
||||
- Integration between components (e.g., full pipeline from EvidenceUnit to recommendation)
|
||||
- Error handling paths (DB errors, missing data, feature flag states)
|
||||
|
||||
### Property Test Organization
|
||||
|
||||
| Test File | Properties Covered | Module Under Test |
|
||||
|---|---|---|
|
||||
| `tests/test_pbt_v3_reliability.py` | 1, 2, 3, 4, 21 | scoring.py (q_i, p_correct, LLR) |
|
||||
| `tests/test_pbt_v3_clustering.py` | 5, 6 | worker.py (n_eff, cluster LLR) |
|
||||
| `tests/test_pbt_v3_posterior.py` | 7, 8, 9, 10, 12, 20 | bayesian.py, contradiction.py, worker.py |
|
||||
| `tests/test_pbt_v3_layers.py` | 13, 14, 15 | interpolation.py, signal_propagation.py |
|
||||
| `tests/test_pbt_v3_projection.py` | 16 | projection.py |
|
||||
| `tests/test_pbt_v3_decision.py` | 11, 17, 18, 19, 22 | position_sizer.py, stop_loss_manager.py, eligibility.py |
|
||||
| `tests/test_pbt_v3_tier.py` | 23 | risk/engine.py |
|
||||
|
||||
### Hypothesis Strategy Design
|
||||
|
||||
Key custom strategies for generating valid inputs:
|
||||
|
||||
```python
|
||||
from hypothesis import strategies as st
|
||||
|
||||
# EvidenceUnit generator
|
||||
evidence_units = st.builds(
|
||||
EvidenceUnit,
|
||||
symbol=st.text(min_size=1, max_size=5),
|
||||
layer=st.sampled_from(["company", "macro", "competitive"]),
|
||||
direction=st.sampled_from([-1, 0, 1]),
|
||||
sentiment_strength=st.floats(min_value=0.0, max_value=1.0),
|
||||
impact=st.floats(min_value=0.0, max_value=1.0),
|
||||
extraction_conf=st.floats(min_value=0.0, max_value=1.0),
|
||||
source_cred=st.floats(min_value=0.0, max_value=1.0),
|
||||
novelty=st.floats(min_value=0.0, max_value=1.0),
|
||||
event_base_rate=st.floats(min_value=0.01, max_value=1.0),
|
||||
...
|
||||
)
|
||||
|
||||
# Cluster LLR list generator
|
||||
cluster_llrs = st.lists(
|
||||
st.floats(min_value=-2.5, max_value=2.5),
|
||||
min_size=1, max_size=20,
|
||||
)
|
||||
|
||||
# Regime generator
|
||||
regimes = st.sampled_from(["panic", "trend_following", "mean_reversion", "uncertainty"])
|
||||
```
|
||||
|
||||
### Unit Test Coverage
|
||||
|
||||
| Area | Key Example Tests |
|
||||
|---|---|
|
||||
| EvidenceUnit normalization | Company signal → correct fields; macro → correct horizon mapping |
|
||||
| q_i pipeline | Known inputs → known outputs for each sub-formula |
|
||||
| LLR conversion | p_correct=0.60 → LLR≈0.405; direction=-1 → negative LLR |
|
||||
| Clustering | 3 identical articles → n_eff < 3; independent → n_eff = N |
|
||||
| Posterior | Empty evidence → P_up=0.50; strong bullish → P_up > 0.60 |
|
||||
| Contradiction | All bullish → 0; equal split → high score |
|
||||
| Confidence | Zero data quality → near-zero confidence |
|
||||
| Macro LLR | Full exposure → max LLR ≈ 1.10; zero overlap → LLR ≈ 0 |
|
||||
| Competitive | Distance 4 → zero propagation; direct rival → attenuated signal |
|
||||
| Kelly sizing | p_win=0.3, b=2 → f_kelly < 0 → size = 0 |
|
||||
| Stops | Entry=100, stop_dist=0.02 → stop=98, TP > 100 |
|
||||
| Feature flag | Flag false → heuristic path; flag true → v3 path |
|
||||
| Error fallback | v3 raises → heuristic runs, error logged |
|
||||
|
||||
### Integration Tests
|
||||
|
||||
- Full pipeline: raw signals → EvidenceUnit → q_i → LLR → cluster → posterior → recommendation
|
||||
- Feature flag toggle: verify clean switch between pipelines mid-run
|
||||
- JSONB round-trip: store v3 output in PostgreSQL JSONB, retrieve and verify
|
||||
- Regime transitions: price series that crosses regime boundaries
|
||||
@@ -0,0 +1,314 @@
|
||||
# Requirements Document
|
||||
|
||||
## Introduction
|
||||
|
||||
This specification defines the requirements for upgrading the Stonks Oracle signal processing engine from the current dual-mode pipeline (heuristic + probabilistic) to the v3 Calibrated Evidence Engine. The v3 engine replaces arbitrary weighted-sentiment scoring with a principled probabilistic pipeline: calibrated reliability estimation, log-likelihood ratio (LLR) evidence accumulation, correlation-aware clustering, Bayesian posterior assembly, return distribution modeling, and Kelly-criterion position sizing. The upgrade preserves the existing three-layer architecture (company, macro, competitive), the WeightedSignal abstraction, database schema compatibility, and the service boundary structure. The entire v3 engine operates behind a `v3_engine_enabled` feature flag with the heuristic mode retained as a fallback.
|
||||
|
||||
## Glossary
|
||||
|
||||
- **V3_Engine**: The new calibrated evidence engine that replaces the current heuristic and probabilistic scoring modes
|
||||
- **EvidenceUnit**: The canonical normalized shape for all signals (company, macro, competitive) before aggregation
|
||||
- **LLR**: Log-Likelihood Ratio — the calibrated evidence contribution of a single signal or cluster, measured in log-odds units
|
||||
- **Cluster**: A group of correlated signals sharing (symbol, horizon, event_type, source_group, time_bucket)
|
||||
- **n_eff**: Effective evidence count within a cluster after de-correlation adjustment
|
||||
- **Posterior**: The Bayesian posterior probability P_up computed via log-odds accumulation
|
||||
- **Regime**: Market regime classification (panic, trend_following, mean_reversion, uncertainty) derived from EMA trend and volatility indicators
|
||||
- **Contradiction_Score**: A measure of opposing evidence based on LLR entropy and evidence volume
|
||||
- **Data_Quality_Score**: Multiplicative fail-closed quality metric combining parse rate, confidence, freshness, coverage, and diversity
|
||||
- **Fractional_Kelly**: Position sizing method using Kelly criterion scaled by a conservative fraction (0.25) and further modulated by confidence, data quality, and contradiction
|
||||
- **Portfolio_Heat**: Total stop-defined risk dollars across all open positions as a fraction of portfolio value
|
||||
- **Feature_Flag**: The `v3_engine_enabled` runtime toggle that activates the v3 pipeline without code deployment
|
||||
- **Scoring_Service**: The `services/aggregation/scoring.py` module responsible for signal weight computation
|
||||
- **Bayesian_Service**: The `services/aggregation/bayesian.py` module responsible for posterior computation
|
||||
- **Contradiction_Service**: The `services/aggregation/contradiction.py` module responsible for conflict detection
|
||||
- **Regime_Service**: The `services/aggregation/regime.py` module responsible for market regime classification
|
||||
- **Projection_Service**: The `services/aggregation/projection.py` module responsible for trend projection
|
||||
- **Eligibility_Service**: The `services/recommendation/eligibility.py` module responsible for recommendation gating
|
||||
- **Position_Sizer**: The `services/trading/position_sizer.py` module responsible for trade sizing
|
||||
- **Stop_Loss_Manager**: The `services/trading/stop_loss_manager.py` module responsible for stop/TP computation
|
||||
- **Risk_Engine**: The `services/risk/engine.py` module responsible for portfolio risk enforcement
|
||||
|
||||
## Requirements
|
||||
|
||||
### Requirement 1: Canonical Evidence Unit Normalization
|
||||
|
||||
**User Story:** As the aggregation engine, I want all signals normalized into a canonical EvidenceUnit shape, so that company, macro, and competitive signals flow through a single unified pipeline.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN a company signal is received, THE V3_Engine SHALL normalize the signal into an EvidenceUnit containing symbol, layer (set to "company"), event_type, source_id, source_group, timestamp, horizon (one of: intraday, 1d, 7d, 30d, 90d), direction (-1/0/+1), sentiment_strength [0,1], impact [0,1], extraction_conf [0,1], source_cred [0,1], novelty [0,1], event_base_rate (float in (0.0, 1.0]), and cluster_id
|
||||
2. WHEN a macro signal is received, THE V3_Engine SHALL normalize the signal into an EvidenceUnit with layer set to "macro", symbol mapped from the macro impact record's ticker, direction mapped from impact_direction (positive→+1, negative→-1, neutral→0), impact mapped from macro_impact_score, source_cred mapped from event confidence, extraction_conf mapped from event confidence, novelty set to 1.0 for new events, source_id mapped from the global event id, source_group set to "macro", and horizon derived from the event's estimated_duration (short_term→7d, medium_term→30d, long_term→90d)
|
||||
3. WHEN a competitive signal is received, THE V3_Engine SHALL normalize the signal into an EvidenceUnit with layer set to "competitive", symbol set to the target ticker, direction mapped from signal_direction (bullish→+1, bearish→-1, neutral→0), impact mapped from signal_strength × relationship_strength, source_cred mapped from pattern_confidence, extraction_conf set to pattern_confidence, novelty set to 1.0, source_id mapped from source_document_id, source_group set to "competitive", and horizon derived from the pattern's time_horizon field
|
||||
4. THE V3_Engine SHALL assign direction value of +1 for signals with sentiment or impact_direction equal to "positive" or "bullish", -1 for "negative" or "bearish", and 0 for "neutral" or "mixed"
|
||||
5. WHEN a signal has direction equal to 0 (neutral), THE V3_Engine SHALL include the signal in quality, coverage, and contradiction context computations but SHALL exclude the signal from directional posterior voting
|
||||
6. IF a required source field (symbol, timestamp, or source_id) is missing or null in the incoming signal, THEN THE V3_Engine SHALL reject the signal, log a warning identifying the signal source and missing field, and not produce an EvidenceUnit for that signal
|
||||
7. IF an optional numeric field (sentiment_strength, impact, extraction_conf, source_cred, novelty) is missing or null, THEN THE V3_Engine SHALL substitute a default value of 0.5 for the missing field
|
||||
8. THE V3_Engine SHALL assign event_base_rate from a configured lookup by event_type, defaulting to 0.10 when no event_type-specific base rate is configured
|
||||
|
||||
### Requirement 2: Calibrated Reliability Computation
|
||||
|
||||
**User Story:** As the scoring engine, I want to compute a calibrated reliability q_i for each signal, so that evidence quality is measured probabilistically instead of via arbitrary weight products.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Scoring_Service SHALL compute extraction reliability as q_ext = sigmoid(k_ext × (extraction_conf - m_ext)) with defaults k_ext = 8.0 and m_ext = 0.55, where extraction_conf is in [0.0, 1.0] and q_ext output is in (0.0, 1.0)
|
||||
2. THE Scoring_Service SHALL compute source reliability using Bayesian shrinkage: E[theta_s] = (alpha_0 + hits_s) / (alpha_0 + beta_0 + hits_s + misses_s) with defaults alpha_0 = 3, beta_0 = 3, and q_source = clamp((E[theta_s] - 0.50) / 0.35, 0.0, 1.0), where hits_s and misses_s are non-negative integers representing the source's historical correct and incorrect directional predictions
|
||||
3. IF a source has zero historical outcomes (hits_s = 0 AND misses_s = 0), THEN THE Scoring_Service SHALL compute q_source = 0.0 from the prior (E[theta_s] = 0.5)
|
||||
4. THE Scoring_Service SHALL compute recency reliability as q_recency = 2^(-age_hours / tau_adaptive) where tau_adaptive = tau_base × (1 + 0.75 × impact + 0.50 × surprise) and surprise = clamp(-log2(event_base_rate) / 5, 0, 1), with age_hours = max((reference_time - signal_timestamp).total_seconds() / 3600, 0.0)
|
||||
5. IF event_base_rate is unavailable or equal to zero, THEN THE Scoring_Service SHALL use a default event_base_rate of 0.10 to prevent undefined logarithm computation
|
||||
6. THE Scoring_Service SHALL use horizon-specific half-life defaults: intraday=2h, 1d=12h, 7d=72h, 30d=240h, 90d=720h
|
||||
7. THE Scoring_Service SHALL compute uniqueness as q_uniqueness = clamp(0.50 + 0.50 × novelty, 0.50, 1.00) × (1 / sqrt(1 + duplicate_count_before)), where duplicate_count_before is the number of other signals in the same cluster that were ingested before this signal
|
||||
8. THE Scoring_Service SHALL compute final signal reliability as q_i = clamp(q_ext × q_source × source_cred × q_recency × q_uniqueness, 0.0, 1.0)
|
||||
9. WHEN q_recency falls below 0.01, THE Scoring_Service SHALL apply a floor of 0.01 only for explainability display output and SHALL use the unmodified q_recency value (including zero) for posterior voting computation
|
||||
|
||||
### Requirement 3: Log-Likelihood Ratio Conversion
|
||||
|
||||
**User Story:** As the posterior engine, I want signals converted to calibrated log-likelihood ratios, so that evidence accumulation follows proper Bayesian updating rules.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Scoring_Service SHALL compute directional correctness probability as p_correct = clamp(0.50 + 0.35 × q_i × impact × sentiment_strength, 0.501, 0.85)
|
||||
2. THE Scoring_Service SHALL compute the signal log-likelihood ratio as LLR_i = direction × ln(p_correct / (1 - p_correct)), where ln denotes the natural logarithm (base e), consistent with the logit function used in posterior assembly
|
||||
3. WHEN direction equals 0 (neutral signal), THE Scoring_Service SHALL produce LLR_i = 0.0, excluding the signal from directional posterior voting while retaining it for quality and contradiction context
|
||||
4. THE Scoring_Service SHALL clamp p_correct to a minimum of 0.501 to ensure LLR_i is always nonzero for directional signals (direction ≠ 0), producing a minimum absolute LLR magnitude of approximately 0.004
|
||||
5. THE Scoring_Service SHALL clamp p_correct to a maximum of 0.85 to prevent any single signal from dominating the posterior, producing a maximum absolute LLR magnitude of approximately 1.735
|
||||
6. FOR ALL valid directional signals (direction ∈ {-1, +1}), THE Scoring_Service SHALL produce LLR_i values with the same sign as direction
|
||||
|
||||
### Requirement 4: Correlation-Aware Evidence Clustering
|
||||
|
||||
**User Story:** As the aggregation engine, I want correlated signals grouped and de-duplicated before posterior assembly, so that near-identical articles cannot inflate evidence counts.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE V3_Engine SHALL cluster signals by (symbol, horizon, event_type, source_group, time_bucket)
|
||||
2. THE V3_Engine SHALL compute effective evidence count as n_eff_c = (sum_i w_i)^2 / (sum_i w_i^2 + 2 × sum_{i<j}(rho_ij × w_i × w_j)) where w_i = abs(LLR_i)
|
||||
3. THE V3_Engine SHALL use default pairwise correlations: rho=0.80 for same wire/story/source group, rho=0.50 for same event different publisher, rho=0.25 for same theme different event, rho=0.00 for independent events
|
||||
4. THE V3_Engine SHALL compute cluster LLR as LLR_c = weighted_mean(LLR_i, abs(LLR_i)) × sqrt(n_eff_c)
|
||||
5. THE V3_Engine SHALL clamp each cluster LLR to the range [-2.5, 2.5] to prevent any single cluster from dominating the posterior
|
||||
|
||||
### Requirement 5: Posterior Assembly via Log-Odds
|
||||
|
||||
**User Story:** As the Bayesian engine, I want to assemble a posterior probability from cluster LLRs and a regime-aware prior, so that the trading decision is based on calibrated belief.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Bayesian_Service SHALL use a neutral base prior of P_prior = 0.50 unless a calibrated symbol/sector prior is stored in the risk_configs table for the given ticker or its sector
|
||||
2. THE Bayesian_Service SHALL compute posterior log-odds as logit(P_up) = logit(P_prior) + sum_c(gamma_regime × LLR_c) where gamma_regime is the regime evidence multiplier from Requirement 6 criterion 6
|
||||
3. THE Bayesian_Service SHALL compute P_up = sigmoid(logit(P_up)) = 1/(1+exp(-logit(P_up))) and P_down = 1 - P_up, clamping P_up to the range [1e-10, 1 - 1e-10] to avoid numerical boundary issues
|
||||
4. THE Bayesian_Service SHALL compute trend strength as strength = abs(2 × P_up - 1), producing a value in [0.0, 1.0] where 0.0 indicates maximum uncertainty and 1.0 indicates maximum directional conviction
|
||||
5. THE Bayesian_Service SHALL apply regime-specific direction thresholds to classify direction from P_up: panic (bullish when P_up >= 0.68, bearish when P_up <= 0.32), trend_following (bullish when P_up >= 0.60, bearish when P_up <= 0.40), mean_reversion (bullish when P_up >= 0.63, bearish when P_up <= 0.37), uncertainty (bullish when P_up >= 0.65, bearish when P_up <= 0.35). WHEN P_up falls between the bullish and bearish thresholds, THE Bayesian_Service SHALL classify direction as neutral
|
||||
6. WHEN P_prior is calibrated from the risk_configs table, THE Bayesian_Service SHALL clamp P_prior to the range [0.40, 0.60] before computing logit(P_prior)
|
||||
7. IF the risk_configs lookup for a calibrated prior fails due to a database error, THEN THE Bayesian_Service SHALL fall back to the neutral base prior of 0.50 and log a warning
|
||||
|
||||
### Requirement 6: Regime Detection v3
|
||||
|
||||
**User Story:** As the regime service, I want to classify market regimes using z-scored indicators and apply regime-appropriate evidence multipliers, so that the engine adapts its sensitivity to market conditions.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Regime_Service SHALL compute trend_z = (EMA_20 - EMA_100) / ATR_20 where EMA_20 and EMA_100 are exponential moving averages of closing prices, and ATR_20 is the 20-day Average True Range. THE Regime_Service SHALL compute vol_ratio = sigma_20 / sigma_100 where sigma_20 and sigma_100 are standard deviations of daily returns
|
||||
2. THE Regime_Service SHALL classify panic when vol_ratio > 1.5 OR abs(trend_z) > 2.5. Panic classification SHALL take priority over all other regimes
|
||||
3. IF the regime is not panic, THEN THE Regime_Service SHALL classify trend_following when abs(trend_z) >= 0.75 AND vol_ratio < 1.3
|
||||
4. IF the regime is neither panic nor trend_following, THEN THE Regime_Service SHALL classify mean_reversion when abs(trend_z) < 0.50 AND vol_ratio < 1.0
|
||||
5. THE Regime_Service SHALL classify uncertainty for all conditions not matching panic, trend_following, or mean_reversion
|
||||
6. THE Regime_Service SHALL apply regime evidence multipliers (gamma_regime): panic=0.70, trend_following=1.10, mean_reversion=0.90, uncertainty=0.80
|
||||
7. THE Regime_Service SHALL apply regime confidence multipliers: panic=0.70, trend_following=1.00, mean_reversion=0.95, uncertainty=0.85
|
||||
8. IF market data is insufficient to compute EMA_100 (fewer than 100 closing prices) or ATR_20 (fewer than 20 bars) or sigma_100 (fewer than 100 daily returns), THEN THE Regime_Service SHALL default to the uncertainty regime
|
||||
|
||||
### Requirement 7: LLR Entropy Contradiction
|
||||
|
||||
**User Story:** As the contradiction service, I want to measure meaningful opposing evidence using LLR entropy, so that contradiction reflects genuine disagreement scaled by evidence volume.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Contradiction_Service SHALL compute E_pos = sum_c(max(LLR_c, 0)) and E_neg = sum_c(max(-LLR_c, 0)) and E_total = E_pos + E_neg
|
||||
2. WHEN E_total equals zero (no directional evidence from any cluster), THE Contradiction_Service SHALL return a contradiction score of 0.0
|
||||
3. WHEN E_total is greater than zero, THE Contradiction_Service SHALL compute f_pos = E_pos / E_total and f_neg = E_neg / E_total where f_pos + f_neg = 1.0
|
||||
4. THE Contradiction_Service SHALL compute H_conflict = -f_pos × log2(f_pos) - f_neg × log2(f_neg), treating 0 × log2(0) as 0 for the boundary case. H_conflict ranges from 0.0 (all one direction) to 1.0 (equal split)
|
||||
5. THE Contradiction_Service SHALL compute volume_factor = 1 - exp(-E_total / 3.0), where 3.0 represents the evidence mass at which contradiction becomes 95% significant
|
||||
6. THE Contradiction_Service SHALL compute the final contradiction score as H_conflict × volume_factor, producing a value in [0.0, 1.0]
|
||||
7. WHEN only one direction of evidence exists (E_pos = 0 or E_neg = 0 but E_total > 0), THE Contradiction_Service SHALL return a contradiction score of 0.0
|
||||
|
||||
### Requirement 8: Multiplicative Confidence v3
|
||||
|
||||
**User Story:** As the aggregation engine, I want confidence computed multiplicatively from independent quality dimensions, so that one weak dimension suppresses the trade rather than being averaged away.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE V3_Engine SHALL compute C_evidence = 1 - exp(-n_eff_total / 5.0) where n_eff_total = sum_c(n_eff_c)
|
||||
2. THE V3_Engine SHALL compute C_quality = weighted_mean(q_i, weight=abs(LLR_i))
|
||||
3. THE V3_Engine SHALL compute confidence = clamp(C_evidence × sqrt(C_quality) × sqrt(max(strength, 0.05)) × regime_confidence_multiplier × (1 - contradiction) × data_quality_score, 0, 1)
|
||||
4. THE V3_Engine SHALL use strength = abs(2 × P_up - 1) as the directional separation term
|
||||
5. WHEN any single confidence dimension is near zero, THE V3_Engine SHALL produce a near-zero final confidence due to the multiplicative formula
|
||||
|
||||
### Requirement 9: Macro Layer v3 (Noisy-OR Exposure)
|
||||
|
||||
**User Story:** As the macro interpolation service, I want to compute normalized exposure via noisy-OR and emit macro evidence as LLR into the shared posterior, so that macro signals integrate with company evidence without special post-hoc modifiers.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE V3_Engine SHALL compute macro exposure as E_raw = 1 - product_k(1 - w_k × O_k) with default weights w_geo=0.35, w_supply=0.25, w_commodity=0.25, w_sector=0.15
|
||||
2. THE V3_Engine SHALL normalize macro exposure as E_macro = E_raw / E_max where E_max = 1 - product_k(1 - w_k)
|
||||
3. THE V3_Engine SHALL apply resilience dampener per market position tier: global_leader=0.70, multinational=0.85, regional=1.00, domestic=1.20
|
||||
4. THE V3_Engine SHALL compute macro likelihood ratio as LLR_macro = macro_direction × log(p_macro / (1 - p_macro)) where p_macro = clamp(0.50 + 0.30 × macro_impact × event_confidence × q_recency, 0.501, 0.80)
|
||||
5. THE V3_Engine SHALL feed macro LLR into the same posterior engine as company evidence without requiring a separate post-hoc modifier
|
||||
|
||||
### Requirement 10: Competitive Layer v3 (Correlation-Shrunk Propagation)
|
||||
|
||||
**User Story:** As the signal propagation service, I want to use correlation-shrunk attenuation for competitive signals, so that propagated evidence is properly discounted by distance and relationship strength.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE V3_Engine SHALL compute shrunk correlation as rho_shrunk = (n / (n + 30)) × rho_rolling + (30 / (n + 30)) × rho_prior with rho_prior_same_sector = 0.30 and rho_prior_cross_sector = 0.10
|
||||
2. THE V3_Engine SHALL compute rho_effective = max(rho_shrunk, 0) to use only positive propagation unless the relationship is explicitly inverse
|
||||
3. THE V3_Engine SHALL compute graph attenuation as attenuation = rho_effective × exp(-0.85 × d_network) with max_distance = 3
|
||||
4. THE V3_Engine SHALL compute competitive LLR as LLR_competitive = LLR_source × attenuation × pattern_confidence
|
||||
5. THE V3_Engine SHALL clamp competitive LLR to the range [-1.25, 1.25] to prevent competitive signals from dominating the posterior
|
||||
|
||||
### Requirement 11: Trend Projection v3 (Posterior State)
|
||||
|
||||
**User Story:** As the projection service, I want to project trends using a posterior state with regime-aware decay, so that projections are grounded in the same Bayesian framework as current estimates.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Projection_Service SHALL maintain an evidence state A_t = phi_regime × A_{t-1} + sum_c(LLR_c), initialized to A_0 = 0.0 when no prior state exists for a ticker-horizon pair
|
||||
2. THE Projection_Service SHALL use regime-specific decay factors: panic phi=0.35, trend_following phi=0.80, mean_reversion phi=0.55, uncertainty phi=0.50
|
||||
3. THE Projection_Service SHALL compute projected alpha as A_projected_h = phi_regime^h × A_t + expected_known_catalyst_LLR_h, where h is the projection horizon in aggregation cycles and expected_known_catalyst_LLR_h defaults to 0.0 when no known catalysts exist
|
||||
4. THE Projection_Service SHALL compute projected probability as P_up_projected_h = sigmoid(logit(P_prior_h) + A_projected_h)
|
||||
5. THE Projection_Service SHALL compute projected strength as abs(2 × P_up_projected_h - 1)
|
||||
6. THE Projection_Service SHALL flag divergence when sign(P_up_projected_h - 0.5) differs from sign(P_up_t - 0.5)
|
||||
7. WHEN market data is insufficient for regime classification, THE Projection_Service SHALL use the uncertainty decay factor (phi=0.50) as the default
|
||||
|
||||
### Requirement 12: Return Distribution and Expected Value Gate
|
||||
|
||||
**User Story:** As the eligibility service, I want to gate recommendations using a return distribution model, so that only trades with positive risk-adjusted expected value pass through.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Eligibility_Service SHALL compute horizon volatility as sigma_h = realized_vol_20d × sqrt(horizon_days / 252), where horizon_days maps to 1 (intraday/1d), 7 (7d), 30 (30d), or 90 (90d)
|
||||
2. THE Eligibility_Service SHALL compute expected return as mu_h = tanh(A_projected_h / 3.0) × confidence × sigma_h
|
||||
3. THE Eligibility_Service SHALL compute EV_long = mu_h - costs - 0.10 × CVaR_5_loss, where costs = spread_cost + slippage_estimate + commission_estimate, and CVaR_5_loss = sigma_h × 1.645 × 1.4 (Gaussian approximation of expected loss beyond the 5th percentile)
|
||||
4. THE Eligibility_Service SHALL compute regime-specific minimum edge: panic=0.0100, trend_following=0.0035, mean_reversion=0.0050, uncertainty=0.0075
|
||||
5. THE Eligibility_Service SHALL require EV_long > min_edge AND EV_long > max(0.0025, 0.25 × costs) for trade eligibility
|
||||
6. THE Eligibility_Service SHALL require confidence >= regime_confidence_min AND contradiction <= regime_contradiction_max AND n_eff_total >= 2.0 AND data_quality_score >= 0.50 for eligibility
|
||||
7. IF realized_vol_20d is unavailable (fewer than 20 trading days of price data), THEN THE Eligibility_Service SHALL use a default volatility of 0.25 (annualized) for sigma_h computation
|
||||
|
||||
### Requirement 13: Recommendation Eligibility and Mode Escalation v3
|
||||
|
||||
**User Story:** As the recommendation service, I want regime-aware eligibility gates and mode escalation, so that recommendation quality matches the rigor of the v3 posterior.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Eligibility_Service SHALL apply regime-specific eligibility thresholds: panic (confidence_min=0.70, contradiction_max=0.25, strength_min=0.36), trend_following (confidence_min=0.55, contradiction_max=0.40, strength_min=0.20), mean_reversion (confidence_min=0.60, contradiction_max=0.35, strength_min=0.26), uncertainty (confidence_min=0.65, contradiction_max=0.30, strength_min=0.30)
|
||||
2. THE Eligibility_Service SHALL map action as BUY when P_up >= bullish_threshold and EV_long > min_edge, SELL when existing position and EV_exit > EV_hold, HOLD when existing position, and WATCH otherwise
|
||||
3. THE Eligibility_Service SHALL escalate to live_eligible when action is BUY or SELL and confidence >= 0.75 and contradiction <= 0.20 and n_eff_total >= 5 and EV_long > 2 × min_edge and risk_engine_passed
|
||||
4. THE Eligibility_Service SHALL escalate to paper_eligible when action is BUY or SELL and confidence >= 0.60 and EV_long > min_edge and risk_engine_passed
|
||||
5. IF eligibility gates are not met, THEN THE Eligibility_Service SHALL assign mode as informational
|
||||
|
||||
### Requirement 14: Fractional Kelly Position Sizing
|
||||
|
||||
**User Story:** As the position sizer, I want to size positions using fractional Kelly criterion, so that position sizes are proportional to edge and constrained by risk.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Position_Sizer SHALL compute stop_distance_pct = max(ATR_pct × ATR_multiplier_regime, sigma_h × 1.25, 0.005) where ATR_pct = ATR_14 / current_price, with regime ATR multipliers: panic=2.5, trend_following=1.8, mean_reversion=1.4, uncertainty=2.0
|
||||
2. THE Position_Sizer SHALL compute reward ratio b = clamp(1.2 + 2.0 × confidence + 1.0 × strength - contradiction, 1.2, 3.0)
|
||||
3. THE Position_Sizer SHALL compute f_kelly = (p_win × b - (1 - p_win)) / b where p_win = P_up from the Bayesian posterior
|
||||
4. THE Position_Sizer SHALL compute final sizing as portfolio_pct = clamp(max(0, f_kelly) × 0.25 × confidence × data_quality_score × (1 - contradiction), 0, max_position_pct)
|
||||
5. THE Position_Sizer SHALL enforce hard caps by reducing portfolio_pct to the minimum of: max_position_pct from the active risk tier, available_sector_capacity_pct, available_correlation_capacity_pct (0 if weighted average absolute correlation with existing positions exceeds 0.80), and available_heat_capacity_pct
|
||||
6. IF portfolio_pct after all caps is less than 0.005 (0.5% of portfolio), THEN THE Position_Sizer SHALL downgrade the recommendation to WATCH with reason "position_below_minimum"
|
||||
7. IF f_kelly is less than or equal to zero, THEN THE Position_Sizer SHALL produce portfolio_pct = 0 and downgrade the recommendation to WATCH with reason "negative_edge"
|
||||
|
||||
### Requirement 15: Stop-Defined Portfolio Heat
|
||||
|
||||
**User Story:** As the risk engine, I want portfolio heat calculated from stop-defined risk dollars, so that risk measurement reflects actual loss exposure rather than position notional.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Risk_Engine SHALL compute risk_dollars = position_value × stop_distance_pct for each open position
|
||||
2. THE Risk_Engine SHALL compute portfolio_heat = sum of risk_dollars across all open positions
|
||||
3. IF portfolio_heat exceeds max_portfolio_heat × portfolio_value, THEN THE Risk_Engine SHALL reject new position entries
|
||||
4. THE Position_Sizer SHALL compute available_heat_capacity = max_portfolio_heat × portfolio_value - current_portfolio_heat
|
||||
5. THE Position_Sizer SHALL reject a position when the new risk_dollars would exceed available_heat_capacity
|
||||
|
||||
### Requirement 16: Regime-Aware Stop Loss and Take Profit
|
||||
|
||||
**User Story:** As the stop loss manager, I want stops and targets computed from regime-aware volatility and dynamic reward ratios, so that exit levels adapt to current market conditions.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Stop_Loss_Manager SHALL compute stop_distance_pct = max(ATR_pct × regime_ATR_multiplier, sigma_h × z_stop, min_stop_pct) with z_stop = 1.25 and min_stop_pct = 0.005
|
||||
2. THE Stop_Loss_Manager SHALL compute stop_loss = entry_price × (1 - stop_distance_pct) for long positions
|
||||
3. THE Stop_Loss_Manager SHALL compute take_profit = entry_price × (1 + b × stop_distance_pct) where b = clamp(1.2 + 2.0 × confidence + 1.0 × strength - contradiction, 1.2, 3.0)
|
||||
4. THE Stop_Loss_Manager SHALL activate trailing stop when unrealized_gain_pct >= 0.50 × take_profit_distance_pct
|
||||
5. THE Stop_Loss_Manager SHALL compute trailing_stop = max(existing_stop, current_price × (1 - trailing_distance_pct)) where trailing_distance_pct = max(ATR_pct × trailing_ATR_mult, sigma_h × 0.75)
|
||||
|
||||
### Requirement 17: Data Quality v3 (Multiplicative Fail-Closed)
|
||||
|
||||
**User Story:** As the suppression layer, I want data quality computed as a multiplicative fail-closed metric, so that a single catastrophic quality failure suppresses the entire recommendation.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE V3_Engine SHALL compute Q_parse = 1 - extraction_failure_rate
|
||||
2. THE V3_Engine SHALL compute Q_conf = weighted_mean(extraction_conf_i, weight=impact_i)
|
||||
3. THE V3_Engine SHALL compute Q_fresh = exp(-age_newest_hours / 168)
|
||||
4. THE V3_Engine SHALL compute Q_coverage = 1 - exp(-N_valid / 5)
|
||||
5. THE V3_Engine SHALL compute Q_diversity = min(1, log2(1 + N_source_types) / log2(4))
|
||||
6. THE V3_Engine SHALL compute data_quality_score = clamp(Q_parse × sqrt(Q_conf) × Q_fresh × Q_coverage × Q_diversity, 0, 1)
|
||||
7. IF data_quality_score < 0.50 OR N_valid < 2 OR Q_parse < 0.50, THEN THE V3_Engine SHALL force the recommendation to informational mode
|
||||
8. IF company evidence is zero and only macro or competitive evidence exists, THEN THE V3_Engine SHALL force the recommendation to informational mode unless macro_only_enabled is configured
|
||||
|
||||
### Requirement 18: Risk Tier Auto-Adjustment v3
|
||||
|
||||
**User Story:** As the risk tier controller, I want tier adjustments based on risk-adjusted performance metrics, so that the engine self-corrects when performance degrades.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Risk_Engine SHALL track profit_factor_30d (gross_profit / gross_loss over 30 days), max_drawdown_30d (largest peak-to-trough portfolio decline over 30 days as a fraction), calibration_error (mean absolute difference between predicted P_up and realized binary outcome over 30 days), and realized_sharpe_30d (annualized Sharpe ratio of daily returns over 30 days)
|
||||
2. THE Risk_Engine SHALL evaluate tier adjustment conditions once per calendar day after the trading session closes
|
||||
3. THE Risk_Engine SHALL downgrade one tier if any condition is met: profit_factor_30d < 1.0 OR max_drawdown_30d > 0.12 OR calibration_error > 0.20 OR realized_sharpe_30d < 0
|
||||
4. THE Risk_Engine SHALL upgrade one tier only if all conditions are met: profit_factor_30d > 1.35 AND max_drawdown_30d < 0.05 AND calibration_error < 0.12 AND reserve_pool > 0.20 AND N_trades_30d >= 20
|
||||
5. IF a downgrade condition is triggered, THEN THE Risk_Engine SHALL apply the downgrade immediately without waiting for an upgrade evaluation
|
||||
6. THE Risk_Engine SHALL enforce a minimum cooldown of 7 calendar days between consecutive upgrade evaluations to prevent tier oscillation
|
||||
|
||||
### Requirement 19: Feature Flag and Fallback
|
||||
|
||||
**User Story:** As an operator, I want the v3 engine gated behind a runtime feature flag with the heuristic mode as fallback, so that the upgrade can be rolled out safely without downtime.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHILE `v3_engine_enabled` is False, THE V3_Engine SHALL route all aggregation through the existing heuristic pipeline without any v3 computation
|
||||
2. WHILE `v3_engine_enabled` is True, THE V3_Engine SHALL route all aggregation through the v3 calibrated evidence pipeline
|
||||
3. THE V3_Engine SHALL read the `v3_engine_enabled` flag from the risk_configs table at the start of each aggregation cycle without requiring a service restart
|
||||
4. IF the v3 pipeline encounters an unhandled error during aggregation, THEN THE V3_Engine SHALL log the error at ERROR level with full traceback and fall back to heuristic mode for that aggregation cycle, recording the fallback event in output metadata
|
||||
5. THE V3_Engine SHALL store a `pipeline_mode` field value of "v3" or "heuristic" in all output records (TrendSummary, Recommendation) JSONB metadata indicating which pipeline produced the result
|
||||
6. IF the `v3_engine_enabled` flag cannot be read from the database (connection error or missing row), THEN THE V3_Engine SHALL default to heuristic mode and log a warning
|
||||
|
||||
### Requirement 20: Database Compatibility and Output Contract
|
||||
|
||||
**User Story:** As the system architect, I want v3 output stored in existing tables using JSONB metadata, so that no schema migration or downtime is required.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE V3_Engine SHALL store posterior fields (p_up, log_odds, strength, confidence, contradiction, n_eff, data_quality) in the existing TrendSummary JSONB metadata column
|
||||
2. THE V3_Engine SHALL store return model fields (mu_h, sigma_h, ev_long, min_edge) in the Recommendation JSONB metadata column
|
||||
3. THE V3_Engine SHALL expose an explainability payload containing top_positive_clusters, top_negative_clusters, suppression_reasons, and risk_adjustments
|
||||
4. THE V3_Engine SHALL preserve the existing WeightedSignal abstraction as an intermediate representation before LLR conversion
|
||||
5. THE V3_Engine SHALL preserve all existing database table schemas without requiring new migrations for core functionality
|
||||
|
||||
### Requirement 21: Mathematical Correctness Properties
|
||||
|
||||
**User Story:** As a developer, I want property-based tests validating all v3 mathematical invariants, so that correctness is verified across the input space.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. FOR ALL valid EvidenceUnits, THE V3_Engine SHALL produce q_i values in the range [0, 1]
|
||||
2. FOR ALL valid q_i values, THE V3_Engine SHALL produce p_correct values in the range [0.501, 0.85]
|
||||
3. FOR ALL valid signals with direction != 0, THE V3_Engine SHALL produce LLR values with the same sign as direction
|
||||
4. FOR ALL valid cluster configurations, THE V3_Engine SHALL produce n_eff_c values satisfying 0 < n_eff_c <= N (where N is the cluster size)
|
||||
5. FOR ALL valid cluster LLRs, THE Bayesian_Service SHALL produce P_up values in the range (0, 1) exclusive
|
||||
6. FOR ALL valid inputs, THE V3_Engine SHALL produce confidence values in the range [0, 1]
|
||||
7. FOR ALL valid inputs with no opposing evidence, THE Contradiction_Service SHALL produce a contradiction score of 0
|
||||
8. FOR ALL valid inputs, THE Position_Sizer SHALL produce portfolio_pct values in the range [0, max_position_pct]
|
||||
9. FOR ALL valid inputs where f_kelly <= 0, THE Position_Sizer SHALL produce a portfolio_pct of 0
|
||||
10. FOR ALL valid EvidenceUnit sequences, serializing the posterior state to JSON and deserializing SHALL produce an equivalent posterior state (round-trip property)
|
||||
@@ -0,0 +1,451 @@
|
||||
# Implementation Plan: Math Core v3 Engine
|
||||
|
||||
## Overview
|
||||
|
||||
Incremental upgrade of the Stonks Oracle signal processing engine from dual-mode heuristic/probabilistic to the v3 Calibrated Evidence Engine. Implementation follows the v3 math doc section 20 order: EvidenceUnit → reliability → LLR → clustering → posterior → contradiction → confidence → macro/competitive layers → EV gate → Kelly sizing → stop-defined heat → retire heuristic scoring. All v3 code operates behind the `v3_engine_enabled` feature flag with heuristic fallback.
|
||||
|
||||
## Tasks
|
||||
|
||||
- [x] 1. EvidenceUnit and LLR conversion behind feature flag
|
||||
- [x] 1.1 Implement EvidenceUnit dataclass and normalization functions in `services/aggregation/scoring.py`
|
||||
- Add frozen dataclass `EvidenceUnit` with all 16 fields (symbol, layer, event_type, source_id, source_group, timestamp, horizon, direction, sentiment_strength, impact, extraction_conf, source_cred, novelty, event_base_rate, cluster_id)
|
||||
- Implement `normalize_company_signal()`, `normalize_macro_signal()`, `normalize_competitive_signal()`
|
||||
- Validate required fields (symbol, timestamp, source_id) — reject with warning on missing
|
||||
- Substitute 0.5 for missing optional numeric fields
|
||||
- Map direction from sentiment/impact_direction strings (+1/-1/0)
|
||||
- Assign event_base_rate from EVENT_TYPE_BASE_RATES lookup (default 0.10)
|
||||
- _Requirements: 1.1, 1.2, 1.3, 1.4, 1.5, 1.6, 1.7, 1.8_
|
||||
|
||||
- [x] 1.2 Implement calibrated reliability pipeline (`compute_v3_reliability`) in `services/aggregation/scoring.py`
|
||||
- Add `SourceStats` dataclass (source_id, hits, misses, alpha_0=3, beta_0=3)
|
||||
- Add `ReliabilityComponents` dataclass (q_ext, q_source, q_recency, q_uniqueness, q_i)
|
||||
- Implement q_ext = sigmoid(8.0 × (extraction_conf - 0.55))
|
||||
- Implement q_source via Bayesian shrinkage: E[theta_s] = (alpha_0 + hits) / (alpha_0 + beta_0 + hits + misses), then clamp((E - 0.50) / 0.35, 0, 1)
|
||||
- Implement q_recency = 2^(-age_hours / tau_adaptive) with adaptive half-life formula
|
||||
- Implement q_uniqueness = clamp(0.5 + 0.5 × novelty, 0.5, 1.0) × (1 / sqrt(1 + dup_count))
|
||||
- Implement q_i = clamp(q_ext × q_source × source_cred × q_recency × q_uniqueness, 0, 1)
|
||||
- Apply floor of 0.01 on q_recency only for explainability display
|
||||
- _Requirements: 2.1, 2.2, 2.3, 2.4, 2.5, 2.6, 2.7, 2.8, 2.9_
|
||||
|
||||
- [x] 1.3 Implement LLR conversion (`compute_llr`) in `services/aggregation/scoring.py`
|
||||
- Compute p_correct = clamp(0.50 + 0.35 × q_i × impact × sentiment_strength, 0.501, 0.85)
|
||||
- Compute LLR_i = direction × ln(p_correct / (1 - p_correct))
|
||||
- Return 0.0 for neutral signals (direction == 0)
|
||||
- Ensure LLR sign always matches direction for directional signals
|
||||
- _Requirements: 3.1, 3.2, 3.3, 3.4, 3.5, 3.6_
|
||||
|
||||
- [x] 1.4 Add feature flag routing in `services/aggregation/worker.py`
|
||||
- Read `v3_engine_enabled` from risk_configs table at start of each aggregation cycle
|
||||
- Route to v3 pipeline when True, heuristic when False
|
||||
- Default to heuristic mode if DB read fails (log warning)
|
||||
- Wrap v3 pipeline in try/except — fall back to heuristic on unhandled error (log ERROR with traceback)
|
||||
- Store `pipeline_mode` field ("v3" or "heuristic") in output metadata
|
||||
- _Requirements: 19.1, 19.2, 19.3, 19.4, 19.5, 19.6_
|
||||
|
||||
- [x] 1.5 Write property tests for EvidenceUnit, reliability, and LLR (`tests/test_pbt_v3_reliability.py`)
|
||||
- **Property 1: Reliability q_i is bounded in [0, 1]**
|
||||
- **Property 2: p_correct is bounded in [0.501, 0.85]**
|
||||
- **Property 3: LLR sign matches direction and magnitude is bounded**
|
||||
- **Property 4: Neutral signals produce zero LLR**
|
||||
- **Property 21: EvidenceUnit normalization preserves field ranges**
|
||||
- **Validates: Requirements 1.1–1.8, 2.1–2.9, 3.1–3.6, 21.1–21.3**
|
||||
- Use Hypothesis with `@settings(max_examples=100)`
|
||||
|
||||
- [x] 1.6 Write unit tests for EvidenceUnit normalization and LLR conversion (`tests/test_v3_evidence_unit.py`)
|
||||
- Test company signal → EvidenceUnit with correct field mapping
|
||||
- Test macro signal → EvidenceUnit with correct horizon mapping (short_term→7d, medium_term→30d, long_term→90d)
|
||||
- Test competitive signal → EvidenceUnit with correct direction mapping
|
||||
- Test missing required fields → rejection with warning
|
||||
- Test missing optional fields → default 0.5 substitution
|
||||
- Test known inputs through full q_i pipeline → expected outputs
|
||||
- Test LLR boundary cases: p_correct at clamp boundaries
|
||||
- _Requirements: 1.1–1.8, 2.1–2.9, 3.1–3.6_
|
||||
|
||||
- [x] 2. Evidence clustering and n_eff
|
||||
|
||||
- [x] 2.1 Implement correlation-aware clustering in `services/aggregation/worker.py`
|
||||
- Add `EvidenceCluster` dataclass (cluster_id, units, llrs, n_eff, cluster_llr)
|
||||
- Implement `cluster_evidence()` — group EvidenceUnits by (symbol, horizon, event_type, source_group, time_bucket)
|
||||
- Compute cluster_id as hash of grouping key
|
||||
- Define time_bucket resolution per horizon (intraday=1h, 1d=4h, 7d=24h, 30d=72h, 90d=168h)
|
||||
- _Requirements: 4.1_
|
||||
|
||||
- [x] 2.2 Implement n_eff computation in `services/aggregation/worker.py`
|
||||
- Implement `compute_n_eff(llrs, correlations)` using formula: (sum w_i)² / (sum w_i² + 2 × sum_{i<j} rho_ij × w_i × w_j)
|
||||
- Use default pairwise correlations: same wire=0.80, same event diff publisher=0.50, same theme diff event=0.25, independent=0.00
|
||||
- Guard against division by zero (denominator → return n_eff=1.0)
|
||||
- _Requirements: 4.2, 4.3_
|
||||
|
||||
- [x] 2.3 Implement cluster LLR computation in `services/aggregation/worker.py`
|
||||
- Implement `compute_cluster_llr(llrs, n_eff)` = clamp(weighted_mean(LLR_i, |LLR_i|) × sqrt(n_eff), -2.5, 2.5)
|
||||
- Handle single-signal clusters (LLR_c = LLR_i clamped)
|
||||
- Handle all-zero LLR clusters (cluster_llr = 0.0)
|
||||
- _Requirements: 4.4, 4.5_
|
||||
|
||||
- [x] 2.4 Write property tests for clustering (`tests/test_pbt_v3_clustering.py`)
|
||||
- **Property 5: Effective evidence count n_eff is bounded by cluster size**
|
||||
- **Property 6: Cluster LLR is clamped to [-2.5, 2.5]**
|
||||
- **Validates: Requirements 4.2, 4.3, 4.4, 4.5, 21.4**
|
||||
- Use Hypothesis with `@settings(max_examples=100)`
|
||||
|
||||
- [x] 2.5 Write unit tests for clustering (`tests/test_v3_clustering.py`)
|
||||
- Test 3 identical articles from same source → n_eff < 3
|
||||
- Test 3 independent articles → n_eff ≈ 3
|
||||
- Test single signal cluster → n_eff = 1.0
|
||||
- Test cluster LLR clamp at ±2.5
|
||||
- Test grouping by correct key dimensions
|
||||
- _Requirements: 4.1–4.5_
|
||||
|
||||
- [x] 3. Checkpoint - Verify foundation layer
|
||||
- Ensure all tests pass for EvidenceUnit, reliability, LLR, and clustering.
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [x] 4. Replace trend assembly with posterior P_up
|
||||
|
||||
- [x] 4.1 Implement regime detection v3 in `services/aggregation/regime.py`
|
||||
- Add `V3RegimeClassification` dataclass (regime, trend_z, vol_ratio, evidence_multiplier, confidence_multiplier, phi_decay, atr_multiplier)
|
||||
- Implement `classify_regime_v3(closing_prices, daily_returns, atr_20)` using ATR-normalized trend_z = (EMA_20 - EMA_100) / ATR_20
|
||||
- Compute vol_ratio = sigma_20 / sigma_100
|
||||
- Classification rules in priority: panic (vol_ratio > 1.5 OR |trend_z| > 2.5), trend_following (|trend_z| >= 0.75 AND vol_ratio < 1.3), mean_reversion (|trend_z| < 0.50 AND vol_ratio < 1.0), uncertainty (default)
|
||||
- Assign regime parameters: gamma, confidence_mult, phi, ATR_mult, min_edge
|
||||
- Default to uncertainty when data insufficient
|
||||
- _Requirements: 6.1, 6.2, 6.3, 6.4, 6.5, 6.6, 6.7, 6.8_
|
||||
|
||||
- [x] 4.2 Implement posterior assembly via log-odds in `services/aggregation/bayesian.py`
|
||||
- Add `V3Posterior` dataclass (p_up, p_down, log_odds, strength, direction, n_eff_total, regime)
|
||||
- Implement `compute_v3_posterior(clusters, regime, p_prior=0.50)`
|
||||
- Compute logit(P_up) = logit(P_prior) + sum(gamma_regime × LLR_c)
|
||||
- Compute P_up = sigmoid(log_odds), clamp to [1e-10, 1 - 1e-10]
|
||||
- Compute strength = abs(2 × P_up - 1)
|
||||
- Classify direction using regime-specific thresholds (panic: 0.68/0.32, trend_following: 0.60/0.40, mean_reversion: 0.63/0.37, uncertainty: 0.65/0.35)
|
||||
- Load calibrated prior from risk_configs if available (clamp to [0.40, 0.60]), fall back to 0.50 on error
|
||||
- _Requirements: 5.1, 5.2, 5.3, 5.4, 5.5, 5.6, 5.7_
|
||||
|
||||
- [x] 4.3 Write property tests for posterior and regime (`tests/test_pbt_v3_posterior.py`)
|
||||
- **Property 7: Posterior P_up is in open interval (0, 1)**
|
||||
- **Property 20: Regime classification is exhaustive and deterministic**
|
||||
- **Validates: Requirements 5.3, 6.2–6.5, 21.5**
|
||||
- Use Hypothesis with `@settings(max_examples=100)`
|
||||
|
||||
- [x] 4.4 Write unit tests for posterior assembly (`tests/test_v3_posterior.py`)
|
||||
- Test empty clusters → P_up = 0.50 (neutral prior)
|
||||
- Test all bullish clusters → P_up > 0.50
|
||||
- Test regime direction thresholds at boundary values
|
||||
- Test prior clamp [0.40, 0.60]
|
||||
- Test regime classification with known inputs
|
||||
- _Requirements: 5.1–5.7, 6.1–6.8_
|
||||
|
||||
- [x] 5. Replace contradiction with LLR entropy
|
||||
|
||||
- [x] 5.1 Implement LLR entropy contradiction in `services/aggregation/contradiction.py`
|
||||
- Add `compute_v3_contradiction(clusters: list[EvidenceCluster]) -> float`
|
||||
- Compute E_pos = sum(max(LLR_c, 0)), E_neg = sum(max(-LLR_c, 0)), E_total = E_pos + E_neg
|
||||
- When E_total == 0 → return 0.0
|
||||
- When only one direction exists (E_pos == 0 or E_neg == 0) → return 0.0
|
||||
- Compute f_pos = E_pos / E_total, f_neg = E_neg / E_total
|
||||
- Compute H_conflict = -f_pos × log2(f_pos) - f_neg × log2(f_neg), treating 0×log2(0) = 0
|
||||
- Compute volume_factor = 1 - exp(-E_total / 3.0)
|
||||
- Return H_conflict × volume_factor, bounded in [0.0, 1.0]
|
||||
- _Requirements: 7.1, 7.2, 7.3, 7.4, 7.5, 7.6, 7.7_
|
||||
|
||||
- [x] 5.2 Write property tests for contradiction (`tests/test_pbt_v3_posterior.py`)
|
||||
- **Property 8: Contradiction is zero when evidence is unidirectional**
|
||||
- **Property 9: Contradiction score is bounded in [0, 1]**
|
||||
- **Validates: Requirements 7.6, 7.7, 21.7**
|
||||
- Use Hypothesis with `@settings(max_examples=100)`
|
||||
|
||||
- [x] 5.3 Write unit tests for LLR entropy contradiction (`tests/test_v3_contradiction.py`)
|
||||
- Test all bullish clusters → contradiction = 0.0
|
||||
- Test equal split of evidence → high contradiction near 1.0
|
||||
- Test E_total = 0 → contradiction = 0.0
|
||||
- Test volume_factor growth: small evidence mass → suppressed score
|
||||
- _Requirements: 7.1–7.7_
|
||||
|
||||
- [x] 6. Replace confidence formula
|
||||
|
||||
- [x] 6.1 Implement multiplicative confidence v3 in `services/aggregation/worker.py`
|
||||
- Add `compute_v3_confidence(n_eff_total, q_values, llrs, strength, regime_confidence_mult, contradiction, data_quality) -> float`
|
||||
- Compute C_evidence = 1 - exp(-n_eff_total / 5.0)
|
||||
- Compute C_quality = weighted_mean(q_i, weight=|LLR_i|)
|
||||
- Compute confidence = clamp(C_evidence × sqrt(C_quality) × sqrt(max(strength, 0.05)) × regime_confidence_mult × (1 - contradiction) × data_quality, 0, 1)
|
||||
- _Requirements: 8.1, 8.2, 8.3, 8.4, 8.5_
|
||||
|
||||
- [x] 6.2 Implement data quality v3 in `services/aggregation/worker.py`
|
||||
- Add `compute_v3_data_quality(units, extraction_failure_rate, age_newest_hours, n_source_types) -> float`
|
||||
- Q_parse = 1 - extraction_failure_rate
|
||||
- Q_conf = weighted_mean(extraction_conf_i, weight=impact_i)
|
||||
- Q_fresh = exp(-age_newest_hours / 168)
|
||||
- Q_coverage = 1 - exp(-N_valid / 5)
|
||||
- Q_diversity = min(1, log2(1 + N_source_types) / log2(4))
|
||||
- data_quality_score = clamp(Q_parse × sqrt(Q_conf) × Q_fresh × Q_coverage × Q_diversity, 0, 1)
|
||||
- Force informational mode when data_quality < 0.50 OR N_valid < 2 OR Q_parse < 0.50
|
||||
- Force informational when only macro/competitive evidence unless macro_only_enabled
|
||||
- _Requirements: 17.1, 17.2, 17.3, 17.4, 17.5, 17.6, 17.7, 17.8_
|
||||
|
||||
- [x] 6.3 Write property tests for confidence and data quality (`tests/test_pbt_v3_posterior.py`)
|
||||
- **Property 10: Multiplicative confidence is bounded in [0, 1] and suppressed by weak dimensions**
|
||||
- **Property 17: Data quality score is bounded in [0, 1]**
|
||||
- **Validates: Requirements 8.3, 8.5, 17.6, 21.6**
|
||||
- Use Hypothesis with `@settings(max_examples=100)`
|
||||
|
||||
- [x] 6.4 Write unit tests for confidence and data quality (`tests/test_v3_confidence.py`)
|
||||
- Test zero data quality → near-zero confidence
|
||||
- Test full contradiction (1.0) → zero confidence
|
||||
- Test low n_eff → suppressed C_evidence
|
||||
- Test data quality boundary cases (Q_parse < 0.50 → informational)
|
||||
- _Requirements: 8.1–8.5, 17.1–17.8_
|
||||
|
||||
- [x] 7. Checkpoint - Verify core pipeline
|
||||
- Ensure all tests pass for posterior, contradiction, confidence, and data quality.
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [x] 8. Convert macro and competitive layers to emit LLR
|
||||
|
||||
- [x] 8.1 Implement noisy-OR macro exposure and LLR emission in `services/aggregation/interpolation.py`
|
||||
- Add `compute_normalized_macro_exposure(overlaps: dict[str, float]) -> float`
|
||||
- E_raw = 1 - product(1 - w_k × O_k) with weights: w_geo=0.35, w_supply=0.25, w_commodity=0.25, w_sector=0.15
|
||||
- E_max = 1 - product(1 - w_k)
|
||||
- E_macro = E_raw / E_max (normalized to [0, 1])
|
||||
- Apply resilience dampener per tier: global_leader=0.70, multinational=0.85, regional=1.00, domestic=1.20
|
||||
- Add `compute_macro_llr(macro_impact, event_confidence, q_recency, macro_direction) -> float`
|
||||
- p_macro = clamp(0.50 + 0.30 × macro_impact × event_confidence × q_recency, 0.501, 0.80)
|
||||
- LLR_macro = macro_direction × ln(p_macro / (1 - p_macro))
|
||||
- Feed macro LLR into shared posterior without separate post-hoc modifier
|
||||
- _Requirements: 9.1, 9.2, 9.3, 9.4, 9.5_
|
||||
|
||||
- [x] 8.2 Implement correlation-shrunk competitive propagation in `services/aggregation/signal_propagation.py`
|
||||
- Add `compute_shrunk_correlation(rho_rolling, n_observations, same_sector) -> float`
|
||||
- rho_prior = 0.30 (same_sector) or 0.10 (cross_sector)
|
||||
- rho_shrunk = (n/(n+30)) × rho_rolling + (30/(n+30)) × rho_prior
|
||||
- rho_effective = max(rho_shrunk, 0)
|
||||
- Add `compute_competitive_llr(llr_source, rho_effective, d_network, pattern_confidence) -> float`
|
||||
- attenuation = rho_effective × exp(-0.85 × d_network), max_distance = 3
|
||||
- LLR_competitive = clamp(llr_source × attenuation × pattern_confidence, -1.25, 1.25)
|
||||
- _Requirements: 10.1, 10.2, 10.3, 10.4, 10.5_
|
||||
|
||||
- [x] 8.3 Write property tests for macro and competitive layers (`tests/test_pbt_v3_layers.py`)
|
||||
- **Property 13: Noisy-OR normalized exposure is bounded in [0, 1]**
|
||||
- **Property 14: Competitive LLR is clamped to [-1.25, 1.25]**
|
||||
- **Property 15: Graph attenuation is zero beyond max distance**
|
||||
- **Validates: Requirements 9.1, 9.2, 10.3, 10.4, 10.5**
|
||||
- Use Hypothesis with `@settings(max_examples=100)`
|
||||
|
||||
- [x] 8.4 Write unit tests for macro and competitive layers (`tests/test_v3_layers.py`)
|
||||
- Test noisy-OR: all O_k = 1.0 → E_macro = 1.0; all O_k = 0 → E_macro = 0
|
||||
- Test resilience dampener per tier
|
||||
- Test macro LLR at boundary values
|
||||
- Test shrunk correlation convergence (n → ∞ approaches rho_rolling)
|
||||
- Test competitive LLR clamp at ±1.25
|
||||
- Test distance > 3 → zero attenuation
|
||||
- _Requirements: 9.1–9.5, 10.1–10.5_
|
||||
|
||||
- [x] 9. Replace EV gate with expected-return distribution
|
||||
|
||||
- [x] 9.1 Implement posterior state projection in `services/aggregation/projection.py`
|
||||
- Add `V3ProjectionState` dataclass (a_t, p_up_projected, projected_strength, diverges, phi_regime)
|
||||
- Implement `compute_v3_projection(a_prev, cluster_llrs, regime, p_prior, projection_horizon, known_catalyst_llr=0.0)`
|
||||
- Evidence state: A_t = phi_regime × A_{t-1} + sum(LLR_c), init A_0 = 0.0
|
||||
- Regime decay: panic=0.35, trend_following=0.80, mean_reversion=0.55, uncertainty=0.50
|
||||
- Projected alpha: A_projected = phi^h × A_t + known_catalyst_LLR
|
||||
- P_up_projected = sigmoid(logit(P_prior) + A_projected)
|
||||
- Projected strength = abs(2 × P_up_projected - 1)
|
||||
- Flag divergence when sign(P_up_projected - 0.5) ≠ sign(P_up_t - 0.5)
|
||||
- _Requirements: 11.1, 11.2, 11.3, 11.4, 11.5, 11.6, 11.7_
|
||||
|
||||
- [x] 9.2 Implement return distribution and EV gate in `services/recommendation/eligibility.py`
|
||||
- Add `ReturnDistribution` dataclass (sigma_h, mu_h, ev_long, min_edge, eligible)
|
||||
- Add `compute_return_distribution(a_projected, confidence, realized_vol_20d, horizon_days, costs, regime)`
|
||||
- sigma_h = realized_vol_20d × sqrt(horizon_days / 252); default vol = 0.25 if unavailable
|
||||
- mu_h = tanh(A_projected / 3.0) × confidence × sigma_h
|
||||
- CVaR_5 = sigma_h × 1.645 × 1.4
|
||||
- EV_long = mu_h - costs - 0.10 × CVaR_5
|
||||
- Regime min_edge: panic=0.0100, trend_following=0.0035, mean_reversion=0.0050, uncertainty=0.0075
|
||||
- Eligibility: EV_long > min_edge AND EV_long > max(0.0025, 0.25 × costs)
|
||||
- Also require: confidence >= regime_confidence_min, contradiction <= regime_contradiction_max, n_eff_total >= 2.0, data_quality >= 0.50
|
||||
- _Requirements: 12.1, 12.2, 12.3, 12.4, 12.5, 12.6, 12.7_
|
||||
|
||||
- [x] 9.3 Implement regime-aware eligibility and mode escalation in `services/recommendation/eligibility.py`
|
||||
- Add regime-specific eligibility thresholds: panic (conf≥0.70, contra≤0.25, str≥0.36), trend_following (conf≥0.55, contra≤0.40, str≥0.20), mean_reversion (conf≥0.60, contra≤0.35, str≥0.26), uncertainty (conf≥0.65, contra≤0.30, str≥0.30)
|
||||
- Action mapping: BUY when P_up >= bullish_threshold and EV > min_edge; SELL when existing position and EV_exit > EV_hold; HOLD when existing; WATCH otherwise
|
||||
- live_eligible: BUY/SELL + conf >= 0.75 + contra <= 0.20 + n_eff >= 5 + EV > 2×min_edge + risk_passed
|
||||
- paper_eligible: BUY/SELL + conf >= 0.60 + EV > min_edge + risk_passed
|
||||
- Otherwise: informational
|
||||
- _Requirements: 13.1, 13.2, 13.3, 13.4, 13.5_
|
||||
|
||||
- [x] 9.4 Write property tests for projection (`tests/test_pbt_v3_projection.py`)
|
||||
- **Property 16: Projection evidence state decays toward zero**
|
||||
- **Validates: Requirements 11.1, 11.3**
|
||||
- Use Hypothesis with `@settings(max_examples=100)`
|
||||
|
||||
- [x] 9.5 Write unit tests for EV gate and eligibility (`tests/test_v3_eligibility.py`)
|
||||
- Test EV_long positive → eligible
|
||||
- Test EV_long negative → ineligible
|
||||
- Test regime-specific min_edge thresholds
|
||||
- Test mode escalation: live vs paper vs informational
|
||||
- Test projection decay convergence
|
||||
- Test divergence flag behavior
|
||||
- _Requirements: 11.1–11.7, 12.1–12.7, 13.1–13.5_
|
||||
|
||||
- [x] 10. Checkpoint - Verify decision layer
|
||||
- Ensure all tests pass for projection, EV gate, eligibility, and layer integrations.
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [x] 11. Replace sizing with fractional Kelly under existing risk caps
|
||||
|
||||
- [x] 11.1 Implement fractional Kelly position sizing in `services/trading/position_sizer.py`
|
||||
- Add `compute_kelly_sizing(p_win, b, confidence, data_quality, contradiction, max_position_pct, available_caps) -> float`
|
||||
- Compute reward ratio b = clamp(1.2 + 2.0 × confidence + 1.0 × strength - contradiction, 1.2, 3.0)
|
||||
- Compute f_kelly = (p_win × b - (1 - p_win)) / b
|
||||
- portfolio_pct = clamp(max(0, f_kelly) × 0.25 × confidence × data_quality × (1 - contradiction), 0, max_position_pct)
|
||||
- Apply min of: max_position_pct, sector_capacity, correlation_capacity (0 if avg corr > 0.80), heat_capacity
|
||||
- If portfolio_pct < 0.005 → downgrade to WATCH (reason: position_below_minimum)
|
||||
- If f_kelly <= 0 → portfolio_pct = 0, downgrade to WATCH (reason: negative_edge)
|
||||
- _Requirements: 14.1, 14.2, 14.3, 14.4, 14.5, 14.6, 14.7_
|
||||
|
||||
- [x] 11.2 Implement regime-aware stop loss and take profit in `services/trading/stop_loss_manager.py`
|
||||
- Add v3 stop computation: stop_distance_pct = max(ATR_pct × regime_ATR_mult, sigma_h × 1.25, 0.005)
|
||||
- stop_loss = entry_price × (1 - stop_distance_pct)
|
||||
- take_profit = entry_price × (1 + b × stop_distance_pct) where b = reward ratio
|
||||
- Activate trailing stop when unrealized_gain >= 0.50 × TP distance
|
||||
- trailing_stop = max(existing_stop, current_price × (1 - trailing_distance_pct))
|
||||
- trailing_distance_pct = max(ATR_pct × trailing_ATR_mult, sigma_h × 0.75)
|
||||
- Trailing stop must be monotonically non-decreasing
|
||||
- _Requirements: 16.1, 16.2, 16.3, 16.4, 16.5_
|
||||
|
||||
- [x] 11.3 Write property tests for Kelly sizing and stops (`tests/test_pbt_v3_decision.py`)
|
||||
- **Property 11: Fractional Kelly sizing is bounded and respects negative edge**
|
||||
- **Property 18: Stop loss is below entry price and take profit is above**
|
||||
- **Property 19: Trailing stop never decreases**
|
||||
- **Validates: Requirements 14.4, 14.7, 16.2, 16.3, 16.5, 21.8, 21.9**
|
||||
- Use Hypothesis with `@settings(max_examples=100)`
|
||||
|
||||
- [x] 11.4 Write unit tests for Kelly sizing and stops (`tests/test_v3_sizing.py`)
|
||||
- Test p_win=0.3, b=2 → f_kelly < 0 → size = 0
|
||||
- Test p_win=0.7, b=2 → positive size within caps
|
||||
- Test cap enforcement (sector, correlation, heat)
|
||||
- Test position_below_minimum downgrade
|
||||
- Test stop/TP computation with known inputs
|
||||
- Test trailing stop monotonicity over a price sequence
|
||||
- _Requirements: 14.1–14.7, 16.1–16.5_
|
||||
|
||||
- [x] 12. Replace portfolio heat with stop-defined risk dollars
|
||||
|
||||
- [x] 12.1 Implement stop-defined portfolio heat in `services/risk/engine.py`
|
||||
- Add `compute_portfolio_heat(positions, stop_distances) -> float`
|
||||
- risk_dollars = position_value × stop_distance_pct for each position
|
||||
- portfolio_heat = sum(risk_dollars)
|
||||
- Add `check_heat_capacity(current_heat, new_risk_dollars, max_heat_pct, portfolio_value) -> bool`
|
||||
- Reject new entry when current_heat + new_risk_dollars > max_heat_pct × portfolio_value
|
||||
- Integrate available_heat_capacity into Kelly sizing pipeline
|
||||
- _Requirements: 15.1, 15.2, 15.3, 15.4, 15.5_
|
||||
|
||||
- [x] 12.2 Implement risk tier auto-adjustment v3 in `services/risk/engine.py`
|
||||
- Add `TierMetrics` dataclass (profit_factor_30d, max_drawdown_30d, calibration_error, realized_sharpe_30d, n_trades_30d, reserve_pool_pct)
|
||||
- Add `evaluate_tier_adjustment(metrics) -> str` returning 'upgrade'|'downgrade'|'hold'
|
||||
- Downgrade if ANY: profit_factor < 1.0 OR drawdown > 0.12 OR calibration_error > 0.20 OR sharpe < 0
|
||||
- Upgrade only if ALL: profit_factor > 1.35 AND drawdown < 0.05 AND calibration_error < 0.12 AND reserve > 0.20 AND N_trades >= 20
|
||||
- Apply downgrade immediately; enforce 7-day upgrade cooldown
|
||||
- Evaluate once per calendar day after session close
|
||||
- _Requirements: 18.1, 18.2, 18.3, 18.4, 18.5, 18.6_
|
||||
|
||||
- [x] 12.3 Write property tests for heat and tier adjustment (`tests/test_pbt_v3_decision.py`)
|
||||
- **Property 22: Portfolio heat rejection is correct**
|
||||
- **Property 23: Tier auto-adjustment obeys downgrade-any, upgrade-all logic**
|
||||
- **Validates: Requirements 15.3, 15.5, 18.3, 18.4**
|
||||
- Use Hypothesis with `@settings(max_examples=100)`
|
||||
|
||||
- [x] 12.4 Write unit tests for heat and tier (`tests/test_v3_risk.py`)
|
||||
- Test heat computation: 3 positions with known stops → expected heat
|
||||
- Test heat rejection: heat at limit → new entry blocked
|
||||
- Test tier downgrade: single bad metric triggers downgrade
|
||||
- Test tier upgrade: all metrics good → upgrade
|
||||
- Test tier upgrade: one metric bad → hold (not upgrade)
|
||||
- Test 7-day cooldown enforcement
|
||||
- _Requirements: 15.1–15.5, 18.1–18.6_
|
||||
|
||||
- [x] 13. Checkpoint - Verify sizing and risk layer
|
||||
- Ensure all tests pass for Kelly sizing, stops, heat, and tier adjustment.
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [x] 14. Retire heuristic scoring to explainability-only mode
|
||||
|
||||
- [x] 14.1 Wire v3 pipeline end-to-end in `services/aggregation/worker.py`
|
||||
- Orchestrate full pipeline: EvidenceUnit → q_i → LLR → cluster → posterior → contradiction → confidence → data_quality
|
||||
- Store v3 posterior in TrendSummary JSONB metadata (p_up, log_odds, strength, confidence, contradiction, n_eff, data_quality, regime)
|
||||
- Store return model in Recommendation JSONB metadata (mu_h, sigma_h, ev_long, min_edge)
|
||||
- Store explainability payload (top_positive_clusters, top_negative_clusters, suppression_reasons, risk_adjustments)
|
||||
- Preserve WeightedSignal as intermediate representation before LLR conversion
|
||||
- _Requirements: 20.1, 20.2, 20.3, 20.4, 20.5_
|
||||
|
||||
- [x] 14.2 Retain heuristic pipeline as fallback with explainability overlay
|
||||
- Keep existing heuristic scoring path fully functional (no removal)
|
||||
- Mark heuristic outputs with `pipeline_mode: "heuristic"` in metadata
|
||||
- Ensure heuristic mode still produces valid TrendSummary and Recommendation objects
|
||||
- Test feature flag toggle: v3 → heuristic → v3 round-trip
|
||||
- _Requirements: 19.1, 19.2, 19.5, 20.4_
|
||||
|
||||
- [x] 14.3 Write property test for JSON round-trip (`tests/test_pbt_v3_posterior.py`)
|
||||
- **Property 12: Posterior state JSON round-trip**
|
||||
- **Validates: Requirements 20.1, 21.10**
|
||||
- Serialize V3Posterior to JSON and deserialize, verify equivalence within 1e-10
|
||||
- Use Hypothesis with `@settings(max_examples=100)`
|
||||
|
||||
- [x] 14.4 Write integration tests for full pipeline (`tests/test_v3_integration.py`)
|
||||
- Test full path: raw signals → EvidenceUnit → q_i → LLR → cluster → posterior → recommendation
|
||||
- Test feature flag false → heuristic path, flag true → v3 path
|
||||
- Test v3 exception → heuristic fallback + error logged
|
||||
- Test output JSONB contains expected v3 fields
|
||||
- _Requirements: 19.1–19.6, 20.1–20.5_
|
||||
|
||||
- [x] 15. Final checkpoint - All tests green
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
- Run full test suite: `.venv/bin/python -m pytest tests/test_pbt_v3_*.py tests/test_v3_*.py -x --tb=short -q`
|
||||
- Verify no regressions in existing heuristic pipeline tests
|
||||
|
||||
## Notes
|
||||
|
||||
- Tasks marked with `*` are optional and can be skipped for faster MVP
|
||||
- Each task references specific requirements for traceability
|
||||
- Checkpoints ensure incremental validation after each logical phase
|
||||
- Property tests validate the 23 correctness properties defined in the design
|
||||
- Unit tests validate specific examples, edge cases, and error handling
|
||||
- The implementation preserves the existing heuristic pipeline as a fully functional fallback
|
||||
- All v3 code is gated behind `v3_engine_enabled` — no changes to production behavior until flag is flipped
|
||||
- Run tests with: `.venv/bin/python -m pytest tests/ -x --tb=short -q`
|
||||
- Property tests use Hypothesis: `@settings(max_examples=100)`
|
||||
|
||||
## Task Dependency Graph
|
||||
|
||||
```json
|
||||
{
|
||||
"waves": [
|
||||
{ "id": 0, "tasks": ["1.1"] },
|
||||
{ "id": 1, "tasks": ["1.2", "1.4"] },
|
||||
{ "id": 2, "tasks": ["1.3"] },
|
||||
{ "id": 3, "tasks": ["1.5", "1.6"] },
|
||||
{ "id": 4, "tasks": ["2.1"] },
|
||||
{ "id": 5, "tasks": ["2.2"] },
|
||||
{ "id": 6, "tasks": ["2.3"] },
|
||||
{ "id": 7, "tasks": ["2.4", "2.5"] },
|
||||
{ "id": 8, "tasks": ["4.1"] },
|
||||
{ "id": 9, "tasks": ["4.2"] },
|
||||
{ "id": 10, "tasks": ["4.3", "4.4"] },
|
||||
{ "id": 11, "tasks": ["5.1"] },
|
||||
{ "id": 12, "tasks": ["5.2", "5.3"] },
|
||||
{ "id": 13, "tasks": ["6.1", "6.2"] },
|
||||
{ "id": 14, "tasks": ["6.3", "6.4"] },
|
||||
{ "id": 15, "tasks": ["8.1", "8.2"] },
|
||||
{ "id": 16, "tasks": ["8.3", "8.4"] },
|
||||
{ "id": 17, "tasks": ["9.1"] },
|
||||
{ "id": 18, "tasks": ["9.2", "9.3"] },
|
||||
{ "id": 19, "tasks": ["9.4", "9.5"] },
|
||||
{ "id": 20, "tasks": ["11.1", "11.2"] },
|
||||
{ "id": 21, "tasks": ["11.3", "11.4"] },
|
||||
{ "id": 22, "tasks": ["12.1", "12.2"] },
|
||||
{ "id": 23, "tasks": ["12.3", "12.4"] },
|
||||
{ "id": 24, "tasks": ["14.1"] },
|
||||
{ "id": 25, "tasks": ["14.2"] },
|
||||
{ "id": 26, "tasks": ["14.3", "14.4"] }
|
||||
]
|
||||
}
|
||||
```
|
||||
@@ -0,0 +1 @@
|
||||
{"specId": "b595d834-7e72-4fab-87a9-65c92115a069", "workflowType": "requirements-first", "specType": "feature"}
|
||||
@@ -0,0 +1,975 @@
|
||||
# Design Document — Model Validation, Calibration, and Signal Quality
|
||||
|
||||
## Overview
|
||||
|
||||
This design adds a closed-loop model validation layer to Stonks Oracle. The system currently generates trend summaries and trading recommendations with confidence scores, but has no mechanism to evaluate whether those predictions are accurate, whether confidence scores are well-calibrated, which sources contribute to correct predictions, or whether the system outperforms simple benchmarks.
|
||||
|
||||
The validation layer introduces six new service modules under `services/validation/`, a quality gate in `services/trading/`, seven new API endpoints under `/api/validation/`, a database migration (035) with four new tables and two SQL views, and an upgraded OpsModel dashboard page. The architecture follows the existing patterns: pure computation modules with asyncpg for persistence, FastAPI endpoints in `services/api/app.py`, and React/TanStack Query hooks on the frontend.
|
||||
|
||||
### Design Rationale
|
||||
|
||||
A prediction engine without outcome tracking is flying blind. The validation layer closes the feedback loop by:
|
||||
|
||||
1. **Capturing immutable snapshots** at prediction time — preventing hindsight bias in evaluation
|
||||
2. **Evaluating outcomes** across multiple horizons (1h, 6h, 1d, 7d, 30d) — matching the system's multi-window trend architecture
|
||||
3. **Computing calibration metrics** (ECE, Brier score) — measuring whether confidence scores mean what they claim
|
||||
4. **Tracking information coefficients** (IC, Rank IC) — measuring linear and ordinal predictive power
|
||||
5. **Attributing performance** to sources, catalysts, and signal layers — identifying the most valuable information channels
|
||||
6. **Recalibrating confidence** via Bayesian shrinkage — learning from the system's own track record
|
||||
7. **Gating live trading** on minimum quality thresholds — preventing real capital risk on a poorly performing model
|
||||
|
||||
The design reuses existing infrastructure (asyncpg, FastAPI, TanStack Query, Recharts) and integrates with the existing `source_accuracy` table from the signal-math-upgrade spec.
|
||||
|
||||
---
|
||||
|
||||
## Architecture
|
||||
|
||||
### High-Level Data Flow
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
subgraph "Prediction Capture (Real-time)"
|
||||
A[Recommendation Engine] -->|generates| B[Prediction_Snapshot_Writer]
|
||||
B --> C[prediction_snapshots table]
|
||||
B --> D[signal_evidence_links table]
|
||||
B -->|computes| E[canonical_evidence_key<br/>duplicate detection<br/>contribution scores]
|
||||
end
|
||||
|
||||
subgraph "Outcome Evaluation (Periodic)"
|
||||
F[Outcome_Evaluator<br/>scheduled job] -->|reads matured snapshots| C
|
||||
F -->|fetches future prices| G[market_snapshots table]
|
||||
F -->|computes returns| H[prediction_outcomes table]
|
||||
F -->|evaluates 5 horizons| H
|
||||
end
|
||||
|
||||
subgraph "Metrics Computation (Periodic)"
|
||||
I[Metrics_Engine] -->|reads| H
|
||||
I -->|reads| C
|
||||
I -->|reads| D
|
||||
I -->|computes| J[model_metric_snapshots table]
|
||||
I -->|computes| K[Calibration: ECE, Brier]
|
||||
I -->|computes| L[IC, Rank IC by horizon]
|
||||
I -->|computes| M[Benchmark: excess returns]
|
||||
end
|
||||
|
||||
subgraph "Attribution (Periodic)"
|
||||
N[Attribution_Engine] -->|joins| D
|
||||
N -->|joins| H
|
||||
N -->|computes| O[Per-source metrics]
|
||||
N -->|computes| P[Per-catalyst metrics]
|
||||
N -->|computes| Q[Per-layer metrics]
|
||||
end
|
||||
|
||||
subgraph "Calibration (Periodic)"
|
||||
R[Calibration_Engine] -->|reads| H
|
||||
R -->|reads| D
|
||||
R -->|computes Bayesian shrinkage| S[source_accuracy table<br/>reliability scores]
|
||||
end
|
||||
|
||||
subgraph "Safety Gate (Per-cycle)"
|
||||
T[Quality_Gate] -->|reads latest| J
|
||||
T -->|evaluates thresholds| U{Pass?}
|
||||
U -->|yes| V[Live trading allowed]
|
||||
U -->|no| W[Force paper mode]
|
||||
T -->|stores result| X[risk_configs table<br/>model_quality_gate key]
|
||||
end
|
||||
|
||||
subgraph "Dashboard (Frontend)"
|
||||
Y[Dashboard_API<br/>7 endpoints] -->|reads| J
|
||||
Y -->|reads| C
|
||||
Y -->|reads| H
|
||||
Y -->|reads| D
|
||||
Z[OpsModel.tsx<br/>upgraded page] -->|fetches| Y
|
||||
end
|
||||
|
||||
subgraph "Backtest Integration"
|
||||
AA[BacktestReplay] -->|validation mode| B
|
||||
AA -->|validation mode| F
|
||||
AA -->|triggers| I
|
||||
end
|
||||
```
|
||||
|
||||
### Scheduling Strategy
|
||||
|
||||
The validation components run on different cadences:
|
||||
|
||||
| Component | Trigger | Cadence |
|
||||
|-----------|---------|---------|
|
||||
| Prediction_Snapshot_Writer | Synchronous — called by recommendation engine | Every recommendation |
|
||||
| Outcome_Evaluator | Scheduled job | Every 1 hour |
|
||||
| Metrics_Engine | After Outcome_Evaluator completes | Every 1 hour |
|
||||
| Attribution_Engine | Called by Metrics_Engine | Every 1 hour |
|
||||
| Calibration_Engine | After Metrics_Engine completes | Every 6 hours |
|
||||
| Quality_Gate | Start of each aggregation cycle | Every aggregation cycle |
|
||||
|
||||
### Sector ETF Mapping
|
||||
|
||||
The system needs a mapping from company sectors to sector ETFs for benchmark comparison. This is stored as a configuration constant:
|
||||
|
||||
```python
|
||||
SECTOR_ETF_MAP: dict[str, str] = {
|
||||
"Technology": "XLK",
|
||||
"Consumer Cyclical": "XLY",
|
||||
"Financial Services": "XLF",
|
||||
"Healthcare": "XLV",
|
||||
"Energy": "XLE",
|
||||
"Communication Services": "XLC",
|
||||
"Industrials": "XLI",
|
||||
"Consumer Defensive": "XLP",
|
||||
"Real Estate": "XLRE",
|
||||
"Utilities": "XLU",
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Components and Interfaces
|
||||
|
||||
### New Modules
|
||||
|
||||
| Module | File | Responsibility |
|
||||
|--------|------|----------------|
|
||||
| Prediction Snapshot Writer | `services/validation/prediction_snapshot.py` | Captures immutable prediction state at generation time |
|
||||
| Outcome Evaluator | `services/validation/outcome_evaluator.py` | Matches predictions with realized market outcomes |
|
||||
| Metrics Engine | `services/validation/metrics.py` | Computes calibration, IC, Brier, benchmark metrics |
|
||||
| Attribution Engine | `services/validation/attribution.py` | Per-source, per-catalyst, per-layer performance |
|
||||
| Calibration Engine | `services/validation/calibration.py` | Bayesian shrinkage source reliability, weight adjustment |
|
||||
| Quality Gate | `services/trading/model_quality_gate.py` | Safety gate for live trading eligibility |
|
||||
|
||||
### Modified Modules
|
||||
|
||||
| Module | File | Changes |
|
||||
|--------|------|---------|
|
||||
| Query API | `services/api/app.py` | 7 new `/api/validation/*` endpoints |
|
||||
| Aggregation Worker | `services/aggregation/worker.py` | Call Quality_Gate at cycle start |
|
||||
| Recommendation Engine | `services/recommendation/eligibility.py` | Call Prediction_Snapshot_Writer after recommendation |
|
||||
| Backtest Replay | `services/trading/backtest_replay.py` | Validation mode support |
|
||||
| Frontend Hooks | `frontend/src/api/hooks.ts` | 7 new validation hooks |
|
||||
| OpsModel Page | `frontend/src/pages/OpsModel.tsx` | Full dashboard upgrade |
|
||||
| AppLayout | `frontend/src/components/AppLayout.tsx` | Nav item update (if needed) |
|
||||
|
||||
### Component Interface Details
|
||||
|
||||
#### 1. Prediction Snapshot Writer (`services/validation/prediction_snapshot.py`)
|
||||
|
||||
```python
|
||||
SECTOR_ETF_MAP: dict[str, str] = {
|
||||
"Technology": "XLK",
|
||||
"Consumer Cyclical": "XLY",
|
||||
"Financial Services": "XLF",
|
||||
"Healthcare": "XLV",
|
||||
"Energy": "XLE",
|
||||
"Communication Services": "XLC",
|
||||
"Industrials": "XLI",
|
||||
"Consumer Defensive": "XLP",
|
||||
"Real Estate": "XLRE",
|
||||
"Utilities": "XLU",
|
||||
}
|
||||
|
||||
EVALUATION_HORIZONS: list[str] = ["1h", "6h", "1d", "7d", "30d"]
|
||||
|
||||
MAX_SINGLE_DOCUMENT_WEIGHT: float = 1.0
|
||||
|
||||
|
||||
@dataclass
|
||||
class PredictionSnapshot:
|
||||
"""Immutable snapshot of a prediction at generation time."""
|
||||
id: str # UUID
|
||||
generated_at: datetime
|
||||
ticker: str
|
||||
window: str
|
||||
horizon: str
|
||||
direction: str # bullish/bearish/mixed/neutral
|
||||
action: str # buy/sell/hold/watch
|
||||
mode: str # informational/paper_eligible/live_eligible
|
||||
strength: float
|
||||
confidence: float
|
||||
contradiction: float
|
||||
p_bull: float | None
|
||||
p_bear: float | None
|
||||
score_company: float
|
||||
score_macro: float
|
||||
score_competitive: float
|
||||
evidence_count: int
|
||||
unique_source_count: int
|
||||
duplicate_evidence_count: int
|
||||
price_at_prediction: float | None
|
||||
spy_price_at_prediction: float | None
|
||||
sector_etf_price_at_prediction: float | None
|
||||
metadata: dict
|
||||
|
||||
|
||||
@dataclass
|
||||
class SignalEvidenceLink:
|
||||
"""Link between a prediction and a contributing evidence document."""
|
||||
id: str # UUID
|
||||
prediction_id: str
|
||||
document_id: str
|
||||
signal_id: str
|
||||
ticker: str
|
||||
source: str
|
||||
source_type: str
|
||||
catalyst_type: str
|
||||
sentiment: str
|
||||
impact: float
|
||||
extraction_confidence: float
|
||||
weight: float # clamped to MAX_SINGLE_DOCUMENT_WEIGHT
|
||||
is_duplicate: bool
|
||||
canonical_evidence_key: str
|
||||
contribution_score: float # weight / total_weight, sums to 1.0
|
||||
metadata: dict
|
||||
|
||||
|
||||
def compute_canonical_evidence_key(title: str, url: str) -> str:
|
||||
"""SHA256 of normalized(title) + normalized(url).
|
||||
|
||||
Normalization: lowercase, strip whitespace for title;
|
||||
lowercase, strip query params for URL.
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
async def create_prediction_snapshot(
|
||||
pool: asyncpg.Pool,
|
||||
recommendation: Recommendation,
|
||||
trend_summary: TrendSummary,
|
||||
evidence_signals: list[WeightedSignal],
|
||||
evidence_docs: list[dict], # document metadata from recommendation_evidence
|
||||
) -> PredictionSnapshot:
|
||||
"""Create and persist a prediction snapshot with evidence links.
|
||||
|
||||
1. Fetches current prices (ticker, SPY, sector ETF) from market_snapshots
|
||||
2. Computes canonical evidence keys and duplicate detection
|
||||
3. Clamps individual document weights to MAX_SINGLE_DOCUMENT_WEIGHT
|
||||
4. Computes contribution scores (one-vote-per-canonical-key dedup)
|
||||
5. Persists snapshot and evidence links in a transaction
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
async def fetch_latest_close_price(
|
||||
pool: asyncpg.Pool,
|
||||
ticker: str,
|
||||
) -> float | None:
|
||||
"""Fetch most recent close price from market_snapshots for a ticker."""
|
||||
...
|
||||
```
|
||||
|
||||
#### 2. Outcome Evaluator (`services/validation/outcome_evaluator.py`)
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class PredictionOutcome:
|
||||
"""Realized outcome for a prediction at a specific horizon."""
|
||||
id: str # UUID
|
||||
prediction_id: str
|
||||
evaluated_at: datetime
|
||||
horizon: str # 1h, 6h, 1d, 7d, 30d
|
||||
future_price: float
|
||||
future_return: float
|
||||
spy_future_price: float | None
|
||||
spy_return: float | None
|
||||
sector_etf_future_price: float | None
|
||||
sector_etf_return: float | None
|
||||
excess_return_vs_spy: float | None
|
||||
excess_return_vs_sector: float | None
|
||||
direction_correct: bool
|
||||
profitable: bool
|
||||
metadata: dict
|
||||
|
||||
|
||||
HORIZON_DURATIONS: dict[str, timedelta] = {
|
||||
"1h": timedelta(hours=1),
|
||||
"6h": timedelta(hours=6),
|
||||
"1d": timedelta(days=1),
|
||||
"7d": timedelta(days=7),
|
||||
"30d": timedelta(days=30),
|
||||
}
|
||||
|
||||
|
||||
async def evaluate_matured_predictions(
|
||||
pool: asyncpg.Pool,
|
||||
) -> int:
|
||||
"""Evaluate all matured prediction snapshots.
|
||||
|
||||
Finds snapshots where horizon has elapsed and outcome not yet recorded.
|
||||
For each, fetches future prices and computes returns.
|
||||
Skips horizons where future price is unavailable (retries next run).
|
||||
|
||||
Returns count of outcomes recorded.
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
async def evaluate_single_prediction(
|
||||
pool: asyncpg.Pool,
|
||||
snapshot: PredictionSnapshot,
|
||||
horizon: str,
|
||||
) -> PredictionOutcome | None:
|
||||
"""Evaluate a single prediction at a specific horizon.
|
||||
|
||||
Returns None if future price is unavailable.
|
||||
"""
|
||||
...
|
||||
```
|
||||
|
||||
#### 3. Metrics Engine (`services/validation/metrics.py`)
|
||||
|
||||
```python
|
||||
CONFIDENCE_BUCKETS: list[tuple[float, float]] = [
|
||||
(0.50, 0.60),
|
||||
(0.60, 0.70),
|
||||
(0.70, 0.80),
|
||||
(0.80, 0.90),
|
||||
(0.90, 1.00),
|
||||
]
|
||||
|
||||
LOOKBACK_WINDOWS: list[str] = ["7d", "30d", "90d", "all"]
|
||||
|
||||
|
||||
@dataclass
|
||||
class CalibrationBucket:
|
||||
"""Calibration metrics for a single confidence bucket."""
|
||||
bucket_low: float
|
||||
bucket_high: float
|
||||
avg_confidence: float
|
||||
observed_win_rate: float
|
||||
prediction_count: int
|
||||
miscalibrated: bool # |avg_confidence - win_rate| > 0.15
|
||||
|
||||
|
||||
@dataclass
|
||||
class ModelMetricSnapshot:
|
||||
"""Aggregate model quality metrics for a lookback/horizon combination."""
|
||||
id: str
|
||||
generated_at: datetime
|
||||
lookback_window: str
|
||||
horizon: str
|
||||
prediction_count: int
|
||||
win_rate: float
|
||||
directional_accuracy: float
|
||||
information_coefficient: float | None
|
||||
rank_information_coefficient: float | None
|
||||
avg_return: float
|
||||
avg_excess_return_vs_spy: float
|
||||
avg_excess_return_vs_sector: float
|
||||
calibration_error: float # ECE
|
||||
brier_score: float
|
||||
buy_win_rate: float
|
||||
sell_win_rate: float
|
||||
hold_win_rate: float
|
||||
metadata: dict
|
||||
|
||||
|
||||
def compute_calibration_error(
|
||||
confidences: list[float],
|
||||
outcomes: list[bool],
|
||||
) -> tuple[float, list[CalibrationBucket]]:
|
||||
"""Compute ECE and calibration buckets.
|
||||
|
||||
ECE = Σ (n_b / N) * |avg_conf_b - win_rate_b|
|
||||
|
||||
Returns (ece, buckets).
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
def compute_brier_score(
|
||||
p_bulls: list[float],
|
||||
outcomes: list[bool],
|
||||
) -> float:
|
||||
"""Brier score = mean((p_bull - outcome)^2).
|
||||
|
||||
outcome is 1.0 when price moved in predicted direction, 0.0 otherwise.
|
||||
Returns value in [0.0, 1.0].
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
def compute_information_coefficient(
|
||||
scores: list[float],
|
||||
returns: list[float],
|
||||
) -> float | None:
|
||||
"""Pearson correlation between prediction scores and future returns.
|
||||
|
||||
Returns None when fewer than 30 data points.
|
||||
Returns value in [-1.0, 1.0].
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
def compute_rank_information_coefficient(
|
||||
scores: list[float],
|
||||
returns: list[float],
|
||||
) -> float | None:
|
||||
"""Spearman rank correlation between prediction scores and future returns.
|
||||
|
||||
Returns None when fewer than 30 data points.
|
||||
Returns value in [-1.0, 1.0].
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
def compute_contribution_scores(
|
||||
weights: list[float],
|
||||
) -> list[float]:
|
||||
"""Compute contribution scores from document weights.
|
||||
|
||||
Each score = weight_i / sum(weights). Sums to 1.0.
|
||||
Each score in [0.0, 1.0].
|
||||
Returns empty list for empty input.
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
async def compute_and_store_metric_snapshots(
|
||||
pool: asyncpg.Pool,
|
||||
) -> list[ModelMetricSnapshot]:
|
||||
"""Compute metric snapshots for all lookback/horizon combinations.
|
||||
|
||||
Lookback windows: 7d, 30d, 90d, all-time.
|
||||
Horizons: 1h, 6h, 1d, 7d, 30d.
|
||||
"""
|
||||
...
|
||||
```
|
||||
|
||||
#### 4. Attribution Engine (`services/validation/attribution.py`)
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class SourceAttribution:
|
||||
"""Performance metrics for a single source."""
|
||||
source: str
|
||||
source_type: str
|
||||
prediction_count: int
|
||||
avg_weight: float
|
||||
avg_contribution_score: float
|
||||
win_rate: float
|
||||
avg_future_return: float
|
||||
avg_excess_return_vs_spy: float
|
||||
information_coefficient: float | None
|
||||
duplicate_rate: float
|
||||
|
||||
|
||||
@dataclass
|
||||
class CatalystAttribution:
|
||||
"""Performance metrics for a single catalyst type."""
|
||||
catalyst_type: str
|
||||
prediction_count: int
|
||||
win_rate: float
|
||||
avg_future_return: float
|
||||
avg_excess_return_vs_spy: float
|
||||
information_coefficient: float | None
|
||||
|
||||
|
||||
@dataclass
|
||||
class LayerAttribution:
|
||||
"""Performance metrics for a signal layer."""
|
||||
layer: str # company, macro, competitive
|
||||
avg_contribution_pct: float
|
||||
dominant_win_rate: float # win rate when this layer > 30% contribution
|
||||
dominant_ic: float | None # IC when this layer > 30% contribution
|
||||
|
||||
|
||||
async def compute_source_attribution(
|
||||
pool: asyncpg.Pool,
|
||||
lookback_days: int = 30,
|
||||
horizon: str = "7d",
|
||||
) -> list[SourceAttribution]:
|
||||
...
|
||||
|
||||
|
||||
async def compute_catalyst_attribution(
|
||||
pool: asyncpg.Pool,
|
||||
lookback_days: int = 30,
|
||||
horizon: str = "7d",
|
||||
) -> list[CatalystAttribution]:
|
||||
...
|
||||
|
||||
|
||||
async def compute_layer_attribution(
|
||||
pool: asyncpg.Pool,
|
||||
lookback_days: int = 30,
|
||||
horizon: str = "7d",
|
||||
) -> list[LayerAttribution]:
|
||||
...
|
||||
```
|
||||
|
||||
#### 5. Calibration Engine (`services/validation/calibration.py`)
|
||||
|
||||
```python
|
||||
def compute_source_reliability(
|
||||
observed_win_rate: float,
|
||||
sample_count: int,
|
||||
prior_strength: int = 30,
|
||||
) -> float:
|
||||
"""Bayesian shrinkage source reliability.
|
||||
|
||||
reliability = 0.5 + (n / (n + prior_strength)) * (observed_win_rate - 0.5)
|
||||
|
||||
Returns value in [0.0, 1.0].
|
||||
When n=0, returns 0.5 (prior mean).
|
||||
As n→∞, approaches observed_win_rate.
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
def compute_adjusted_evidence_weight(
|
||||
base_weight: float,
|
||||
reliability: float,
|
||||
) -> float:
|
||||
"""Adjusted weight = base_weight * (0.5 + reliability), clamped to [0.1, 2.0]."""
|
||||
...
|
||||
|
||||
|
||||
async def update_source_reliabilities(
|
||||
pool: asyncpg.Pool,
|
||||
) -> int:
|
||||
"""Recompute and store source reliability scores from latest outcomes.
|
||||
|
||||
Uses the existing source_accuracy table, updating accuracy_ratio
|
||||
with the Bayesian shrinkage formula.
|
||||
|
||||
Returns count of sources updated.
|
||||
"""
|
||||
...
|
||||
```
|
||||
|
||||
#### 6. Quality Gate (`services/trading/model_quality_gate.py`)
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class QualityGateConfig:
|
||||
"""Configurable thresholds for live trading eligibility."""
|
||||
min_prediction_count: int = 100
|
||||
min_ic: float = 0.03
|
||||
min_win_rate: float = 0.53
|
||||
max_ece: float = 0.15
|
||||
min_excess_return_vs_spy: float = 0.0
|
||||
max_snapshot_age_hours: int = 24
|
||||
|
||||
|
||||
@dataclass
|
||||
class GateThresholdResult:
|
||||
"""Result for a single threshold check."""
|
||||
name: str
|
||||
threshold: float
|
||||
actual: float
|
||||
passed: bool
|
||||
|
||||
|
||||
@dataclass
|
||||
class QualityGateResult:
|
||||
"""Full gate evaluation result."""
|
||||
passed: bool
|
||||
evaluated_at: datetime
|
||||
threshold_results: list[GateThresholdResult]
|
||||
reason: str # "all thresholds met" or "failed: ..."
|
||||
snapshot_id: str | None
|
||||
config: QualityGateConfig
|
||||
|
||||
|
||||
async def evaluate_quality_gate(
|
||||
pool: asyncpg.Pool,
|
||||
config: QualityGateConfig | None = None,
|
||||
) -> QualityGateResult:
|
||||
"""Evaluate model quality gate from latest metric snapshot.
|
||||
|
||||
Reads the most recent model_metric_snapshot for the 30d lookback
|
||||
and 7d horizon (the primary evaluation window).
|
||||
|
||||
If no snapshot exists or snapshot is stale (>24h), defaults to
|
||||
paper-only mode (fail-safe).
|
||||
|
||||
Stores result in risk_configs under 'model_quality_gate' key.
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
async def load_gate_config_from_db(
|
||||
pool: asyncpg.Pool,
|
||||
) -> QualityGateConfig:
|
||||
"""Load gate thresholds from risk_configs, with defaults."""
|
||||
...
|
||||
```
|
||||
|
||||
#### 7. Dashboard API Endpoints
|
||||
|
||||
Seven new endpoints added to `services/api/app.py`:
|
||||
|
||||
| Endpoint | Method | Returns |
|
||||
|----------|--------|---------|
|
||||
| `/api/validation/summary` | GET | Latest model metric snapshot + gate status |
|
||||
| `/api/validation/calibration` | GET | Calibration table with buckets |
|
||||
| `/api/validation/ic-by-horizon` | GET | IC and Rank IC per horizon |
|
||||
| `/api/validation/attribution/sources` | GET | Per-source performance |
|
||||
| `/api/validation/attribution/catalysts` | GET | Per-catalyst performance |
|
||||
| `/api/validation/attribution/layers` | GET | Per-layer performance |
|
||||
| `/api/validation/gate-status` | GET | Quality gate evaluation detail |
|
||||
|
||||
All endpoints accept optional `lookback` (default "30d") and `horizon` (default "7d") query parameters.
|
||||
|
||||
---
|
||||
|
||||
## Data Models
|
||||
|
||||
### Database Schema (Migration 035)
|
||||
|
||||
#### prediction_snapshots
|
||||
|
||||
```sql
|
||||
CREATE TABLE IF NOT EXISTS prediction_snapshots (
|
||||
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||
generated_at TIMESTAMPTZ NOT NULL,
|
||||
ticker VARCHAR(20) NOT NULL,
|
||||
window VARCHAR(20) NOT NULL,
|
||||
horizon VARCHAR(20) NOT NULL,
|
||||
direction VARCHAR(20) NOT NULL,
|
||||
action VARCHAR(20) NOT NULL,
|
||||
mode VARCHAR(30) NOT NULL,
|
||||
strength FLOAT NOT NULL,
|
||||
confidence FLOAT NOT NULL,
|
||||
contradiction FLOAT NOT NULL DEFAULT 0.0,
|
||||
p_bull FLOAT,
|
||||
p_bear FLOAT,
|
||||
score_company FLOAT NOT NULL DEFAULT 0.0,
|
||||
score_macro FLOAT NOT NULL DEFAULT 0.0,
|
||||
score_competitive FLOAT NOT NULL DEFAULT 0.0,
|
||||
evidence_count INTEGER NOT NULL DEFAULT 0,
|
||||
unique_source_count INTEGER NOT NULL DEFAULT 0,
|
||||
duplicate_evidence_count INTEGER NOT NULL DEFAULT 0,
|
||||
price_at_prediction FLOAT,
|
||||
spy_price_at_prediction FLOAT,
|
||||
sector_etf_price_at_prediction FLOAT,
|
||||
metadata JSONB DEFAULT '{}',
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
|
||||
);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS idx_pred_snap_ticker ON prediction_snapshots(ticker);
|
||||
CREATE INDEX IF NOT EXISTS idx_pred_snap_generated ON prediction_snapshots(generated_at);
|
||||
CREATE INDEX IF NOT EXISTS idx_pred_snap_horizon ON prediction_snapshots(horizon);
|
||||
```
|
||||
|
||||
#### prediction_outcomes
|
||||
|
||||
```sql
|
||||
CREATE TABLE IF NOT EXISTS prediction_outcomes (
|
||||
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||
prediction_id UUID NOT NULL REFERENCES prediction_snapshots(id),
|
||||
evaluated_at TIMESTAMPTZ NOT NULL,
|
||||
horizon VARCHAR(20) NOT NULL,
|
||||
future_price FLOAT,
|
||||
future_return FLOAT,
|
||||
spy_future_price FLOAT,
|
||||
spy_return FLOAT,
|
||||
sector_etf_future_price FLOAT,
|
||||
sector_etf_return FLOAT,
|
||||
excess_return_vs_spy FLOAT,
|
||||
excess_return_vs_sector FLOAT,
|
||||
direction_correct BOOLEAN,
|
||||
profitable BOOLEAN,
|
||||
metadata JSONB DEFAULT '{}',
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
|
||||
);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS idx_pred_out_prediction ON prediction_outcomes(prediction_id);
|
||||
CREATE INDEX IF NOT EXISTS idx_pred_out_horizon ON prediction_outcomes(horizon);
|
||||
CREATE INDEX IF NOT EXISTS idx_pred_out_evaluated ON prediction_outcomes(evaluated_at);
|
||||
```
|
||||
|
||||
#### signal_evidence_links
|
||||
|
||||
```sql
|
||||
CREATE TABLE IF NOT EXISTS signal_evidence_links (
|
||||
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||
prediction_id UUID NOT NULL REFERENCES prediction_snapshots(id),
|
||||
document_id VARCHAR(200),
|
||||
signal_id VARCHAR(200),
|
||||
ticker VARCHAR(20),
|
||||
source VARCHAR(200),
|
||||
source_type VARCHAR(50),
|
||||
catalyst_type VARCHAR(50),
|
||||
sentiment VARCHAR(20),
|
||||
impact FLOAT,
|
||||
extraction_confidence FLOAT,
|
||||
weight FLOAT,
|
||||
is_duplicate BOOLEAN NOT NULL DEFAULT FALSE,
|
||||
canonical_evidence_key VARCHAR(64),
|
||||
contribution_score FLOAT,
|
||||
metadata JSONB DEFAULT '{}',
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
|
||||
);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS idx_sig_ev_prediction ON signal_evidence_links(prediction_id);
|
||||
CREATE INDEX IF NOT EXISTS idx_sig_ev_document ON signal_evidence_links(document_id);
|
||||
CREATE INDEX IF NOT EXISTS idx_sig_ev_ticker ON signal_evidence_links(ticker);
|
||||
```
|
||||
|
||||
#### model_metric_snapshots
|
||||
|
||||
```sql
|
||||
CREATE TABLE IF NOT EXISTS model_metric_snapshots (
|
||||
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||
generated_at TIMESTAMPTZ NOT NULL,
|
||||
lookback_window VARCHAR(20) NOT NULL,
|
||||
horizon VARCHAR(20) NOT NULL,
|
||||
prediction_count INTEGER NOT NULL DEFAULT 0,
|
||||
win_rate FLOAT,
|
||||
directional_accuracy FLOAT,
|
||||
information_coefficient FLOAT,
|
||||
rank_information_coefficient FLOAT,
|
||||
avg_return FLOAT,
|
||||
avg_excess_return_vs_spy FLOAT,
|
||||
avg_excess_return_vs_sector FLOAT,
|
||||
calibration_error FLOAT,
|
||||
brier_score FLOAT,
|
||||
buy_win_rate FLOAT,
|
||||
sell_win_rate FLOAT,
|
||||
hold_win_rate FLOAT,
|
||||
metadata JSONB DEFAULT '{}',
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
|
||||
);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS idx_model_snap_generated ON model_metric_snapshots(generated_at);
|
||||
CREATE INDEX IF NOT EXISTS idx_model_snap_lookback ON model_metric_snapshots(lookback_window);
|
||||
CREATE INDEX IF NOT EXISTS idx_model_snap_horizon ON model_metric_snapshots(horizon);
|
||||
```
|
||||
|
||||
#### SQL Explorer Views
|
||||
|
||||
```sql
|
||||
CREATE OR REPLACE VIEW v_prediction_performance AS
|
||||
SELECT
|
||||
ps.ticker,
|
||||
ps.direction,
|
||||
ps.action,
|
||||
ps.confidence,
|
||||
ps.strength,
|
||||
ps.contradiction,
|
||||
ps.p_bull,
|
||||
ps.score_company,
|
||||
ps.score_macro,
|
||||
ps.score_competitive,
|
||||
ps.evidence_count,
|
||||
ps.unique_source_count,
|
||||
ps.duplicate_evidence_count,
|
||||
ps.price_at_prediction,
|
||||
po.future_return,
|
||||
po.excess_return_vs_spy,
|
||||
po.excess_return_vs_sector,
|
||||
po.direction_correct,
|
||||
po.profitable,
|
||||
po.horizon,
|
||||
ps.generated_at,
|
||||
po.evaluated_at
|
||||
FROM prediction_snapshots ps
|
||||
JOIN prediction_outcomes po ON po.prediction_id = ps.id;
|
||||
|
||||
CREATE OR REPLACE VIEW v_source_performance AS
|
||||
SELECT
|
||||
sel.source,
|
||||
sel.source_type,
|
||||
sel.catalyst_type,
|
||||
sel.sentiment,
|
||||
sel.weight,
|
||||
sel.contribution_score,
|
||||
sel.is_duplicate,
|
||||
po.direction_correct,
|
||||
po.future_return,
|
||||
po.excess_return_vs_spy,
|
||||
po.horizon,
|
||||
ps.generated_at
|
||||
FROM signal_evidence_links sel
|
||||
JOIN prediction_snapshots ps ON ps.id = sel.prediction_id
|
||||
JOIN prediction_outcomes po ON po.prediction_id = sel.prediction_id;
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Correctness Properties
|
||||
|
||||
*A property is a characteristic or behavior that should hold true across all valid executions of a system — essentially, a formal statement about what the system should do. Properties serve as the bridge between human-readable specifications and machine-verifiable correctness guarantees.*
|
||||
|
||||
The following properties were derived from the acceptance criteria through systematic prework analysis. Each property is universally quantified and maps to specific requirements. After reflection, 7 unique properties remain — one for each PBT requirement in Requirement 17. Redundant properties from Requirements 2, 5, 6, 8, and 11 were consolidated with their corresponding Requirement 17 counterparts.
|
||||
|
||||
### Property 1: Calibration Error Range and Round-Trip
|
||||
|
||||
*For any* valid distribution of predictions across confidence buckets (where each prediction has a confidence in [0.5, 1.0] and a boolean outcome), the Expected Calibration Error (ECE) SHALL be in [0.0, 1.0]. Furthermore, when every bucket's observed win rate exactly matches its average confidence, ECE SHALL be 0.0.
|
||||
|
||||
**Validates: Requirements 5.1, 5.3, 17.1**
|
||||
|
||||
### Property 2: Brier Score Range and Perfect Prediction
|
||||
|
||||
*For any* list of (p_bull, outcome) pairs where p_bull ∈ [0.0, 1.0] and outcome ∈ {0.0, 1.0}, the Brier score SHALL be in [0.0, 1.0]. Furthermore, when all predictions have p_bull = 1.0 and outcome = 1.0 (or p_bull = 0.0 and outcome = 0.0), the Brier score SHALL be 0.0.
|
||||
|
||||
**Validates: Requirements 5.4, 17.2**
|
||||
|
||||
### Property 3: Information Coefficient Range and Perfect Correlation
|
||||
|
||||
*For any* list of (score, return) pairs with at least 30 elements where scores and returns are finite floats, the Information Coefficient (Pearson correlation) SHALL be in [-1.0, 1.0]. Furthermore, when scores and returns are perfectly positively linearly correlated (returns = a * scores + b, a > 0), IC SHALL be 1.0 (within floating-point tolerance).
|
||||
|
||||
**Validates: Requirements 6.1, 6.2, 17.3**
|
||||
|
||||
### Property 4: Canonical Evidence Key Determinism and Normalization Idempotence
|
||||
|
||||
*For any* (title, url) string pair, computing the canonical evidence key SHALL be deterministic — the same inputs always produce the same key. Furthermore, normalizing an already-normalized input (lowercased, trimmed title; lowercased, query-stripped URL) and computing the key SHALL produce the same key as the original computation (idempotence).
|
||||
|
||||
**Validates: Requirements 2.3, 17.4**
|
||||
|
||||
### Property 5: Source Reliability Bayesian Shrinkage Bounds and Convergence
|
||||
|
||||
*For any* observed_win_rate ∈ [0.0, 1.0] and sample_count ≥ 0, the source reliability computed via Bayesian shrinkage SHALL be in [0.0, 1.0]. When sample_count = 0, reliability SHALL be exactly 0.5. As sample_count increases toward infinity, reliability SHALL approach the observed_win_rate monotonically.
|
||||
|
||||
**Validates: Requirements 8.1, 8.2, 17.5**
|
||||
|
||||
### Property 6: Quality Gate Determinism and Threshold Monotonicity
|
||||
|
||||
*For any* set of model metric values and quality gate configuration, the gate evaluation result SHALL be deterministic — the same inputs always produce the same pass/fail result. Furthermore, for any configuration where the gate passes, relaxing any single threshold (increasing min values or decreasing max values to make them easier to satisfy) SHALL NOT cause the gate to fail (monotonicity).
|
||||
|
||||
**Validates: Requirements 11.1, 17.6**
|
||||
|
||||
### Property 7: Contribution Score Sum-to-One and Range
|
||||
|
||||
*For any* non-empty list of positive document weights, the computed contribution scores SHALL each be in [0.0, 1.0] and SHALL sum to 1.0 (within floating-point tolerance of 1e-9). For an empty weight list, the result SHALL be an empty list.
|
||||
|
||||
**Validates: Requirements 2.5, 17.7**
|
||||
|
||||
---
|
||||
|
||||
## Error Handling
|
||||
|
||||
### Price Data Unavailability
|
||||
|
||||
| Scenario | Handling |
|
||||
|----------|----------|
|
||||
| Ticker price unavailable at snapshot time | Store NULL for `price_at_prediction`, log warning, continue |
|
||||
| SPY price unavailable at snapshot time | Store NULL for `spy_price_at_prediction`, log warning, continue |
|
||||
| Sector ETF price unavailable at snapshot time | Store NULL for `sector_etf_price_at_prediction`, log warning, continue |
|
||||
| Sector not found in SECTOR_ETF_MAP | Store NULL for sector ETF price, log warning |
|
||||
| Future price unavailable at evaluation time | Skip that horizon, retry on next Outcome_Evaluator run |
|
||||
| SPY/sector ETF future price unavailable | Store NULL for excess returns, still compute ticker return |
|
||||
|
||||
### Metrics Computation Edge Cases
|
||||
|
||||
| Scenario | Handling |
|
||||
|----------|----------|
|
||||
| Zero predictions in a confidence bucket | Exclude bucket from ECE computation |
|
||||
| Fewer than 30 predictions for IC/Rank IC | Return NULL instead of unreliable correlation |
|
||||
| All predictions in same confidence bucket | ECE = |avg_confidence - win_rate| for that single bucket |
|
||||
| Division by zero in contribution scores (total weight = 0) | Return equal contribution scores (1/n) |
|
||||
| Single prediction | Contribution score = 1.0 |
|
||||
| NaN/infinity in metric computation | Guard with `math.isnan`/`math.isinf` checks, return 0.0 or NULL |
|
||||
|
||||
### Quality Gate Failures
|
||||
|
||||
| Scenario | Handling |
|
||||
|----------|----------|
|
||||
| No model_metric_snapshots exist | Default to paper-only mode (fail-safe) |
|
||||
| Most recent snapshot older than 24 hours | Default to paper-only mode (fail-safe) |
|
||||
| risk_configs table unreachable | Default to paper-only mode, log warning |
|
||||
| Invalid threshold values in risk_configs | Use default thresholds, log warning |
|
||||
| Gate evaluation fails mid-computation | Default to paper-only mode, log error |
|
||||
|
||||
### Database Failures
|
||||
|
||||
| Scenario | Handling |
|
||||
|----------|----------|
|
||||
| prediction_snapshots insert fails | Log error, do not block recommendation generation |
|
||||
| signal_evidence_links insert fails | Log error, snapshot still created (partial data) |
|
||||
| prediction_outcomes insert fails | Log error, retry on next Outcome_Evaluator run |
|
||||
| model_metric_snapshots insert fails | Log error, stale metrics used until next successful computation |
|
||||
| source_accuracy update fails | Log error, continue with stale reliability data |
|
||||
|
||||
### Canonical Evidence Key Edge Cases
|
||||
|
||||
| Scenario | Handling |
|
||||
|----------|----------|
|
||||
| Empty title | Use empty string in hash computation |
|
||||
| Empty URL | Use empty string in hash computation |
|
||||
| URL with no query parameters | Use URL as-is after lowercasing |
|
||||
| Non-ASCII characters in title/URL | Encode as UTF-8 before hashing |
|
||||
|
||||
---
|
||||
|
||||
## Testing Strategy
|
||||
|
||||
### Dual Testing Approach
|
||||
|
||||
The model validation feature requires both property-based tests (for mathematical correctness of metric computations) and example-based unit tests (for specific behaviors, integration points, and edge cases). Property-based testing is appropriate here because the feature contains several pure mathematical functions (ECE, Brier score, IC, Bayesian shrinkage, contribution scores) with clear input/output behavior and universal properties.
|
||||
|
||||
### Property-Based Testing
|
||||
|
||||
**Library:** Hypothesis (already in use — `.hypothesis/` directory exists, project convention established)
|
||||
|
||||
**Configuration:**
|
||||
- Minimum 100 iterations per property: `@settings(max_examples=100)`
|
||||
- File naming: `tests/test_pbt_model_validation.py`
|
||||
- Tag format: `# Feature: model-validation-calibration, Property N: <title>`
|
||||
|
||||
**Property tests to implement (one test per correctness property):**
|
||||
|
||||
| Property | Test Function | Key Generators |
|
||||
|----------|---------------|----------------|
|
||||
| 1: ECE range and round-trip | `test_calibration_error_range_and_roundtrip` | `st.lists(st.tuples(st.floats(0.5, 1.0), st.booleans()))` |
|
||||
| 2: Brier score range and perfect | `test_brier_score_range_and_perfect` | `st.lists(st.tuples(st.floats(0.0, 1.0), st.sampled_from([0.0, 1.0])))` |
|
||||
| 3: IC range and perfect correlation | `test_information_coefficient_range_and_perfect` | `st.lists(st.floats(-10, 10), min_size=30)` with linear transform |
|
||||
| 4: Canonical key determinism and idempotence | `test_canonical_key_determinism_and_idempotence` | `st.text()` pairs for title and URL |
|
||||
| 5: Source reliability bounds and convergence | `test_source_reliability_bounds_and_convergence` | `st.floats(0.0, 1.0)` for win_rate, `st.integers(0, 10000)` for n |
|
||||
| 6: Quality gate determinism and monotonicity | `test_quality_gate_determinism_and_monotonicity` | Custom strategy for `QualityGateConfig` and metric values |
|
||||
| 7: Contribution score sum-to-one | `test_contribution_score_sum_to_one` | `st.lists(st.floats(0.01, 100.0), min_size=1)` |
|
||||
|
||||
### Example-Based Unit Tests
|
||||
|
||||
**File:** `tests/test_model_validation_unit.py`
|
||||
|
||||
| Test Area | Examples |
|
||||
|-----------|----------|
|
||||
| Canonical evidence key | Known title/URL → expected SHA256, empty inputs, unicode |
|
||||
| Duplicate detection | 3 docs with 2 sharing a key → 1 marked duplicate |
|
||||
| Contribution scores | [0.5, 0.3, 0.2] → [0.5, 0.3, 0.2], single doc → [1.0] |
|
||||
| ECE specific values | Perfect calibration → 0.0, all overconfident → positive ECE |
|
||||
| Brier score specific values | All correct at p=1.0 → 0.0, all wrong at p=1.0 → 1.0 |
|
||||
| IC specific values | Perfect correlation → 1.0, anti-correlation → -1.0, < 30 → None |
|
||||
| Source reliability | n=0 → 0.5, n=1000 with wr=0.8 → ≈0.8, n=30 with wr=0.7 → 0.6 |
|
||||
| Adjusted evidence weight | reliability=0.5 → base*1.0, clamping to [0.1, 2.0] |
|
||||
| Quality gate | All thresholds met → pass, one failed → fail with reason |
|
||||
| Quality gate fail-safe | No snapshots → paper-only, stale snapshot → paper-only |
|
||||
| Direction correct logic | bullish+positive → true, bullish+negative → false |
|
||||
| Profitable logic | buy+positive → true, sell+negative → true |
|
||||
| Future return computation | price 100→110 → 0.10, price 100→90 → -0.10 |
|
||||
| Excess return | ticker 10%, SPY 5% → excess 5% |
|
||||
| Weight clamping | weight 1.5 → clamped to 1.0 |
|
||||
|
||||
### Frontend Tests
|
||||
|
||||
**File:** `frontend/src/test/pages.test.tsx` (extend existing)
|
||||
|
||||
| Test Area | Strategy |
|
||||
|-----------|----------|
|
||||
| OpsModel page renders validation tabs | MSW mock for `/api/validation/summary` |
|
||||
| Calibration table renders buckets | MSW mock for `/api/validation/calibration` |
|
||||
| Gate status indicator | MSW mock for `/api/validation/gate-status` |
|
||||
| Miscalibration warning badge | Mock data with miscalibrated bucket |
|
||||
|
||||
### Integration Tests
|
||||
|
||||
**File:** `tests/test_model_validation_integration.py`
|
||||
|
||||
| Test Area | Strategy |
|
||||
|-----------|----------|
|
||||
| Snapshot creation with mock DB | asyncpg mock, verify INSERT queries |
|
||||
| Outcome evaluation with mock prices | asyncpg mock, verify return computation |
|
||||
| Metrics computation end-to-end | In-memory data, verify all metrics computed |
|
||||
| API endpoint responses | FastAPI TestClient with mock pool |
|
||||
|
||||
### Test File Structure
|
||||
|
||||
```
|
||||
tests/
|
||||
├── test_pbt_model_validation.py # 7 property-based tests
|
||||
├── test_model_validation_unit.py # Example-based unit tests
|
||||
└── test_model_validation_integration.py # Integration tests (optional)
|
||||
|
||||
frontend/src/test/
|
||||
└── pages.test.tsx # Extended with validation page tests
|
||||
```
|
||||
@@ -0,0 +1,286 @@
|
||||
# Requirements Document — Model Validation, Calibration, and Signal Quality
|
||||
|
||||
## Introduction
|
||||
|
||||
The Stonks Oracle platform generates trend summaries and trading recommendations from a three-layer signal aggregation engine. While the pipeline produces directional predictions with confidence scores, there is no systematic mechanism to evaluate whether those predictions are accurate, whether confidence scores are well-calibrated, which sources and signal types contribute to correct predictions, or whether the system outperforms simple benchmarks. The platform also lacks safety gates that prevent live trading when model quality is insufficient.
|
||||
|
||||
This feature adds a complete model validation layer: prediction outcome tracking, calibration analysis, information coefficient metrics, signal and source attribution, evidence deduplication quality tracking, confidence recalibration, benchmark comparison, an upgraded Model Performance dashboard, and safety gates for live trading eligibility. The goal is to transform Stonks Oracle from a signal dashboard with paper trading into a statistically validated prediction engine with closed-loop feedback.
|
||||
|
||||
## Glossary
|
||||
|
||||
- **Prediction_Snapshot_Writer**: A new service component in `services/validation/prediction_snapshot.py` that captures the full state of every recommendation and trend prediction at generation time, including prices, evidence links, and duplicate counts.
|
||||
- **Outcome_Evaluator**: A new service component in `services/validation/outcome_evaluator.py` that runs periodically to compute realized future returns and directional accuracy for matured prediction snapshots across multiple horizons.
|
||||
- **Metrics_Engine**: A new service component in `services/validation/metrics.py` that computes aggregate model quality metrics including calibration error, information coefficient, Brier score, and win rates over configurable lookback windows.
|
||||
- **Attribution_Engine**: A new service component in `services/validation/attribution.py` that computes per-source, per-catalyst-type, and per-signal-layer performance metrics by joining evidence links with prediction outcomes.
|
||||
- **Calibration_Engine**: A new service component in `services/validation/calibration.py` that computes source reliability scores using Bayesian shrinkage and adjusts evidence weights based on historical source performance.
|
||||
- **Quality_Gate**: A new service component in `services/trading/model_quality_gate.py` that evaluates aggregate model metrics against configurable thresholds and determines whether the system meets minimum quality standards for live trading.
|
||||
- **Information_Coefficient**: The Pearson correlation between predicted scores and realized future returns, measuring the linear predictive power of the model. Abbreviated as IC.
|
||||
- **Rank_Information_Coefficient**: The Spearman rank correlation between predicted scores and realized future returns, measuring ordinal predictive power. Abbreviated as Rank IC.
|
||||
- **Calibration_Error**: The Expected Calibration Error (ECE), computed as the weighted average of the absolute difference between predicted confidence and observed win rate across confidence buckets.
|
||||
- **Brier_Score**: The mean squared error between the predicted bullish probability and the binary actual outcome (1 if price went up, 0 otherwise), measuring probabilistic forecast accuracy.
|
||||
- **Canonical_Evidence_Key**: A normalized identifier for a piece of evidence, computed as SHA256 of the normalized title concatenated with the normalized URL, used to detect duplicate evidence across different ingestion paths.
|
||||
- **Excess_Return**: The return of a prediction minus the return of a benchmark (SPY for broad market, sector ETF for sector-relative) over the same horizon, measuring alpha generation.
|
||||
- **Prediction_Snapshot**: A frozen record of a prediction at generation time, capturing all inputs (prices, scores, evidence) needed to evaluate the prediction against future outcomes without hindsight bias.
|
||||
- **Model_Metric_Snapshot**: A periodic aggregate of model quality metrics over a lookback window and horizon, stored for time-series analysis of model performance trends.
|
||||
- **Source_Reliability**: A Bayesian-shrunk estimate of a source's historical win rate, computed as `0.5 + (n/(n+30)) * (observed_win_rate - 0.5)`, which regresses toward 0.5 for sources with few observations.
|
||||
- **Dashboard_API**: The set of API endpoints under `/api/validation/` that serve model quality metrics, calibration tables, attribution data, and gate status to the frontend.
|
||||
|
||||
---
|
||||
|
||||
## Requirements
|
||||
|
||||
### Requirement 1: Prediction Snapshot Capture
|
||||
|
||||
**User Story:** As a quantitative analyst, I want every recommendation and trend prediction captured as an immutable snapshot at generation time, so that I can evaluate predictions against future outcomes without hindsight bias.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN a recommendation is generated by the Recommendation_Engine, THE Prediction_Snapshot_Writer SHALL create a prediction_snapshots record containing the ticker, generation timestamp, trend window, prediction horizon, direction, action, mode, strength, confidence, contradiction score, bullish probability, bearish probability, company score, macro score, competitive score, evidence count, unique source count, duplicate evidence count, price at prediction time, SPY price at prediction time, and sector ETF price at prediction time.
|
||||
2. WHEN a prediction snapshot is created, THE Prediction_Snapshot_Writer SHALL record the current market price for the predicted ticker by querying the most recent close price from the market_snapshots table.
|
||||
3. WHEN a prediction snapshot is created, THE Prediction_Snapshot_Writer SHALL record the current SPY price by querying the most recent close price for ticker SPY from the market_snapshots table.
|
||||
4. WHEN a prediction snapshot is created, THE Prediction_Snapshot_Writer SHALL record the current sector ETF price by looking up the sector for the predicted ticker and querying the most recent close price for the corresponding sector ETF from the market_snapshots table.
|
||||
5. IF the market price, SPY price, or sector ETF price is unavailable at snapshot time, THEN THE Prediction_Snapshot_Writer SHALL store NULL for the unavailable price fields and log a warning, rather than failing the snapshot creation.
|
||||
6. THE Prediction_Snapshot_Writer SHALL store prediction snapshots in a new `prediction_snapshots` database table with a UUID primary key and indexed columns for ticker, generated_at, and horizon.
|
||||
7. WHEN a prediction snapshot is created, THE Prediction_Snapshot_Writer SHALL store a JSONB metadata field containing any additional context from the trend summary market_context and recommendation risk_checks fields.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 2: Signal Evidence Link Tracking
|
||||
|
||||
**User Story:** As a quantitative analyst, I want to know which specific evidence documents contributed to each prediction, so that I can attribute prediction success or failure to individual sources and signal types.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN a prediction snapshot is created, THE Prediction_Snapshot_Writer SHALL create signal_evidence_links records for each document that contributed to the prediction, linking the prediction_id to the document_id and signal_id.
|
||||
2. THE signal_evidence_links record SHALL capture the source identifier, source type, catalyst type, sentiment, impact score, extraction confidence, weight assigned during aggregation, duplicate status, canonical evidence key, and contribution score for each contributing document.
|
||||
3. WHEN recording evidence links, THE Prediction_Snapshot_Writer SHALL compute the canonical_evidence_key as the SHA256 hash of the concatenation of the normalized (lowercased, whitespace-trimmed) document title and the normalized (lowercased, query-parameters-stripped) document URL.
|
||||
4. WHEN recording evidence links, THE Prediction_Snapshot_Writer SHALL mark a link as `is_duplicate = true` when another link for the same prediction and ticker shares the same canonical_evidence_key.
|
||||
5. THE Prediction_Snapshot_Writer SHALL compute the contribution_score for each evidence link as the ratio of that document's effective weight to the total effective weight across all documents for the prediction.
|
||||
6. THE signal_evidence_links table SHALL have a foreign key constraint from prediction_id to prediction_snapshots(id) and indexes on prediction_id, document_id, and ticker.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 3: Evidence Deduplication Quality Tracking
|
||||
|
||||
**User Story:** As a quantitative analyst, I want the system to track evidence deduplication quality per prediction, so that I can identify when predictions are inflated by counting the same information multiple times from different sources.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN creating a prediction snapshot, THE Prediction_Snapshot_Writer SHALL compute the unique_source_count as the number of distinct source identifiers across all non-duplicate evidence links for that prediction.
|
||||
2. WHEN creating a prediction snapshot, THE Prediction_Snapshot_Writer SHALL compute the duplicate_evidence_count as the number of evidence links marked as `is_duplicate = true` for that prediction.
|
||||
3. THE Prediction_Snapshot_Writer SHALL enforce a maximum single-document weight cap of 1.0, clamping any individual document's effective weight to prevent a single piece of evidence from dominating the prediction.
|
||||
4. WHEN computing contribution scores, THE Prediction_Snapshot_Writer SHALL count each canonical evidence key at most once per ticker per window, applying the one-vote-per-canonical-document deduplication rule.
|
||||
5. THE Metrics_Engine SHALL compute a duplicate_rate metric as the ratio of duplicate_evidence_count to total evidence_count across predictions in the lookback window.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 4: Prediction Outcome Evaluation
|
||||
|
||||
**User Story:** As a quantitative analyst, I want realized market outcomes automatically matched to historical predictions, so that I can measure whether the system's directional calls and confidence scores correspond to actual price movements.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Outcome_Evaluator SHALL run on a periodic schedule, evaluating prediction snapshots whose horizon has elapsed and whose outcome has not yet been recorded.
|
||||
2. WHEN evaluating a prediction snapshot, THE Outcome_Evaluator SHALL compute the future_return as `(future_price - price_at_prediction) / price_at_prediction` using the closing price at the horizon endpoint.
|
||||
3. WHEN evaluating a prediction snapshot, THE Outcome_Evaluator SHALL compute the SPY return over the same horizon as `(spy_future_price - spy_price_at_prediction) / spy_price_at_prediction`.
|
||||
4. WHEN evaluating a prediction snapshot, THE Outcome_Evaluator SHALL compute the sector ETF return over the same horizon as `(sector_etf_future_price - sector_etf_price_at_prediction) / sector_etf_price_at_prediction`.
|
||||
5. WHEN evaluating a prediction snapshot, THE Outcome_Evaluator SHALL compute excess_return_vs_spy as `future_return - spy_return` and excess_return_vs_sector as `future_return - sector_etf_return`.
|
||||
6. WHEN evaluating a prediction snapshot, THE Outcome_Evaluator SHALL determine direction_correct as true when the prediction direction is bullish and future_return is positive, or when the prediction direction is bearish and future_return is negative.
|
||||
7. WHEN evaluating a prediction snapshot, THE Outcome_Evaluator SHALL determine profitable as true when the prediction action is buy and future_return is positive, or when the prediction action is sell and future_return is negative.
|
||||
8. THE Outcome_Evaluator SHALL evaluate each prediction across all applicable horizons: 1 hour, 6 hours, 1 day, 7 days, and 30 days.
|
||||
9. THE Outcome_Evaluator SHALL store evaluation results in a new `prediction_outcomes` table with a foreign key to prediction_snapshots and indexed columns for prediction_id, horizon, and evaluated_at.
|
||||
10. IF the future price is unavailable at the horizon endpoint (market data gap), THEN THE Outcome_Evaluator SHALL skip that horizon evaluation and retry on the next run.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 5: Calibration Analysis
|
||||
|
||||
**User Story:** As a quantitative analyst, I want to measure how well the system's confidence scores predict actual win rates, so that I can identify overconfident or underconfident predictions and recalibrate the model.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Metrics_Engine SHALL compute calibration metrics by grouping evaluated predictions into confidence buckets: [0.50, 0.60), [0.60, 0.70), [0.70, 0.80), [0.80, 0.90), [0.90, 1.00].
|
||||
2. FOR EACH confidence bucket, THE Metrics_Engine SHALL compute the average confidence, the observed win rate (fraction of direction_correct outcomes), and the prediction count.
|
||||
3. THE Metrics_Engine SHALL compute the Expected Calibration Error (ECE) as the weighted average of `|avg_confidence - observed_win_rate|` across all buckets, weighted by the fraction of predictions in each bucket.
|
||||
4. THE Metrics_Engine SHALL compute the Brier Score as `mean((p_bull - actual_outcome)^2)` across all evaluated predictions, where actual_outcome is 1.0 when the price moved in the predicted direction and 0.0 otherwise.
|
||||
5. THE Metrics_Engine SHALL flag calibration buckets where `|avg_confidence - observed_win_rate| > 0.15` as miscalibrated for dashboard highlighting.
|
||||
6. THE Metrics_Engine SHALL compute calibration metrics separately for each prediction horizon (1h, 6h, 1d, 7d, 30d).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 6: Information Coefficient Metrics
|
||||
|
||||
**User Story:** As a quantitative analyst, I want to measure the correlation between the system's prediction scores and realized returns, so that I can assess whether higher-scored predictions actually produce higher returns.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Metrics_Engine SHALL compute the Information Coefficient (IC) as the Pearson correlation between prediction scores and future returns across all evaluated predictions in the lookback window.
|
||||
2. THE Metrics_Engine SHALL compute the Rank Information Coefficient (Rank IC) as the Spearman rank correlation between prediction scores and future returns across all evaluated predictions in the lookback window.
|
||||
3. THE Metrics_Engine SHALL compute IC and Rank IC separately for each prediction horizon (1h, 6h, 1d, 7d, 30d).
|
||||
4. THE Metrics_Engine SHALL compute return statistics by confidence decile, grouping predictions into 10 equal-sized bins by confidence and computing the average future return and average excess return for each decile.
|
||||
5. WHEN fewer than 30 evaluated predictions exist for a given horizon, THE Metrics_Engine SHALL report IC and Rank IC as NULL rather than computing unreliable correlations from small samples.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 7: Source and Signal Attribution
|
||||
|
||||
**User Story:** As a quantitative analyst, I want to know which sources, source types, and catalyst types contribute to accurate predictions, so that I can identify the most valuable information channels and deprioritize unreliable ones.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Attribution_Engine SHALL compute per-source performance metrics by joining signal_evidence_links with prediction_outcomes, grouping by source identifier.
|
||||
2. FOR EACH source, THE Attribution_Engine SHALL compute: prediction count, average weight, average contribution score, win rate, average future return, average excess return vs SPY, and information coefficient.
|
||||
3. THE Attribution_Engine SHALL compute the same performance metrics grouped by source_type (e.g., news_api, filings_api, web_scrape, market_api).
|
||||
4. THE Attribution_Engine SHALL compute the same performance metrics grouped by catalyst_type (e.g., earnings, product, legal, macro, m_and_a).
|
||||
5. THE Attribution_Engine SHALL compute layer attribution metrics for the three signal layers (company, macro, competitive) by using the score_company, score_macro, and score_competitive fields from prediction snapshots.
|
||||
6. FOR EACH layer, THE Attribution_Engine SHALL compute the average contribution percentage, the win rate when that layer is the dominant contributor, and the IC of predictions where that layer contributes more than 30% of the total score.
|
||||
7. THE Attribution_Engine SHALL compute a per-source duplicate_rate as the fraction of evidence links from that source marked as is_duplicate.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 8: Confidence Recalibration via Source Reliability
|
||||
|
||||
**User Story:** As a quantitative analyst, I want source credibility weights adjusted based on historical prediction accuracy using Bayesian shrinkage, so that the system learns from its own track record and improves over time.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Calibration_Engine SHALL compute source reliability using Bayesian shrinkage: `reliability = 0.5 + (n / (n + 30)) * (observed_win_rate - 0.5)`, where n is the number of evaluated predictions involving that source and observed_win_rate is the fraction of correct directional calls.
|
||||
2. WHEN a source has zero evaluated predictions, THE Calibration_Engine SHALL assign a reliability of 0.5 (the prior mean).
|
||||
3. THE Calibration_Engine SHALL compute an adjusted evidence weight for each source as `adjusted_weight = base_weight * (0.5 + reliability)`, clamped to the range [0.1, 2.0].
|
||||
4. THE Calibration_Engine SHALL update source reliability scores after each outcome evaluation cycle, using the latest prediction outcomes.
|
||||
5. THE Calibration_Engine SHALL store source reliability scores in the existing `source_accuracy` table, extending it with a reliability column or using the existing accuracy_ratio field with the Bayesian shrinkage formula.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 9: Benchmark Comparison
|
||||
|
||||
**User Story:** As a quantitative analyst, I want the system's prediction performance compared against simple benchmarks, so that I can determine whether the model adds value beyond naive strategies.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Metrics_Engine SHALL compute the average excess return of all buy predictions versus a buy-and-hold SPY strategy over the same horizons.
|
||||
2. THE Metrics_Engine SHALL compute the average excess return of all buy predictions versus a buy-and-hold sector ETF strategy over the same horizons.
|
||||
3. THE Metrics_Engine SHALL compute the win rate of the system's directional predictions compared to a random 50/50 baseline, reporting the statistical significance using a binomial test when the prediction count exceeds 100.
|
||||
4. THE Metrics_Engine SHALL compute the hit rate improvement, defined as `(system_win_rate - 0.5) / 0.5`, representing the percentage improvement over random guessing.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 10: Model Metric Snapshots
|
||||
|
||||
**User Story:** As a quantitative analyst, I want aggregate model metrics stored as time-series snapshots, so that I can track whether model quality is improving or degrading over time.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Metrics_Engine SHALL periodically compute and store model_metric_snapshots containing all aggregate metrics for each combination of lookback window and prediction horizon.
|
||||
2. EACH model_metric_snapshot SHALL contain: prediction count, win rate, directional accuracy, IC, Rank IC, average return, average excess return vs SPY, average excess return vs sector, calibration error (ECE), Brier score, and per-action win rates (buy, sell, hold).
|
||||
3. THE Metrics_Engine SHALL store model_metric_snapshots in a new `model_metric_snapshots` database table with a UUID primary key and indexed columns for generated_at, lookback_window, and horizon.
|
||||
4. THE Metrics_Engine SHALL compute snapshots for lookback windows of 7 days, 30 days, 90 days, and all-time.
|
||||
5. THE Metrics_Engine SHALL store a JSONB metadata field in each snapshot for extensibility, containing any additional computed metrics not captured in dedicated columns.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 11: Safety Gate for Live Trading
|
||||
|
||||
**User Story:** As a platform operator, I want live trading automatically disabled when model quality metrics fall below minimum thresholds, so that the system does not risk real capital on a poorly performing model.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Quality_Gate SHALL evaluate the following minimum thresholds for live trading eligibility: minimum prediction count of 100, minimum IC of 0.03, minimum win rate of 0.53, maximum ECE of 0.15, and minimum excess return vs SPY of 0.0.
|
||||
2. WHEN any threshold is not met, THE Quality_Gate SHALL force all recommendations to paper mode, overriding any live_eligible mode assignments.
|
||||
3. THE Quality_Gate SHALL evaluate gate status at the start of each aggregation cycle by reading the most recent model_metric_snapshot.
|
||||
4. THE Quality_Gate SHALL log the gate evaluation result including which thresholds passed and which failed, with their actual values.
|
||||
5. THE Quality_Gate SHALL store the gate evaluation result in the `risk_configs` table under a `model_quality_gate` key, making it available to the recommendation engine and dashboard.
|
||||
6. IF the model_metric_snapshots table is empty or the most recent snapshot is older than 24 hours, THEN THE Quality_Gate SHALL default to paper-only mode (fail-safe behavior).
|
||||
7. THE Quality_Gate SHALL support configurable thresholds via the `risk_configs` table, with the default values specified in acceptance criterion 1 used when no override is configured.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 12: Model Performance Dashboard Upgrade
|
||||
|
||||
**User Story:** As a platform operator, I want a comprehensive model performance dashboard showing prediction accuracy, calibration, attribution, and gate status, so that I can monitor model quality and make informed decisions about live trading.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Dashboard_API SHALL expose a `/api/validation/summary` endpoint returning the latest model metric snapshot with summary cards for: prediction count, win rate, directional accuracy, IC, Rank IC, Brier score, calibration error, average excess return vs SPY, average excess return vs sector, and live trading gate status.
|
||||
2. THE Dashboard_API SHALL expose a `/api/validation/calibration` endpoint returning the calibration table with confidence buckets, average confidence, observed win rate, prediction count, and miscalibration flag for each bucket.
|
||||
3. THE Dashboard_API SHALL expose a `/api/validation/ic-by-horizon` endpoint returning IC and Rank IC values for each prediction horizon.
|
||||
4. THE Dashboard_API SHALL expose a `/api/validation/attribution/sources` endpoint returning per-source performance metrics including win rate, IC, average return, and duplicate rate.
|
||||
5. THE Dashboard_API SHALL expose a `/api/validation/attribution/catalysts` endpoint returning per-catalyst-type performance metrics.
|
||||
6. THE Dashboard_API SHALL expose a `/api/validation/attribution/layers` endpoint returning per-signal-layer (company, macro, competitive) performance metrics.
|
||||
7. THE Dashboard_API SHALL expose a `/api/validation/gate-status` endpoint returning the current quality gate evaluation with pass/fail status for each threshold.
|
||||
8. THE frontend OpsModel page SHALL be upgraded to display the model validation summary cards, calibration table, IC-by-horizon table, source performance table, catalyst truth table, layer attribution table, and gate status indicator.
|
||||
9. THE frontend SHALL highlight miscalibrated confidence buckets where `|avg_confidence - observed_win_rate| > 0.15` with a visual warning indicator.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 13: Recommendation Display Enhancements
|
||||
|
||||
**User Story:** As a platform operator, I want each recommendation to display its validation context including calibrated confidence, historical win rate, and evidence quality indicators, so that I can assess the reliability of individual predictions.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN displaying a recommendation, THE frontend SHALL show the original confidence alongside the calibrated confidence (based on the historical win rate for that confidence bucket).
|
||||
2. WHEN displaying a recommendation, THE frontend SHALL show the historical win rate for predictions with similar confidence levels.
|
||||
3. WHEN displaying a recommendation, THE frontend SHALL show the evidence count, unique evidence count, and duplicate evidence count.
|
||||
4. WHEN displaying a recommendation, THE frontend SHALL show a source reliability indicator based on the Bayesian-shrunk reliability score of the primary contributing sources.
|
||||
5. WHEN displaying a recommendation, THE frontend SHALL show the live eligibility status with the reason (gate passed, or which threshold failed).
|
||||
6. WHEN the duplicate evidence count exceeds 20% of the total evidence count, THE frontend SHALL display a warning badge indicating potential evidence inflation.
|
||||
7. WHEN the primary contributing source has a reliability score below 0.4, THE frontend SHALL display a warning badge indicating unknown or low source reliability.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 14: SQL Explorer Views
|
||||
|
||||
**User Story:** As a quantitative analyst, I want pre-built SQL views joining predictions with outcomes and evidence with performance, so that I can run ad-hoc analysis in the SQL Explorer without writing complex joins.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE database migration SHALL create a view `v_prediction_performance` that joins prediction_snapshots with prediction_outcomes on prediction_id, providing a single flat table with prediction inputs and realized outcomes.
|
||||
2. THE database migration SHALL create a view `v_source_performance` that joins signal_evidence_links with prediction_outcomes (via prediction_id), providing per-evidence-link outcome data for source attribution analysis.
|
||||
3. THE v_prediction_performance view SHALL include columns for ticker, direction, action, confidence, strength, price_at_prediction, future_return, excess_return_vs_spy, direction_correct, profitable, horizon, generated_at, and evaluated_at.
|
||||
4. THE v_source_performance view SHALL include columns for source, source_type, catalyst_type, sentiment, weight, contribution_score, is_duplicate, direction_correct, future_return, and excess_return_vs_spy.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 15: Backtest Replay Integration
|
||||
|
||||
**User Story:** As a quantitative analyst, I want to replay historical data through the prediction snapshot and outcome evaluation pipeline, so that I can assess model quality on historical data without future data leakage.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Backtest_Replay service SHALL support a validation mode that generates prediction snapshots and evaluates outcomes using only data available at each historical point in time.
|
||||
2. WHEN running in validation mode, THE Backtest_Replay service SHALL process historical recommendations chronologically, creating prediction snapshots with the market prices that were available at each recommendation's generation time.
|
||||
3. WHEN running in validation mode, THE Backtest_Replay service SHALL evaluate prediction outcomes using market prices from the appropriate future horizon relative to each prediction's generation time.
|
||||
4. THE Backtest_Replay service SHALL prevent future data leakage by ensuring that no market data with a timestamp after the prediction generation time is used during snapshot creation.
|
||||
5. WHEN a backtest validation run completes, THE Backtest_Replay service SHALL trigger a model metrics computation over the backtest period, storing the results as model_metric_snapshots tagged with the backtest_id.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 16: Database Schema
|
||||
|
||||
**User Story:** As a developer, I want the new database tables created via a migration script following the existing migration conventions, so that the schema changes are applied consistently across all environments.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE database migration SHALL create the `prediction_snapshots` table with columns: id (UUID PK), generated_at (TIMESTAMPTZ), ticker (VARCHAR), window (VARCHAR), horizon (VARCHAR), direction (VARCHAR), action (VARCHAR), mode (VARCHAR), strength (FLOAT), confidence (FLOAT), contradiction (FLOAT), p_bull (FLOAT), p_bear (FLOAT), score_company (FLOAT), score_macro (FLOAT), score_competitive (FLOAT), evidence_count (INTEGER), unique_source_count (INTEGER), duplicate_evidence_count (INTEGER), price_at_prediction (FLOAT), spy_price_at_prediction (FLOAT), sector_etf_price_at_prediction (FLOAT), metadata (JSONB), created_at (TIMESTAMPTZ).
|
||||
2. THE database migration SHALL create the `prediction_outcomes` table with columns: id (UUID PK), prediction_id (UUID FK to prediction_snapshots), evaluated_at (TIMESTAMPTZ), horizon (VARCHAR), future_price (FLOAT), future_return (FLOAT), spy_future_price (FLOAT), spy_return (FLOAT), sector_etf_future_price (FLOAT), sector_etf_return (FLOAT), excess_return_vs_spy (FLOAT), excess_return_vs_sector (FLOAT), direction_correct (BOOLEAN), profitable (BOOLEAN), metadata (JSONB), created_at (TIMESTAMPTZ).
|
||||
3. THE database migration SHALL create the `signal_evidence_links` table with columns: id (UUID PK), prediction_id (UUID FK to prediction_snapshots), document_id (VARCHAR), signal_id (VARCHAR), ticker (VARCHAR), source (VARCHAR), source_type (VARCHAR), catalyst_type (VARCHAR), sentiment (VARCHAR), impact (FLOAT), extraction_confidence (FLOAT), weight (FLOAT), is_duplicate (BOOLEAN), canonical_evidence_key (VARCHAR), contribution_score (FLOAT), metadata (JSONB), created_at (TIMESTAMPTZ).
|
||||
4. THE database migration SHALL create the `model_metric_snapshots` table with columns: id (UUID PK), generated_at (TIMESTAMPTZ), lookback_window (VARCHAR), horizon (VARCHAR), prediction_count (INTEGER), win_rate (FLOAT), directional_accuracy (FLOAT), information_coefficient (FLOAT), rank_information_coefficient (FLOAT), avg_return (FLOAT), avg_excess_return_vs_spy (FLOAT), avg_excess_return_vs_sector (FLOAT), calibration_error (FLOAT), brier_score (FLOAT), buy_win_rate (FLOAT), sell_win_rate (FLOAT), hold_win_rate (FLOAT), metadata (JSONB), created_at (TIMESTAMPTZ).
|
||||
5. THE database migration SHALL create appropriate indexes on prediction_snapshots (ticker, generated_at, horizon), prediction_outcomes (prediction_id, horizon), signal_evidence_links (prediction_id, document_id, ticker), and model_metric_snapshots (generated_at, lookback_window, horizon).
|
||||
6. THE database migration SHALL be numbered as `035_model_validation.sql`, following the existing migration numbering convention.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 17: Property-Based Testing for Validation Metrics
|
||||
|
||||
**User Story:** As a developer, I want property-based tests validating the mathematical correctness of all validation metric computations, so that edge cases and numerical stability issues are caught before deployment.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE test suite SHALL include a property-based test for calibration error verifying that ECE is in [0.0, 1.0] for all valid distributions of predictions across confidence buckets, and that ECE is 0.0 when every bucket's observed win rate exactly matches its average confidence (round-trip calibration property).
|
||||
2. THE test suite SHALL include a property-based test for Brier score verifying that the score is in [0.0, 1.0] for all valid probability-outcome pairs, and that the score is 0.0 when all predictions are perfectly correct with probability 1.0.
|
||||
3. THE test suite SHALL include a property-based test for information coefficient verifying that IC is in [-1.0, 1.0] for all valid score-return pairs, and that IC is 1.0 when scores and returns are perfectly positively correlated.
|
||||
4. THE test suite SHALL include a property-based test for the canonical evidence key verifying that the key is deterministic (same inputs always produce the same key) and that normalization is idempotent (normalizing an already-normalized input produces the same key).
|
||||
5. THE test suite SHALL include a property-based test for source reliability Bayesian shrinkage verifying that reliability is always in [0.0, 1.0], that reliability approaches 0.5 as sample count approaches 0, and that reliability approaches the observed win rate as sample count approaches infinity.
|
||||
6. THE test suite SHALL include a property-based test for the quality gate verifying that the gate result is deterministic for the same metric inputs, and that relaxing any single threshold (making it easier to pass) never causes a previously passing gate to fail (monotonicity property).
|
||||
7. THE test suite SHALL include a property-based test for contribution score computation verifying that all contribution scores for a single prediction sum to 1.0 (within floating-point tolerance) and that each individual score is in [0.0, 1.0].
|
||||
@@ -0,0 +1,260 @@
|
||||
# Implementation Plan: Model Validation, Calibration, and Signal Quality
|
||||
|
||||
## Overview
|
||||
|
||||
Add a closed-loop model validation layer to Stonks Oracle: prediction snapshot capture, outcome evaluation, calibration/IC metrics, source/catalyst/layer attribution, Bayesian source reliability, a quality gate for live trading, 7 new API endpoints, an upgraded OpsModel dashboard, and backtest replay integration. Implementation follows the four-phase priority order from the spec, with each phase building on the previous one.
|
||||
|
||||
## Tasks
|
||||
|
||||
- [x] 1. Database migration 035 — schema foundation
|
||||
- [x] 1.1 Create `infra/migrations/035_model_validation.sql` with all tables, indexes, and views
|
||||
- Create `prediction_snapshots` table with all columns from design (id UUID PK, generated_at, ticker, window, horizon, direction, action, mode, strength, confidence, contradiction, p_bull, p_bear, score_company, score_macro, score_competitive, evidence_count, unique_source_count, duplicate_evidence_count, price_at_prediction, spy_price_at_prediction, sector_etf_price_at_prediction, metadata JSONB, created_at)
|
||||
- Create `prediction_outcomes` table with FK to prediction_snapshots (id UUID PK, prediction_id, evaluated_at, horizon, future_price, future_return, spy_future_price, spy_return, sector_etf_future_price, sector_etf_return, excess_return_vs_spy, excess_return_vs_sector, direction_correct, profitable, metadata JSONB, created_at)
|
||||
- Create `signal_evidence_links` table with FK to prediction_snapshots (id UUID PK, prediction_id, document_id, signal_id, ticker, source, source_type, catalyst_type, sentiment, impact, extraction_confidence, weight, is_duplicate, canonical_evidence_key, contribution_score, metadata JSONB, created_at)
|
||||
- Create `model_metric_snapshots` table (id UUID PK, generated_at, lookback_window, horizon, prediction_count, win_rate, directional_accuracy, information_coefficient, rank_information_coefficient, avg_return, avg_excess_return_vs_spy, avg_excess_return_vs_sector, calibration_error, brier_score, buy_win_rate, sell_win_rate, hold_win_rate, metadata JSONB, created_at)
|
||||
- Create indexes on prediction_snapshots (ticker, generated_at, horizon), prediction_outcomes (prediction_id, horizon, evaluated_at), signal_evidence_links (prediction_id, document_id, ticker), model_metric_snapshots (generated_at, lookback_window, horizon)
|
||||
- Create `v_prediction_performance` view joining prediction_snapshots with prediction_outcomes
|
||||
- Create `v_source_performance` view joining signal_evidence_links with prediction_snapshots and prediction_outcomes
|
||||
- _Requirements: 16.1, 16.2, 16.3, 16.4, 16.5, 16.6, 14.1, 14.2, 14.3, 14.4_
|
||||
|
||||
- [x] 2. Phase 1 — Prediction capture, outcome evaluation, core metrics, and dashboard API
|
||||
- [x] 2.1 Implement Prediction Snapshot Writer (`services/validation/prediction_snapshot.py`)
|
||||
- Create `services/validation/__init__.py`
|
||||
- Define `SECTOR_ETF_MAP`, `EVALUATION_HORIZONS`, `MAX_SINGLE_DOCUMENT_WEIGHT` constants
|
||||
- Implement `PredictionSnapshot` and `SignalEvidenceLink` dataclasses
|
||||
- Implement `compute_canonical_evidence_key(title, url)` — SHA256 of normalized title + normalized URL (lowercase, strip whitespace for title; lowercase, strip query params for URL)
|
||||
- Implement `fetch_latest_close_price(pool, ticker)` — query most recent close from market_snapshots
|
||||
- Implement `create_prediction_snapshot(pool, recommendation, trend_summary, evidence_signals, evidence_docs)` — fetch prices (ticker, SPY, sector ETF), compute canonical keys, detect duplicates, clamp weights to MAX_SINGLE_DOCUMENT_WEIGHT, compute contribution scores (one-vote-per-canonical-key), persist snapshot + evidence links in a transaction
|
||||
- Implement `compute_contribution_scores(weights)` — each score = weight_i / sum(weights), sums to 1.0
|
||||
- Handle NULL prices gracefully (log warning, store NULL, don't fail)
|
||||
- _Requirements: 1.1, 1.2, 1.3, 1.4, 1.5, 1.6, 1.7, 2.1, 2.2, 2.3, 2.4, 2.5, 2.6, 3.1, 3.2, 3.3, 3.4_
|
||||
|
||||
- [x] 2.2 Write property test for canonical evidence key determinism and idempotence
|
||||
- **Property 4: Canonical Evidence Key Determinism and Normalization Idempotence**
|
||||
- Test that same (title, url) always produces same key
|
||||
- Test that normalizing already-normalized input produces same key
|
||||
- **Validates: Requirements 2.3, 17.4**
|
||||
|
||||
- [x] 2.3 Write property test for contribution score sum-to-one and range
|
||||
- **Property 7: Contribution Score Sum-to-One and Range**
|
||||
- Test that all scores in [0.0, 1.0] and sum to 1.0 (within 1e-9 tolerance)
|
||||
- Test that empty input returns empty list
|
||||
- **Validates: Requirements 2.5, 17.7**
|
||||
|
||||
- [x] 2.4 Implement Outcome Evaluator (`services/validation/outcome_evaluator.py`)
|
||||
- Define `PredictionOutcome` dataclass and `HORIZON_DURATIONS` mapping
|
||||
- Implement `evaluate_matured_predictions(pool)` — find snapshots where horizon elapsed and outcome not recorded, evaluate each
|
||||
- Implement `evaluate_single_prediction(pool, snapshot, horizon)` — fetch future price at horizon endpoint, compute future_return, SPY return, sector ETF return, excess returns, direction_correct, profitable; return None if future price unavailable
|
||||
- Evaluate across all 5 horizons: 1h, 6h, 1d, 7d, 30d
|
||||
- Skip horizons where future price is unavailable (retry next run)
|
||||
- Store results in prediction_outcomes table
|
||||
- _Requirements: 4.1, 4.2, 4.3, 4.4, 4.5, 4.6, 4.7, 4.8, 4.9, 4.10_
|
||||
|
||||
- [x] 2.5 Implement Metrics Engine (`services/validation/metrics.py`)
|
||||
- Define `CONFIDENCE_BUCKETS`, `LOOKBACK_WINDOWS` constants
|
||||
- Define `CalibrationBucket` and `ModelMetricSnapshot` dataclasses
|
||||
- Implement `compute_calibration_error(confidences, outcomes)` — group into 5 confidence buckets, compute ECE as weighted average of |avg_conf - win_rate|, flag miscalibrated buckets (|diff| > 0.15)
|
||||
- Implement `compute_brier_score(p_bulls, outcomes)` — mean((p_bull - outcome)^2)
|
||||
- Implement `compute_information_coefficient(scores, returns)` — Pearson correlation, return None when < 30 data points
|
||||
- Implement `compute_rank_information_coefficient(scores, returns)` — Spearman rank correlation, return None when < 30 data points
|
||||
- Implement `compute_contribution_scores(weights)` — weight_i / sum(weights), sums to 1.0
|
||||
- Implement benchmark metrics: average excess return vs SPY, vs sector ETF, hit rate improvement
|
||||
- Implement `compute_and_store_metric_snapshots(pool)` — compute for all lookback/horizon combinations (4 lookbacks × 5 horizons), persist to model_metric_snapshots
|
||||
- _Requirements: 5.1, 5.2, 5.3, 5.4, 5.5, 5.6, 6.1, 6.2, 6.3, 6.4, 6.5, 9.1, 9.2, 9.3, 9.4, 10.1, 10.2, 10.3, 10.4, 10.5_
|
||||
|
||||
- [x] 2.6 Write property test for ECE range and round-trip
|
||||
- **Property 1: Calibration Error Range and Round-Trip**
|
||||
- Test ECE in [0.0, 1.0] for all valid distributions
|
||||
- Test ECE = 0.0 when every bucket's win rate matches avg confidence
|
||||
- **Validates: Requirements 5.1, 5.3, 17.1**
|
||||
|
||||
- [x] 2.7 Write property test for Brier score range and perfect prediction
|
||||
- **Property 2: Brier Score Range and Perfect Prediction**
|
||||
- Test Brier in [0.0, 1.0] for all valid (p_bull, outcome) pairs
|
||||
- Test Brier = 0.0 when all predictions perfectly correct
|
||||
- **Validates: Requirements 5.4, 17.2**
|
||||
|
||||
- [x] 2.8 Write property test for IC range and perfect correlation
|
||||
- **Property 3: Information Coefficient Range and Perfect Correlation**
|
||||
- Test IC in [-1.0, 1.0] for all valid (score, return) pairs with ≥30 elements
|
||||
- Test IC = 1.0 for perfectly positively correlated data
|
||||
- **Validates: Requirements 6.1, 6.2, 17.3**
|
||||
|
||||
- [x] 2.9 Implement Dashboard API endpoints in `services/api/app.py`
|
||||
- Add `/api/validation/summary` GET — return latest model_metric_snapshot + gate status
|
||||
- Add `/api/validation/calibration` GET — return calibration table with buckets
|
||||
- Add `/api/validation/ic-by-horizon` GET — return IC and Rank IC per horizon
|
||||
- Add `/api/validation/gate-status` GET — return quality gate evaluation detail
|
||||
- All endpoints accept optional `lookback` (default "30d") and `horizon` (default "7d") query params
|
||||
- _Requirements: 12.1, 12.2, 12.3, 12.7_
|
||||
|
||||
- [x] 2.10 Add frontend validation API hooks in `frontend/src/api/hooks.ts`
|
||||
- Add `useValidationSummary(lookback?, horizon?)` hook for `/api/validation/summary`
|
||||
- Add `useValidationCalibration(lookback?, horizon?)` hook for `/api/validation/calibration`
|
||||
- Add `useValidationICByHorizon(lookback?)` hook for `/api/validation/ic-by-horizon`
|
||||
- Add `useValidationGateStatus()` hook for `/api/validation/gate-status`
|
||||
- _Requirements: 12.1, 12.2, 12.3, 12.7_
|
||||
|
||||
- [x] 2.11 Upgrade OpsModel page (`frontend/src/pages/OpsModel.tsx`) — Phase 1 dashboard
|
||||
- Add tabbed layout: existing "Extraction Performance" tab + new "Model Validation" tab
|
||||
- Add summary cards: prediction count, win rate, directional accuracy, IC, Rank IC, Brier score, ECE, avg excess return vs SPY, gate status
|
||||
- Add calibration table with confidence buckets, avg confidence, observed win rate, count, miscalibration flag
|
||||
- Highlight miscalibrated buckets (|avg_confidence - observed_win_rate| > 0.15) with warning indicator
|
||||
- Add IC-by-horizon table showing IC and Rank IC for each horizon
|
||||
- Add gate status indicator (pass/fail with threshold details)
|
||||
- _Requirements: 12.1, 12.2, 12.3, 12.7, 12.8, 12.9_
|
||||
|
||||
- [x] 3. Checkpoint — Phase 1 verification
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [x] 4. Phase 2 — Attribution engine and source/catalyst truth tables
|
||||
- [x] 4.1 Implement Attribution Engine (`services/validation/attribution.py`)
|
||||
- Define `SourceAttribution`, `CatalystAttribution`, `LayerAttribution` dataclasses
|
||||
- Implement `compute_source_attribution(pool, lookback_days, horizon)` — join signal_evidence_links with prediction_outcomes, group by source; compute prediction count, avg weight, avg contribution score, win rate, avg future return, avg excess return vs SPY, IC, duplicate rate
|
||||
- Implement `compute_catalyst_attribution(pool, lookback_days, horizon)` — same metrics grouped by catalyst_type
|
||||
- Implement `compute_layer_attribution(pool, lookback_days, horizon)` — compute per-layer (company, macro, competitive) avg contribution %, dominant win rate (layer > 30% contribution), dominant IC
|
||||
- _Requirements: 7.1, 7.2, 7.3, 7.4, 7.5, 7.6, 7.7_
|
||||
|
||||
- [x] 4.2 Implement Calibration Engine (`services/validation/calibration.py`)
|
||||
- Implement `compute_source_reliability(observed_win_rate, sample_count, prior_strength=30)` — Bayesian shrinkage: `0.5 + (n / (n + 30)) * (observed_win_rate - 0.5)`; return 0.5 when n=0
|
||||
- Implement `compute_adjusted_evidence_weight(base_weight, reliability)` — `base_weight * (0.5 + reliability)`, clamped to [0.1, 2.0]
|
||||
- Implement `update_source_reliabilities(pool)` — recompute from latest outcomes, update source_accuracy table
|
||||
- _Requirements: 8.1, 8.2, 8.3, 8.4, 8.5_
|
||||
|
||||
- [x] 4.3 Write property test for source reliability Bayesian shrinkage bounds and convergence
|
||||
- **Property 5: Source Reliability Bayesian Shrinkage Bounds and Convergence**
|
||||
- Test reliability in [0.0, 1.0] for all valid inputs
|
||||
- Test reliability = 0.5 when sample_count = 0
|
||||
- Test reliability approaches observed_win_rate as sample_count → ∞
|
||||
- **Validates: Requirements 8.1, 8.2, 17.5**
|
||||
|
||||
- [x] 4.4 Add attribution API endpoints in `services/api/app.py`
|
||||
- Add `/api/validation/attribution/sources` GET — return per-source performance metrics
|
||||
- Add `/api/validation/attribution/catalysts` GET — return per-catalyst performance metrics
|
||||
- Add `/api/validation/attribution/layers` GET — return per-layer performance metrics
|
||||
- All endpoints accept optional `lookback` (default "30d") and `horizon` (default "7d") query params
|
||||
- _Requirements: 12.4, 12.5, 12.6_
|
||||
|
||||
- [x] 4.5 Add frontend attribution hooks in `frontend/src/api/hooks.ts`
|
||||
- Add `useValidationAttributionSources(lookback?, horizon?)` hook
|
||||
- Add `useValidationAttributionCatalysts(lookback?, horizon?)` hook
|
||||
- Add `useValidationAttributionLayers(lookback?, horizon?)` hook
|
||||
- _Requirements: 12.4, 12.5, 12.6_
|
||||
|
||||
- [x] 4.6 Extend OpsModel page with attribution tables
|
||||
- Add source performance table (source, win rate, IC, avg return, duplicate rate)
|
||||
- Add catalyst truth table (catalyst type, win rate, avg return, IC)
|
||||
- Add layer attribution table (company/macro/competitive contribution %, dominant win rate, IC)
|
||||
- _Requirements: 12.4, 12.5, 12.6, 12.8_
|
||||
|
||||
- [x] 5. Checkpoint — Phase 2 verification
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [x] 6. Phase 3 — Quality gate, recommendation enhancements, and pipeline wiring
|
||||
- [x] 6.1 Implement Quality Gate (`services/trading/model_quality_gate.py`)
|
||||
- Define `QualityGateConfig` dataclass with default thresholds (min_prediction_count=100, min_ic=0.03, min_win_rate=0.53, max_ece=0.15, min_excess_return_vs_spy=0.0, max_snapshot_age_hours=24)
|
||||
- Define `GateThresholdResult` and `QualityGateResult` dataclasses
|
||||
- Implement `evaluate_quality_gate(pool, config)` — read most recent model_metric_snapshot (30d lookback, 7d horizon), evaluate each threshold, store result in risk_configs under 'model_quality_gate' key
|
||||
- Implement `load_gate_config_from_db(pool)` — load thresholds from risk_configs with defaults
|
||||
- Default to paper-only mode when no snapshots exist or snapshot is stale (>24h)
|
||||
- Log gate evaluation result with threshold pass/fail details
|
||||
- _Requirements: 11.1, 11.2, 11.3, 11.4, 11.5, 11.6, 11.7_
|
||||
|
||||
- [x] 6.2 Write property test for quality gate determinism and threshold monotonicity
|
||||
- **Property 6: Quality Gate Determinism and Threshold Monotonicity**
|
||||
- Test same inputs always produce same pass/fail result
|
||||
- Test relaxing any threshold never causes a previously passing gate to fail
|
||||
- **Validates: Requirements 11.1, 17.6**
|
||||
|
||||
- [x] 6.3 Wire Quality Gate into aggregation cycle (`services/aggregation/worker.py`)
|
||||
- Call `evaluate_quality_gate` at the start of each aggregation cycle
|
||||
- When gate fails, force all recommendations to paper mode
|
||||
- Log gate status at cycle start
|
||||
- _Requirements: 11.2, 11.3_
|
||||
|
||||
- [x] 6.4 Wire Prediction Snapshot Writer into recommendation engine
|
||||
- After recommendation is generated in `services/recommendation/eligibility.py` or the calling code, call `create_prediction_snapshot` to capture the prediction state
|
||||
- Pass recommendation, trend_summary, evidence signals, and evidence docs
|
||||
- Handle snapshot creation failure gracefully (log error, don't block recommendation)
|
||||
- _Requirements: 1.1, 1.6_
|
||||
|
||||
- [x] 6.5 Enhance recommendation display on frontend
|
||||
- Update `frontend/src/pages/RecommendationDetail` (or relevant recommendation display component) to show:
|
||||
- Original confidence alongside calibrated confidence (historical win rate for that bucket)
|
||||
- Historical win rate for similar confidence levels
|
||||
- Evidence count, unique evidence count, duplicate evidence count
|
||||
- Source reliability indicator for primary contributing sources
|
||||
- Live eligibility status with reason (gate passed or which threshold failed)
|
||||
- Add warning badge when duplicate evidence count > 20% of total evidence count
|
||||
- Add warning badge when primary source reliability < 0.4
|
||||
- _Requirements: 13.1, 13.2, 13.3, 13.4, 13.5, 13.6, 13.7_
|
||||
|
||||
- [x] 7. Checkpoint — Phase 3 verification
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [x] 8. Phase 4 — Backtest replay integration and unit tests
|
||||
- [x] 8.1 Add validation mode to BacktestReplay (`services/trading/backtest_replay.py`)
|
||||
- Add `validation_mode: bool = False` parameter to `BacktestReplay.run()`
|
||||
- When validation_mode=True, create prediction snapshots for each historical recommendation using only data available at that point in time
|
||||
- Evaluate prediction outcomes using market prices from the appropriate future horizon
|
||||
- Prevent future data leakage: no market data after prediction generation time used during snapshot creation
|
||||
- After backtest completes, trigger model metrics computation over the backtest period, tag snapshots with backtest_id
|
||||
- _Requirements: 15.1, 15.2, 15.3, 15.4, 15.5_
|
||||
|
||||
- [x] 8.2 Write unit tests for prediction snapshot writer (`tests/test_model_validation_unit.py`)
|
||||
- Test canonical evidence key: known title/URL → expected SHA256, empty inputs, unicode
|
||||
- Test duplicate detection: 3 docs with 2 sharing a key → 1 marked duplicate
|
||||
- Test contribution scores: [0.5, 0.3, 0.2] → [0.5, 0.3, 0.2], single doc → [1.0]
|
||||
- Test weight clamping: weight 1.5 → clamped to 1.0
|
||||
- _Requirements: 1.1, 2.3, 2.4, 2.5, 3.3_
|
||||
|
||||
- [x] 8.3 Write unit tests for outcome evaluator (`tests/test_model_validation_unit.py`)
|
||||
- Test future return computation: price 100→110 → 0.10, price 100→90 → -0.10
|
||||
- Test direction_correct logic: bullish+positive → true, bullish+negative → false
|
||||
- Test profitable logic: buy+positive → true, sell+negative → true
|
||||
- Test excess return: ticker 10%, SPY 5% → excess 5%
|
||||
- _Requirements: 4.2, 4.5, 4.6, 4.7_
|
||||
|
||||
- [x] 8.4 Write unit tests for metrics engine (`tests/test_model_validation_unit.py`)
|
||||
- Test ECE specific values: perfect calibration → 0.0, all overconfident → positive ECE
|
||||
- Test Brier score: all correct at p=1.0 → 0.0, all wrong at p=1.0 → 1.0
|
||||
- Test IC: perfect correlation → 1.0, anti-correlation → -1.0, < 30 → None
|
||||
- _Requirements: 5.3, 5.4, 6.1, 6.2, 6.5_
|
||||
|
||||
- [x] 8.5 Write unit tests for calibration engine (`tests/test_model_validation_unit.py`)
|
||||
- Test source reliability: n=0 → 0.5, n=1000 with wr=0.8 → ≈0.8, n=30 with wr=0.7 → 0.6
|
||||
- Test adjusted evidence weight: reliability=0.5 → base*1.0, clamping to [0.1, 2.0]
|
||||
- _Requirements: 8.1, 8.2, 8.3_
|
||||
|
||||
- [x] 8.6 Write unit tests for quality gate (`tests/test_model_validation_unit.py`)
|
||||
- Test all thresholds met → pass
|
||||
- Test one threshold failed → fail with reason
|
||||
- Test fail-safe: no snapshots → paper-only, stale snapshot → paper-only
|
||||
- _Requirements: 11.1, 11.6_
|
||||
|
||||
- [x] 8.7 Write frontend tests for validation dashboard (`frontend/src/test/pages.test.tsx`)
|
||||
- Add MSW mock handlers for `/api/validation/summary`, `/api/validation/calibration`, `/api/validation/gate-status`
|
||||
- Test OpsModel page renders validation tab with summary cards
|
||||
- Test calibration table renders buckets with miscalibration warning
|
||||
- Test gate status indicator renders pass/fail
|
||||
- _Requirements: 12.8, 12.9_
|
||||
|
||||
- [x] 9. Final checkpoint — Ensure all tests pass
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
## Notes
|
||||
|
||||
- Tasks marked with `*` are optional and can be skipped for faster MVP
|
||||
- Each task references specific requirements for traceability
|
||||
- Checkpoints ensure incremental validation after each phase
|
||||
- Property tests validate the 7 universal correctness properties from the design document
|
||||
- Unit tests validate specific examples, edge cases, and integration points
|
||||
- The design uses Python for backend and TypeScript for frontend — no language selection needed
|
||||
- Migration number is 035 (existing migrations go up to 034)
|
||||
- All new service modules go under `services/validation/` except the quality gate which goes in `services/trading/`
|
||||
- The 7 new API endpoints are added to the existing `services/api/app.py`
|
||||
- Frontend hooks follow existing patterns in `frontend/src/api/hooks.ts`
|
||||
- Phase 1 delivers the core feedback loop (capture → evaluate → measure → display)
|
||||
- Phase 2 adds attribution depth (which sources/catalysts/layers work best)
|
||||
- Phase 3 adds safety (quality gate) and UX (recommendation warnings)
|
||||
- Phase 4 adds historical analysis (backtest validation mode) and comprehensive tests
|
||||
@@ -0,0 +1 @@
|
||||
{"specId": "f5d99301-94ef-4dc2-8ba4-ccefeee7ecba", "workflowType": "requirements-first", "specType": "bugfix"}
|
||||
@@ -0,0 +1,73 @@
|
||||
# Bugfix Requirements Document
|
||||
|
||||
## Introduction
|
||||
|
||||
Multiple operational bugs discovered in the stonks-beta namespace prevent the validation/calibration feedback loop from functioning and degrade ingestion throughput. The core issue is that the outcome evaluation → metrics computation → quality gate pipeline is completely disconnected from the production scheduler, making the platform unable to self-calibrate or validate predictions. Additionally, Polygon API rate limiting causes ~40% request failures per cycle, a broken config query prevents the v3 engine from being toggled, several periodic snapshot tasks are missing from the scheduler, the lake-publisher deployment is idle/redundant, and order rejection reasons are lost.
|
||||
|
||||
## Bug Analysis
|
||||
|
||||
### Current Behavior (Defect)
|
||||
|
||||
1.1 WHEN the scheduler enqueues ingestion jobs for all 50 tickers' news_api and market_api sources simultaneously THEN the system exhausts the Polygon free-tier rate limit (5 req/min) resulting in ~40% of sources receiving HTTP 429 Too Many Requests every cycle
|
||||
|
||||
1.2 WHEN the aggregation worker reads the v3_engine_enabled flag via `_V3_ENGINE_FLAG_QUERY` THEN the system queries non-existent columns `key` and `value` on the `risk_configs` table (actual schema: `name` varchar, `config` JSONB) causing a PostgreSQL error every aggregation cycle
|
||||
|
||||
1.3 WHEN a scheduler cycle completes THEN the system never calls `evaluate_matured_predictions()` because it is not wired into the scheduler's main loop — only imported in `backtest_replay.py`
|
||||
|
||||
1.4 WHEN a scheduler cycle completes THEN the system never calls `compute_and_store_metric_snapshots()` because it is not wired into the scheduler's main loop — only called from backtest replay
|
||||
|
||||
1.5 WHEN the model quality gate evaluates trading eligibility THEN the system always fails with "no model metric snapshot available — defaulting to paper-only" because `model_metric_snapshots` table is permanently empty (consequence of bug 1.4)
|
||||
|
||||
1.6 WHEN the trading engine runs daily THEN the system never captures portfolio state snapshots to the `portfolio_snapshots` table because no periodic scheduler task invokes this capture
|
||||
|
||||
1.7 WHEN the trading engine runs daily THEN the system never captures risk state snapshots to the `daily_risk_snapshots` table because no periodic scheduler task invokes this capture
|
||||
|
||||
1.8 WHEN a prediction snapshot is created while Polygon rate-limiting has prevented the market data fetch THEN the system stores NULL in `price_at_prediction` (affecting 21% of snapshots), degrading downstream outcome evaluation accuracy
|
||||
|
||||
1.9 WHEN the standalone `lake-publisher` deployment polls `stonks:beta:queue:lake_publish` THEN the queue is always empty (0 items) because all lake publishing happens inline in broker-adapter and recommendation services — the deployment consumes zero work and wastes resources
|
||||
|
||||
1.10 WHEN Alpaca returns HTTP 401 for an order submission THEN the system sets order status to "rejected" but leaves the `rejection_reason` column NULL, capturing the error message only in the `decision_trace` JSONB field
|
||||
|
||||
### Expected Behavior (Correct)
|
||||
|
||||
2.1 WHEN the scheduler enqueues ingestion jobs for Polygon-backed sources (news_api, market_api) THEN the system SHALL pace/stagger requests across the polling interval to stay within the Polygon rate limit, achieving near-zero 429 responses per cycle
|
||||
|
||||
2.2 WHEN the aggregation worker reads the v3_engine_enabled flag THEN the system SHALL query `SELECT config FROM risk_configs WHERE name = 'v3_engine_enabled'` and parse the JSONB value to determine the boolean toggle state
|
||||
|
||||
2.3 WHEN a scheduler cycle completes and sufficient time has elapsed since the last evaluation THEN the system SHALL call `evaluate_matured_predictions()` to evaluate prediction snapshots whose horizon has elapsed, populating the `prediction_outcomes` table
|
||||
|
||||
2.4 WHEN a scheduler cycle completes and sufficient time has elapsed since the last computation THEN the system SHALL call `compute_and_store_metric_snapshots()` to compute aggregate model metrics across all lookback/horizon combinations, populating `model_metric_snapshots`
|
||||
|
||||
2.5 WHEN the model quality gate evaluates trading eligibility THEN the system SHALL have recent metric snapshots available and evaluate thresholds against actual model performance data
|
||||
|
||||
2.6 WHEN market hours close (or on a daily schedule) THEN the system SHALL capture and persist the current portfolio state to `portfolio_snapshots` including value, returns, positions, and risk metrics
|
||||
|
||||
2.7 WHEN market hours close (or on a daily schedule) THEN the system SHALL capture and persist the current risk state to `daily_risk_snapshots` including portfolio value, daily P&L, trade count, and sector positions
|
||||
|
||||
2.8 WHEN a prediction snapshot is created and market price is unavailable due to rate limiting THEN the system SHALL retry the price fetch or defer the snapshot until price data is available, reducing NULL `price_at_prediction` occurrences to near zero
|
||||
|
||||
2.9 WHEN the lake-publisher deployment architecture is reviewed THEN the system SHALL either route lake publish jobs through the Redis queue to the standalone deployment, or remove the redundant deployment — eliminating the idle pod
|
||||
|
||||
2.10 WHEN Alpaca returns an HTTP error (401, 403, or any rejection) for an order submission THEN the system SHALL populate the `rejection_reason` column with the HTTP error message/status in addition to recording it in `decision_trace`
|
||||
|
||||
### Unchanged Behavior (Regression Prevention)
|
||||
|
||||
3.1 WHEN sources with valid rate-limit headroom are enqueued THEN the system SHALL CONTINUE TO enqueue and process them without artificial delay
|
||||
|
||||
3.2 WHEN risk_configs is queried for other configuration keys (e.g., `model_quality_gate_config`, `macro_enabled`) THEN the system SHALL CONTINUE TO read them correctly using the existing `name`/`config` column pattern
|
||||
|
||||
3.3 WHEN the backtest replay module calls `evaluate_matured_predictions()` and `compute_and_store_metric_snapshots()` THEN the system SHALL CONTINUE TO execute them as part of backtest validation
|
||||
|
||||
3.4 WHEN prediction snapshots are created with available market prices THEN the system SHALL CONTINUE TO store the correct `price_at_prediction` value immediately
|
||||
|
||||
3.5 WHEN the existing inline lake publishing in broker-adapter and recommendation services writes facts THEN the system SHALL CONTINUE TO produce correct Parquet partitions in MinIO
|
||||
|
||||
3.6 WHEN orders succeed (HTTP 200 from Alpaca) THEN the system SHALL CONTINUE TO process them normally without modifying the `rejection_reason` column
|
||||
|
||||
3.7 WHEN the scheduler runs ingestion, extraction, aggregation, recommendation, and trading tasks THEN the system SHALL CONTINUE TO execute them on the existing cadence without disruption
|
||||
|
||||
3.8 WHEN the trading engine makes decisions and submits orders THEN the system SHALL CONTINUE TO record full decision context in `decision_trace` JSONB as before
|
||||
|
||||
3.9 WHEN the model quality gate passes (once metrics are populated) THEN the system SHALL CONTINUE TO allow promotion to live trading mode per existing threshold logic
|
||||
|
||||
3.10 WHEN the reporting collector fetches portfolio_snapshots and daily_risk_snapshots for report generation THEN the system SHALL CONTINUE TO query and render them using the existing schema
|
||||
@@ -0,0 +1,396 @@
|
||||
# Technical Design: ops-pipeline-fixes
|
||||
|
||||
## Overview
|
||||
|
||||
This design addresses 10 operational bugs that prevent the validation/calibration feedback loop from functioning and degrade ingestion throughput in the `stonks-beta` namespace. The fixes span the scheduler (rate limiting + periodic tasks), aggregation worker (config query), broker service (rejection reason), prediction snapshot (price fallback), and Helm chart (dead pod removal). All changes are localized with graceful fallbacks and no schema migrations required.
|
||||
|
||||
## Bug Details
|
||||
|
||||
Multiple operational bugs in the `stonks-beta` namespace prevent the validation/calibration feedback loop from functioning and degrade ingestion throughput. The core pipeline (ingestion → extraction → aggregation → recommendation → trading) flows end-to-end, but:
|
||||
- The outcome evaluation → metrics computation → quality gate feedback loop is completely disconnected
|
||||
- Polygon API rate limiting causes ~40% ingestion failures per cycle
|
||||
- A broken config query prevents the v3 engine toggle from working
|
||||
- Portfolio/risk snapshots are never captured
|
||||
- Order rejection reasons are lost
|
||||
|
||||
```mermaid
|
||||
graph TD
|
||||
subgraph "Scheduler (services/scheduler/app.py)"
|
||||
A[schedule_cycle] -->|paced enqueue| B[Ingestion Queue]
|
||||
C[validation_cycle] -->|hourly| D[evaluate_matured_predictions]
|
||||
C -->|after outcomes| E[compute_and_store_metric_snapshots]
|
||||
F[snapshot_cycle] -->|daily 16:30 ET| G[capture_portfolio_snapshot]
|
||||
F -->|daily 16:30 ET| H[capture_risk_snapshot]
|
||||
end
|
||||
|
||||
subgraph "Aggregation (services/aggregation/worker.py)"
|
||||
I[_read_v3_flag] -->|fixed query| J[risk_configs.config JSONB]
|
||||
end
|
||||
|
||||
subgraph "Broker (services/adapters/broker_service.py)"
|
||||
K[persist_order] -->|rejected status| L[orders.rejection_reason]
|
||||
end
|
||||
|
||||
D --> O[prediction_outcomes]
|
||||
E --> P[model_metric_snapshots]
|
||||
P --> Q[Quality Gate]
|
||||
```
|
||||
|
||||
## Expected Behavior
|
||||
|
||||
2.1 The scheduler SHALL pace Polygon API requests within the free-tier limit (~5 req/min), achieving near-zero 429 responses per cycle.
|
||||
|
||||
2.2 The aggregation worker SHALL read `v3_engine_enabled` from the `risk_configs` JSONB `config` column (not non-existent `key`/`value` columns).
|
||||
|
||||
2.3 The scheduler SHALL call `evaluate_matured_predictions()` hourly to populate `prediction_outcomes`.
|
||||
|
||||
2.4 The scheduler SHALL call `compute_and_store_metric_snapshots()` after outcome evaluation to populate `model_metric_snapshots`.
|
||||
|
||||
2.5 The quality gate SHALL have recent metric data available once the validation cycle runs.
|
||||
|
||||
2.6 The scheduler SHALL capture daily portfolio snapshots to `portfolio_snapshots` after market close.
|
||||
|
||||
2.7 The scheduler SHALL capture daily risk snapshots to `daily_risk_snapshots` after market close.
|
||||
|
||||
2.8 Prediction snapshots SHALL fall back to positions table prices when market_snapshots data is unavailable.
|
||||
|
||||
2.9 The lake-publisher deployment SHALL be scaled to 0 (idle pod, wasted resources).
|
||||
|
||||
2.10 The broker service SHALL populate `rejection_reason` on orders when broker or risk engine rejects.
|
||||
|
||||
## Hypothesized Root Cause
|
||||
|
||||
### Bug 1.1 — Polygon Rate Limiting
|
||||
`POLYGON_GLOBAL_RATE_LIMIT = 45` in `services/scheduler/app.py` is set for a paid Polygon plan but the deployed instance uses the free tier (5 req/min). All 50+ sources are attempted per cycle, exhausting the limit instantly.
|
||||
|
||||
### Bug 1.2 — v3_engine_enabled Config Read
|
||||
`_V3_ENGINE_FLAG_QUERY` in `services/aggregation/worker.py` reads `SELECT value FROM risk_configs WHERE key = 'v3_engine_enabled'`. The actual table has columns `name` (varchar) and `config` (JSONB) — no `key` or `value` column exists.
|
||||
|
||||
### Bug 1.3 & 1.4 — Outcome Evaluator & Metrics Never Scheduled
|
||||
`evaluate_matured_predictions()` and `compute_and_store_metric_snapshots()` exist in `services/validation/` but are only imported in `services/trading/backtest_replay.py`. The scheduler main loop in `services/scheduler/app.py` has no call to either function.
|
||||
|
||||
### Bug 1.5 — Quality Gate Permanently Failing
|
||||
`services/trading/model_quality_gate.py` queries `model_metric_snapshots` which is always empty (consequence of 1.4). Returns "no model metric snapshot available — defaulting to paper-only" every time.
|
||||
|
||||
### Bug 1.6 & 1.7 — Portfolio/Risk Snapshots
|
||||
The trading engine has `_persist_daily_snapshot()` but it only executes when the engine's main loop is actively processing trades. The trading-engine pod shows only health checks — its main loop isn't cycling because there are no active trade triggers flowing through it. No fallback capture exists in the scheduler.
|
||||
|
||||
### Bug 1.8 — Market Price Gaps
|
||||
`services/validation/prediction_snapshot.py` queries `market_snapshots` for price at prediction time. When Polygon rate limiting prevents market data fetches, no snapshot exists and `price_at_prediction` is NULL. 21% of snapshots affected.
|
||||
|
||||
### Bug 1.9 — Lake Publisher Idle
|
||||
The standalone `lake-publisher` deployment polls `stonks:beta:queue:lake_publish` but all services (broker-adapter, recommendation) import `services.lake_publisher.worker` directly and publish inline — never pushing to the Redis queue.
|
||||
|
||||
### Bug 1.10 — Order rejection_reason NULL
|
||||
`_INSERT_ORDER` SQL in `services/adapters/broker_service.py` doesn't include `rejection_reason` or `rejected_at` columns. The error is stored in `decision_trace` JSONB but the dedicated column stays NULL. The reconciliation path (`_reconcile_open_orders`) does set these columns, but initial persist does not.
|
||||
|
||||
## Fix Implementation
|
||||
|
||||
### Fix 1: Polygon Rate Limit Constant (Bug 1.1)
|
||||
|
||||
**File:** `services/scheduler/app.py`
|
||||
|
||||
Replace the hardcoded constant with an env-configurable value defaulting to 5:
|
||||
|
||||
```python
|
||||
# Before:
|
||||
POLYGON_GLOBAL_RATE_LIMIT: int = 45
|
||||
|
||||
# After:
|
||||
POLYGON_GLOBAL_RATE_LIMIT: int = int(os.getenv("POLYGON_GLOBAL_RATE_LIMIT", "5"))
|
||||
```
|
||||
|
||||
The existing `check_rate_limit()` function already implements per-minute windowed counting and skips sources once the limit is hit. By reducing the constant to match the free-tier limit, the system will naturally pace — enqueuing ~5 Polygon sources per minute across scheduler ticks (15s interval = 4 ticks/min). Skipped sources are retried next cycle.
|
||||
|
||||
**Validates:** Bugfix 2.1; Regression 3.1, 3.7
|
||||
|
||||
---
|
||||
|
||||
### Fix 2: v3_engine_enabled Config Query (Bug 1.2)
|
||||
|
||||
**File:** `services/aggregation/worker.py`
|
||||
|
||||
Replace the broken query and function:
|
||||
|
||||
```python
|
||||
# Before:
|
||||
_V3_ENGINE_FLAG_QUERY = """
|
||||
SELECT value FROM risk_configs WHERE key = 'v3_engine_enabled'
|
||||
"""
|
||||
|
||||
# After:
|
||||
_V3_ENGINE_FLAG_QUERY = """
|
||||
SELECT config->>'v3_engine_enabled' AS enabled
|
||||
FROM risk_configs
|
||||
WHERE name = 'default' AND active = TRUE
|
||||
LIMIT 1
|
||||
"""
|
||||
|
||||
async def _read_v3_flag(pool: asyncpg.Pool) -> bool:
|
||||
"""Read v3_engine_enabled from risk_configs JSONB. Default False on error."""
|
||||
try:
|
||||
row = await pool.fetchrow(_V3_ENGINE_FLAG_QUERY)
|
||||
if row and row["enabled"]:
|
||||
return row["enabled"].lower() in ("true", "1", "yes")
|
||||
return False
|
||||
except Exception as e:
|
||||
logger.warning("Failed to read v3_engine_enabled flag: %s", e)
|
||||
return False
|
||||
```
|
||||
|
||||
Reads from the `default` active risk_config's JSONB `config` field. Falls back to False (unchanged fail-safe).
|
||||
|
||||
**Validates:** Bugfix 2.2; Regression 3.2
|
||||
|
||||
---
|
||||
|
||||
### Fix 3: Validation Cycle in Scheduler (Bugs 1.3, 1.4, 1.5)
|
||||
|
||||
**File:** `services/scheduler/app.py`
|
||||
|
||||
Add a new periodic task (every ~240 ticks = ~60 minutes):
|
||||
|
||||
```python
|
||||
# New constant:
|
||||
VALIDATION_CYCLE_INTERVAL = int(os.getenv("VALIDATION_CYCLE_INTERVAL", "240"))
|
||||
|
||||
# New counter in main():
|
||||
validation_counter = 0
|
||||
|
||||
# In main loop after existing periodic tasks:
|
||||
validation_counter += 1
|
||||
if validation_counter >= VALIDATION_CYCLE_INTERVAL:
|
||||
validation_counter = 0
|
||||
await run_validation_cycle(pool)
|
||||
```
|
||||
|
||||
New function:
|
||||
|
||||
```python
|
||||
async def run_validation_cycle(pool: asyncpg.Pool) -> None:
|
||||
"""Run outcome evaluation and metric computation (hourly).
|
||||
|
||||
Requirements: 2.3, 2.4, 2.5
|
||||
"""
|
||||
from services.validation.outcome_evaluator import evaluate_matured_predictions
|
||||
from services.validation.metrics import compute_and_store_metric_snapshots
|
||||
|
||||
try:
|
||||
outcomes = await evaluate_matured_predictions(pool)
|
||||
logger.info("Validation: evaluated %d prediction outcomes", outcomes)
|
||||
except Exception:
|
||||
logger.exception("Validation: outcome evaluation failed")
|
||||
return # Skip metrics if outcomes failed
|
||||
|
||||
try:
|
||||
snapshots = await compute_and_store_metric_snapshots(pool)
|
||||
logger.info("Validation: computed %d metric snapshots", len(snapshots))
|
||||
except Exception:
|
||||
logger.exception("Validation: metric computation failed")
|
||||
```
|
||||
|
||||
**Validates:** Bugfix 2.3, 2.4, 2.5; Regression 3.3
|
||||
|
||||
---
|
||||
|
||||
### Fix 4: Daily Portfolio & Risk Snapshots (Bugs 1.6, 1.7)
|
||||
|
||||
**File:** `services/scheduler/app.py`
|
||||
|
||||
Add a daily snapshot task that runs every ~60 minutes but only captures once per day after 16:30 ET:
|
||||
|
||||
```python
|
||||
SNAPSHOT_CYCLE_INTERVAL = int(os.getenv("SNAPSHOT_CYCLE_INTERVAL", "240"))
|
||||
snapshot_counter = 0
|
||||
|
||||
# In main loop:
|
||||
snapshot_counter += 1
|
||||
if snapshot_counter >= SNAPSHOT_CYCLE_INTERVAL:
|
||||
snapshot_counter = 0
|
||||
await maybe_capture_daily_snapshots(pool)
|
||||
```
|
||||
|
||||
New function:
|
||||
|
||||
```python
|
||||
async def maybe_capture_daily_snapshots(pool: asyncpg.Pool) -> None:
|
||||
"""Capture portfolio and risk snapshots once daily after market close.
|
||||
|
||||
Requirements: 2.6, 2.7
|
||||
"""
|
||||
et_now = datetime.now(ZoneInfo("America/New_York"))
|
||||
|
||||
# Only after 4:30 PM ET
|
||||
if et_now.hour < 16 or (et_now.hour == 16 and et_now.minute < 30):
|
||||
return
|
||||
|
||||
today = et_now.date()
|
||||
|
||||
# Already captured today?
|
||||
existing = await pool.fetchval(
|
||||
"SELECT 1 FROM portfolio_snapshots WHERE snapshot_date = $1 LIMIT 1",
|
||||
today,
|
||||
)
|
||||
if existing:
|
||||
return
|
||||
|
||||
# Portfolio snapshot from positions + account data
|
||||
try:
|
||||
positions = await pool.fetch("SELECT * FROM positions WHERE quantity > 0")
|
||||
portfolio_value = sum(
|
||||
float(r["current_price"] or 0) * float(r["quantity"])
|
||||
for r in positions
|
||||
)
|
||||
unrealized_pnl = sum(float(r["unrealized_pnl"] or 0) for r in positions)
|
||||
|
||||
await pool.execute(
|
||||
"""INSERT INTO portfolio_snapshots
|
||||
(snapshot_date, portfolio_value, unrealized_pnl, positions)
|
||||
VALUES ($1, $2, $3, $4::jsonb)""",
|
||||
today, portfolio_value, unrealized_pnl,
|
||||
json.dumps([dict(r) for r in positions], default=str),
|
||||
)
|
||||
logger.info("Captured portfolio snapshot: value=%.2f", portfolio_value)
|
||||
except Exception:
|
||||
logger.exception("Failed to capture portfolio snapshot")
|
||||
|
||||
# Risk snapshot from daily activity
|
||||
try:
|
||||
daily_orders = await pool.fetchval(
|
||||
"SELECT count(*) FROM orders WHERE created_at::date = $1", today
|
||||
)
|
||||
daily_pnl = sum(float(r["unrealized_pnl"] or 0) for r in positions) if positions else 0.0
|
||||
|
||||
await pool.execute(
|
||||
"""INSERT INTO daily_risk_snapshots
|
||||
(account_id, snapshot_date, portfolio_value, daily_pnl, daily_trade_count)
|
||||
VALUES ((SELECT id FROM broker_accounts LIMIT 1), $1, $2, $3, $4)
|
||||
ON CONFLICT DO NOTHING""",
|
||||
today, portfolio_value, daily_pnl, daily_orders or 0,
|
||||
)
|
||||
logger.info("Captured risk snapshot: pnl=%.2f trades=%d", daily_pnl, daily_orders or 0)
|
||||
except Exception:
|
||||
logger.exception("Failed to capture risk snapshot")
|
||||
```
|
||||
|
||||
**Validates:** Bugfix 2.6, 2.7; Regression 3.10
|
||||
|
||||
---
|
||||
|
||||
### Fix 5: Prediction Price Fallback (Bug 1.8)
|
||||
|
||||
**File:** `services/validation/prediction_snapshot.py`
|
||||
|
||||
After the primary `market_snapshots` price lookup returns NULL, add a fallback:
|
||||
|
||||
```python
|
||||
# After market_snapshots lookup:
|
||||
if price_at_prediction is None:
|
||||
pos_row = await conn.fetchrow(
|
||||
"SELECT current_price FROM positions "
|
||||
"WHERE ticker = $1 AND current_price IS NOT NULL LIMIT 1",
|
||||
ticker,
|
||||
)
|
||||
if pos_row:
|
||||
price_at_prediction = float(pos_row["current_price"])
|
||||
```
|
||||
|
||||
Only covers tickers with open positions (currently 10). Acceptable tradeoff — most active tickers are the ones we hold.
|
||||
|
||||
**Validates:** Bugfix 2.8; Regression 3.4
|
||||
|
||||
---
|
||||
|
||||
### Fix 6: Lake Publisher Scale-Down (Bug 1.9)
|
||||
|
||||
**File:** `infra/helm/stonks-oracle/values.yaml`
|
||||
|
||||
```yaml
|
||||
# Change:
|
||||
replicas: 0
|
||||
```
|
||||
|
||||
Keeps the deployment definition intact for future use but schedules no pods.
|
||||
|
||||
**Validates:** Bugfix 2.9; Regression 3.5
|
||||
|
||||
---
|
||||
|
||||
### Fix 7: Order rejection_reason Population (Bug 1.10)
|
||||
|
||||
**File:** `services/adapters/broker_service.py`
|
||||
|
||||
Extend `_INSERT_ORDER` to include `rejection_reason` and `rejected_at`:
|
||||
|
||||
```python
|
||||
_INSERT_ORDER = """
|
||||
INSERT INTO orders (
|
||||
id, recommendation_id, broker_account_id, ticker, side, order_type,
|
||||
quantity, limit_price, stop_price, status, idempotency_key,
|
||||
broker_order_id, decision_trace, submitted_at, filled_at,
|
||||
fill_price, fill_quantity, rejection_reason, rejected_at
|
||||
) VALUES (
|
||||
$1::uuid, $2, $3::uuid, $4, $5, $6,
|
||||
$7, $8, $9, $10, $11,
|
||||
$12, $13::jsonb, $14, $15,
|
||||
$16, $17, $18, $19
|
||||
)
|
||||
ON CONFLICT (idempotency_key) DO UPDATE SET
|
||||
status = EXCLUDED.status,
|
||||
broker_order_id = EXCLUDED.broker_order_id,
|
||||
filled_at = EXCLUDED.filled_at,
|
||||
fill_price = EXCLUDED.fill_price,
|
||||
fill_quantity = EXCLUDED.fill_quantity,
|
||||
rejection_reason = COALESCE(EXCLUDED.rejection_reason, orders.rejection_reason),
|
||||
rejected_at = COALESCE(EXCLUDED.rejected_at, orders.rejected_at),
|
||||
updated_at = NOW()
|
||||
"""
|
||||
```
|
||||
|
||||
Update `persist_order()` to pass the new parameters:
|
||||
|
||||
```python
|
||||
rejection_reason = resp.error if resp.status == OrderStatus.REJECTED else None
|
||||
rejected_at = now if resp.status == OrderStatus.REJECTED else None
|
||||
# Add as params $18, $19
|
||||
```
|
||||
|
||||
**Validates:** Bugfix 2.10; Regression 3.6, 3.8
|
||||
|
||||
---
|
||||
|
||||
## Correctness Properties
|
||||
|
||||
Property 1: Rate limit compliance — After fix, the rolling 1-minute window for Polygon requests SHALL NOT exceed the configured limit (default 5). Existing `check_rate_limit()` windowed counter enforces this; we only change the threshold constant.
|
||||
|
||||
Property 2: Validation cycle completeness — `prediction_outcomes` row count SHALL grow monotonically after the first validation cycle runs. Each run finds matured snapshots not yet evaluated and persists outcomes.
|
||||
|
||||
Property 3: Metric snapshot freshness — `model_metric_snapshots` SHALL contain rows with `generated_at` within the last 2 hours after 2+ validation cycles. The quality gate can then evaluate against real data.
|
||||
|
||||
Property 4: Config read correctness — `_read_v3_flag()` SHALL return True when `risk_configs.config->>'v3_engine_enabled'` is `'true'` and False for all other values including NULL or missing key.
|
||||
|
||||
Property 5: Snapshot idempotency — `portfolio_snapshots` SHALL contain at most 1 row per `snapshot_date`. The `maybe_capture_daily_snapshots` function checks for existing rows before insert.
|
||||
|
||||
Property 6: Rejection reason preservation — Every order with `status = 'rejected'` persisted via `persist_order()` SHALL have a non-NULL `rejection_reason` extracted from the error response.
|
||||
|
||||
## Testing Strategy
|
||||
|
||||
- **Unit tests:** Update `test_scheduler.py` with a test verifying `run_validation_cycle` is called after the counter threshold. Test `_read_v3_flag` with mocked JSONB config returning various values.
|
||||
- **Integration tests:** Verify `persist_order` with rejected status populates `rejection_reason` column.
|
||||
- **Manual verification post-deploy:**
|
||||
- `kubectl logs deployment/scheduler -n stonks-beta --tail=100 | grep Validation` shows outcome counts
|
||||
- `SELECT count(*) FROM prediction_outcomes` starts growing within 1 hour
|
||||
- `SELECT count(*) FROM model_metric_snapshots` populates after outcomes exist
|
||||
- Scheduler logs show significantly fewer "Rate limit hit" warnings
|
||||
- Aggregation logs no longer show "column value does not exist" error
|
||||
- After market close: `SELECT * FROM portfolio_snapshots WHERE snapshot_date = CURRENT_DATE` returns 1 row
|
||||
|
||||
## Glossary
|
||||
|
||||
| Term | Definition |
|
||||
|------|-----------|
|
||||
| Validation cycle | Hourly scheduler task: evaluate_matured_predictions → compute_and_store_metric_snapshots |
|
||||
| Quality gate | Threshold check on model_metric_snapshots that determines if trading can be promoted from paper to live |
|
||||
| Prediction snapshot | Frozen state of a recommendation at generation time (prices, evidence, scores) |
|
||||
| Outcome evaluation | Matching a matured prediction snapshot against realized market returns |
|
||||
| Polygon free tier | API plan with ~5 requests/minute rate limit |
|
||||
@@ -0,0 +1,69 @@
|
||||
# Implementation Plan: ops-pipeline-fixes
|
||||
|
||||
## Overview
|
||||
|
||||
Fix 10 operational bugs preventing the validation/calibration feedback loop from functioning and degrading ingestion throughput. Changes span scheduler (rate limiting + periodic tasks), aggregation worker (config query), broker service (rejection reason), prediction snapshot (price fallback), and Helm chart (dead pod removal).
|
||||
|
||||
## Tasks
|
||||
|
||||
- [x] 1. Fix Polygon global rate limit — In `services/scheduler/app.py`, replace `POLYGON_GLOBAL_RATE_LIMIT: int = 45` with `POLYGON_GLOBAL_RATE_LIMIT: int = int(os.getenv("POLYGON_GLOBAL_RATE_LIMIT", "5"))` to make it env-configurable and default to the free-tier limit
|
||||
- **Validates: Bugfix 2.1; Regression 3.1, 3.7**
|
||||
|
||||
- [x] 2. Fix v3_engine_enabled query — In `services/aggregation/worker.py`, replace `_V3_ENGINE_FLAG_QUERY` from `SELECT value FROM risk_configs WHERE key = 'v3_engine_enabled'` to `SELECT config->>'v3_engine_enabled' AS enabled FROM risk_configs WHERE name = 'default' AND active = TRUE LIMIT 1`, and rewrite `_read_v3_flag()` to parse the returned string (checking for "true"/"1"/"yes"), returning False for NULL/missing/error
|
||||
- **Validates: Bugfix 2.2; Regression 3.2**
|
||||
|
||||
- [x] 3. Add validation cycle constant and counter — In `services/scheduler/app.py`, add `VALIDATION_CYCLE_INTERVAL = int(os.getenv("VALIDATION_CYCLE_INTERVAL", "240"))` constant and `validation_counter = 0` initialization in `main()`
|
||||
- **Validates: Bugfix 2.3, 2.4**
|
||||
|
||||
- [x] 4. Implement run_validation_cycle function — In `services/scheduler/app.py`, implement `run_validation_cycle(pool)` that calls `evaluate_matured_predictions(pool)` followed by `compute_and_store_metric_snapshots(pool)`, with try/except logging for each and skipping metrics if outcomes fail
|
||||
- **Validates: Bugfix 2.3, 2.4, 2.5; Regression 3.3**
|
||||
|
||||
- [x] 5. Wire validation cycle into main loop — In `services/scheduler/app.py` main loop, add the counter increment and conditional call to `run_validation_cycle(pool)` after the existing `report_schedule_counter` block
|
||||
- **Validates: Bugfix 2.3, 2.4, 2.5**
|
||||
|
||||
- [x] 6. Add snapshot cycle constant and counter — In `services/scheduler/app.py`, add `SNAPSHOT_CYCLE_INTERVAL = int(os.getenv("SNAPSHOT_CYCLE_INTERVAL", "240"))` constant and `snapshot_counter = 0` initialization in `main()`
|
||||
- **Validates: Bugfix 2.6, 2.7**
|
||||
|
||||
- [x] 7. Implement maybe_capture_daily_snapshots function — In `services/scheduler/app.py`, implement `maybe_capture_daily_snapshots(pool)` that checks time (after 16:30 ET), checks idempotency (no existing row for today), queries positions table for portfolio value/unrealized PnL, and inserts into `portfolio_snapshots` and `daily_risk_snapshots`
|
||||
- **Validates: Bugfix 2.6, 2.7; Regression 3.10**
|
||||
|
||||
- [x] 8. Wire snapshot cycle into main loop — In `services/scheduler/app.py` main loop, add the counter increment and conditional call to `maybe_capture_daily_snapshots(pool)` after the validation counter block
|
||||
- **Validates: Bugfix 2.6, 2.7**
|
||||
|
||||
- [x] 9. Add prediction price fallback — In `services/validation/prediction_snapshot.py`, after the primary market_snapshots price lookup returns NULL for `price_at_prediction`, add a fallback query to positions table: `SELECT current_price FROM positions WHERE ticker = $1 AND current_price IS NOT NULL LIMIT 1`
|
||||
- **Validates: Bugfix 2.8; Regression 3.4**
|
||||
|
||||
- [x] 10. Scale down lake-publisher — In `infra/helm/stonks-oracle/values.yaml`, change the lake-publisher `replicas` from `1` to `0`
|
||||
- **Validates: Bugfix 2.9; Regression 3.5**
|
||||
|
||||
- [x] 11. Extend _INSERT_ORDER SQL — In `services/adapters/broker_service.py`, extend `_INSERT_ORDER` SQL to include `rejection_reason` and `rejected_at` as parameters $18 and $19, with COALESCE in the ON CONFLICT UPDATE clause to preserve existing values
|
||||
- **Validates: Bugfix 2.10; Regression 3.6, 3.8**
|
||||
|
||||
- [x] 12. Update persist_order parameters — In `services/adapters/broker_service.py`, update `persist_order()` to compute `rejection_reason = resp.error if resp.status == OrderStatus.REJECTED else None` and `rejected_at = now if resp.status == OrderStatus.REJECTED else None`, passing them as the final two parameters in the execute call
|
||||
- **Validates: Bugfix 2.10; Regression 3.6, 3.8**
|
||||
|
||||
- [x] 13. Lint and test — Run `.venv/bin/ruff check services/` and `.venv/bin/python -m pytest tests/ -x --tb=short -q` to verify no regressions
|
||||
- **Validates: Regression 3.1–3.10**
|
||||
|
||||
## Task Dependency Graph
|
||||
|
||||
```json
|
||||
{
|
||||
"waves": [
|
||||
{"tasks": [1, 2, 9, 10]},
|
||||
{"tasks": [3, 6, 11]},
|
||||
{"tasks": [4, 7, 12]},
|
||||
{"tasks": [5, 8]},
|
||||
{"tasks": [13]}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
Tasks 1, 2, 9, 10 are fully independent. Tasks 3/6/11 set up constants needed by 4/7/12. Tasks 5/8 wire into the main loop after their functions exist. Task 13 validates everything last.
|
||||
|
||||
## Notes
|
||||
|
||||
- No database migrations required — all tables already exist with correct columns
|
||||
- All scheduler changes use the existing counter-based periodic task pattern already established for cleanup, aggregation, and report tasks
|
||||
- Lazy imports in `run_validation_cycle` avoid circular imports and keep scheduler startup fast
|
||||
- The `maybe_capture_daily_snapshots` idempotency check prevents duplicate rows on scheduler restart
|
||||
@@ -0,0 +1 @@
|
||||
{"specId": "f5d99301-94ef-4dc2-8ba4-ccefeee7ecba", "workflowType": "requirements-first", "specType": "bugfix"}
|
||||
@@ -0,0 +1,51 @@
|
||||
# Bugfix Requirements Document
|
||||
|
||||
## Introduction
|
||||
|
||||
Five operational bugs in the stonks-beta deployment degrade pipeline health: 1,809 documents stuck in `parsed` status due to recovery batch limits, 26.5% of prediction snapshots missing prices due to incomplete fallback chains, 64% sell bias from uncalibrated NuExtract3 sentiment outputs, idle signal-engine consuming resources while doing nothing, and a quality gate stuck in paper-only mode due to an overly strict staleness threshold interacting with the NULL price problem.
|
||||
|
||||
## Bug Analysis
|
||||
|
||||
### Current Behavior (Defect)
|
||||
|
||||
1.1 WHEN the `recover_stale_documents` task runs with 1,809+ documents stuck in `parsed` status THEN the system only processes 100 per cycle (every ~5 minutes), requiring 90+ cycles (~7.5 hours) to clear the backlog while new documents may continue accumulating
|
||||
|
||||
1.2 WHEN a prediction snapshot is created for a ticker without an open position AND without recent market_snapshots data THEN the system stores NULL in `price_at_prediction` because the fallback chain stops at the positions table (26.5% of snapshots affected — 33,324 of 125,590)
|
||||
|
||||
1.3 WHEN the outcome evaluator encounters a prediction snapshot with NULL `price_at_prediction` THEN the system skips the snapshot entirely, creating a validation blind spot where 26.5% of predictions are never evaluated
|
||||
|
||||
1.4 WHEN the aggregation pipeline processes NuExtract3 extraction outputs THEN the system passes raw `impact_score` and `sentiment` values directly into signal weighting without any distribution normalization, resulting in systematic negative bias producing 64% sell / 23% watch / 12% buy recommendations
|
||||
|
||||
1.5 WHEN the signal-engine pod starts with `dual_pipeline_enabled=False` THEN the system enters an infinite sleep loop consuming CPU (100m request / 500m limit) and memory (128Mi request / 256Mi limit) while producing zero signal evaluations
|
||||
|
||||
1.6 WHEN the quality gate checks `model_metric_snapshots` freshness with a 24-hour staleness threshold AND the validation cycle skips all predictions due to NULL prices (Bug 1.2/1.3) THEN the system permanently defaults to paper-only mode because no fresh metric snapshots are ever generated
|
||||
|
||||
### Expected Behavior (Correct)
|
||||
|
||||
2.1 WHEN the scheduler detects more than 100 documents stuck in `parsed` status older than the threshold THEN the system SHALL increase the batch limit for recovery processing (up to 500 per cycle) and provide a one-time management command to bulk-recover the existing backlog without waiting for periodic sweeps
|
||||
|
||||
2.2 WHEN a prediction snapshot is created and no price is available from market_snapshots (exact time) or positions table THEN the system SHALL query `market_snapshots` with a wider time window (last 24 hours of bar data for the ticker) as an additional fallback before accepting NULL
|
||||
|
||||
2.3 WHEN backfilling existing prediction snapshots with NULL `price_at_prediction` THEN the system SHALL use the extended fallback chain (market_snapshots within 24h of `generated_at`, then positions) to populate prices retroactively via a migration script
|
||||
|
||||
2.4 WHEN the aggregation pipeline computes signal weights from impact records THEN the system SHALL apply z-score normalization to `impact_score` values relative to the rolling 7-day distribution of impact records for the same ticker, preventing systematic model bias from dominating the directional signal
|
||||
|
||||
2.5 WHEN the signal-engine deployment is not ready for production use (`dual_pipeline_enabled=False`) THEN the system SHALL be scaled to 0 replicas in the Helm values files (beta, paper, live) to eliminate wasted CPU, memory, and any GPU time-slice allocations
|
||||
|
||||
2.6 WHEN the quality gate evaluates metric snapshot freshness during the bootstrapping period THEN the system SHALL use a 48-hour staleness threshold (instead of 24h) to tolerate gaps while the validation cycle ramps up after Bug 1.2/1.3 are fixed
|
||||
|
||||
### Unchanged Behavior (Regression Prevention)
|
||||
|
||||
3.1 WHEN documents enter `parsed` status and are processed within the normal threshold window (< 240 minutes) THEN the system SHALL CONTINUE TO leave them for the extraction queue consumer without interference from the recovery task
|
||||
|
||||
3.2 WHEN a prediction snapshot is created and market_snapshots contains a recent bar for the ticker THEN the system SHALL CONTINUE TO use the primary `market_snapshots` close price without invoking any fallback
|
||||
|
||||
3.3 WHEN the aggregation pipeline processes tickers with balanced sentiment distributions (equal bullish/bearish evidence) THEN the system SHALL CONTINUE TO produce neutral/mixed recommendations without artificial skew from the normalization step
|
||||
|
||||
3.4 WHEN the signal-engine is re-enabled in the future (dual_pipeline_enabled=True with replicas > 0) THEN the system SHALL CONTINUE TO function correctly with its existing queue-based architecture and configuration loading
|
||||
|
||||
3.5 WHEN the quality gate evaluates a metric snapshot that is less than 48 hours old and meets all threshold criteria THEN the system SHALL CONTINUE TO promote recommendations to live_eligible mode per existing threshold logic
|
||||
|
||||
3.6 WHEN the outcome evaluator processes prediction snapshots with valid (non-NULL) prices THEN the system SHALL CONTINUE TO evaluate them normally and produce prediction_outcomes records
|
||||
|
||||
3.7 WHEN the `retry_failed_extractions` task handles documents in `extraction_failed` status THEN the system SHALL CONTINUE TO process them on the existing cadence and logic without interference from the parsed-document recovery changes
|
||||
@@ -0,0 +1,320 @@
|
||||
# Pipeline Health Fixes — Bugfix Design
|
||||
|
||||
## Overview
|
||||
|
||||
Five operational bugs degrade stonks-beta pipeline health. This design formalizes the bug conditions, expected fixes, and validation strategy for each:
|
||||
|
||||
1. **Stuck Parsed Docs** — `recover_stale_documents()` batch limit of 100 is too low for 1,809 stuck documents; increase to 500 and lower the stale threshold to 30 minutes.
|
||||
2. **Extended Price Fallback** — Prediction snapshots missing prices (26.5%) because the fallback chain stops at `positions`; add a third fallback querying `market_snapshots` within 24h.
|
||||
3. **Sentiment Z-Score Normalization** — Raw NuExtract3 `impact_score` values produce 64% sell bias; normalize using 7-day rolling z-scores per ticker before signal weighting.
|
||||
4. **Signal Engine Scale Down** — Idle signal-engine pods consume resources; set replicas to 0 in all Helm values files.
|
||||
5. **Quality Gate Threshold** — 24h staleness threshold permanently locks quality gate to paper-only; relax to 48h.
|
||||
|
||||
## Glossary
|
||||
|
||||
- **Bug_Condition (C)**: The specific conditions under which each bug manifests
|
||||
- **Property (P)**: The desired correct behavior after the fix is applied
|
||||
- **Preservation**: Existing behavior that must remain unchanged after the fix
|
||||
- **`recover_stale_documents()`**: Function in `services/scheduler/app.py` that re-enqueues documents stuck in `parsed` status
|
||||
- **`STALE_PARSED_THRESHOLD_MINUTES`**: Constant (currently 240) controlling how long a document must be stuck before recovery
|
||||
- **`fetch_latest_close_price()`**: Function in `services/validation/prediction_snapshot.py` that queries `market_snapshots` for the most recent bar
|
||||
- **`compute_signal_weight()`**: Function in `services/aggregation/scoring.py` that computes combined signal weight from recency, credibility, novelty, confidence, and impact
|
||||
- **`QualityGateConfig.max_snapshot_age_hours`**: Threshold in `services/trading/model_quality_gate.py` controlling when the quality gate defaults to paper-only
|
||||
|
||||
## Bug Details
|
||||
|
||||
### Bug Condition
|
||||
|
||||
The pipeline health degradation manifests across five independent conditions:
|
||||
|
||||
**Formal Specification:**
|
||||
```
|
||||
FUNCTION isBugCondition(input)
|
||||
INPUT: input of type PipelineState
|
||||
OUTPUT: boolean
|
||||
|
||||
-- Bug 1: Parsed docs stuck beyond batch capacity
|
||||
RETURN (input.stuckParsedDocCount > 100
|
||||
AND input.recoveryBatchLimit == 100
|
||||
AND input.docStaleMinutes >= 240)
|
||||
-- Bug 2: Price fallback chain incomplete
|
||||
OR (input.tickerPrice IS NULL
|
||||
AND input.positionPrice IS NULL
|
||||
AND input.marketSnapshotWithin24h IS NOT NULL)
|
||||
-- Bug 3: Raw impact scores without normalization
|
||||
OR (input.impactScoreUsedRaw == TRUE
|
||||
AND input.ticker7dStddev > 0)
|
||||
-- Bug 4: Signal engine running idle
|
||||
OR (input.signalEngineReplicas > 0
|
||||
AND input.dualPipelineEnabled == FALSE)
|
||||
-- Bug 5: Quality gate threshold too strict
|
||||
OR (input.snapshotAgeHours > 24
|
||||
AND input.snapshotAgeHours <= 48
|
||||
AND input.maxSnapshotAgeConfig == 24)
|
||||
END FUNCTION
|
||||
```
|
||||
|
||||
### Examples
|
||||
|
||||
- **Bug 1**: 1,809 documents in `parsed` status older than 4 hours. At 100/cycle every 5 minutes, clearing takes 90+ cycles (~7.5h). With 500/batch, it takes 4 cycles (~20 min).
|
||||
- **Bug 2**: Ticker PLTR has no open position and `fetch_latest_close_price` returns NULL, but `market_snapshots` has a bar from 3 hours ago that could serve as price.
|
||||
- **Bug 3**: NuExtract3 outputs `impact_score` values clustered around -0.3 to -0.1 for a ticker. Without normalization, `weighted_sentiment_average()` systematically produces negative signals → 64% sell recommendations.
|
||||
- **Bug 4**: signal-engine pod starts, detects `dual_pipeline_enabled=False`, enters infinite sleep loop consuming 100m CPU request / 128Mi memory request.
|
||||
- **Bug 5**: Quality gate reads `model_metric_snapshots`, finds the most recent is 26h old (because validation skips NULL-price predictions), fails staleness check, forces paper-only mode permanently.
|
||||
|
||||
## Expected Behavior
|
||||
|
||||
### Preservation Requirements
|
||||
|
||||
**Unchanged Behaviors:**
|
||||
- Documents entering `parsed` status and processed within the normal threshold window (< 30 min after fix) are left alone for the extraction queue consumer
|
||||
- Primary price lookup via `fetch_latest_close_price()` (exact time match from `market_snapshots`) continues as the first-choice price source
|
||||
- Tickers with balanced sentiment distributions continue to produce neutral/mixed recommendations without artificial skew
|
||||
- Signal-engine's queue-based architecture and configuration loading remain functional when re-enabled with replicas > 0
|
||||
- Quality gate threshold logic for snapshots younger than 48h and meeting all criteria continues to promote to `live_eligible`
|
||||
- Outcome evaluator continues to process prediction snapshots with valid (non-NULL) prices normally
|
||||
- `retry_failed_extractions` task continues on its existing cadence without interference
|
||||
|
||||
**Scope:**
|
||||
All inputs that do NOT match the bug conditions above should be completely unaffected by these fixes. The fixes are additive (new fallback path, wider batch, normalization layer) or config-only (replica count, threshold constant).
|
||||
|
||||
## Hypothesized Root Cause
|
||||
|
||||
### Bug 1: Stuck Parsed Docs
|
||||
- **Batch limit too small**: The `LIMIT 100` in the SQL query caps recovery throughput at 100 docs per scheduler cycle (~5 min). When a Redis crash orphans thousands of documents, the recovery rate cannot keep up with the backlog.
|
||||
- **Threshold too conservative**: `STALE_PARSED_THRESHOLD_MINUTES = 240` (4 hours) means documents must be stuck for 4 hours before recovery kicks in. A 30-minute threshold would catch orphans much faster.
|
||||
|
||||
### Bug 2: Incomplete Fallback Chain
|
||||
- **Missing time-window query**: `fetch_latest_close_price()` only checks `market_snapshots` for an exact timestamp match. When market data ingestion is delayed or the prediction happens outside market hours, no exact match exists.
|
||||
- **Positions-only fallback**: The positions table fallback only works for tickers with an active position. 26.5% of snapshots are for tickers without positions.
|
||||
|
||||
### Bug 3: Uncalibrated Impact Scores
|
||||
- **No distribution normalization**: NuExtract3 model outputs are passed raw into `compute_signal_weight()` via `impact_score` parameter. The model has a systematic negative bias in its output distribution that is not corrected.
|
||||
- **Per-ticker variance ignored**: Different tickers receive different volume/types of news, producing different impact_score distributions. A global normalization would be insufficient.
|
||||
|
||||
### Bug 4: Idle Signal Engine
|
||||
- **Replicas set to 1 by default**: `values.yaml` defines `signalEngine.replicas: 1` regardless of whether the dual pipeline feature is enabled. The pod starts, detects the feature is off, and sleeps forever.
|
||||
|
||||
### Bug 5: Overly Strict Staleness
|
||||
- **24h threshold too tight during bootstrapping**: The `max_snapshot_age_hours = 24` default assumes the validation cycle runs frequently. When Bug 2/3 cause most predictions to be skipped, metric snapshots aren't generated, and the 24h window expires.
|
||||
|
||||
## Correctness Properties
|
||||
|
||||
Property 1: Bug Condition — Stuck Parsed Docs Recovery
|
||||
|
||||
_For any_ set of documents stuck in `parsed` status longer than 30 minutes, the fixed `recover_stale_documents()` function SHALL process up to 500 documents per cycle, reducing backlog clearance time by 5x compared to the previous 100-document limit.
|
||||
|
||||
**Validates: Requirements 2.1**
|
||||
|
||||
Property 2: Bug Condition — Extended Price Fallback
|
||||
|
||||
_For any_ prediction snapshot where `fetch_latest_close_price()` returns NULL and the `positions` table has no price, but `market_snapshots` contains a bar for the ticker within 24 hours of the prediction time, the fixed code SHALL use that bar's close price as `price_at_prediction`.
|
||||
|
||||
**Validates: Requirements 2.2, 2.3**
|
||||
|
||||
Property 3: Bug Condition — Sentiment Z-Score Normalization
|
||||
|
||||
_For any_ set of `document_impact_records` for a ticker, the fixed aggregation pipeline SHALL normalize `impact_score` values using the 7-day rolling mean and standard deviation for that ticker before passing them into signal weight computation, preventing systematic model bias.
|
||||
|
||||
**Validates: Requirements 2.4**
|
||||
|
||||
Property 4: Bug Condition — Signal Engine Scale Down
|
||||
|
||||
_For any_ Helm deployment where `dual_pipeline_enabled=False`, the fixed Helm values SHALL specify `signalEngine.replicas: 0`, preventing the pod from being scheduled and consuming resources.
|
||||
|
||||
**Validates: Requirements 2.5**
|
||||
|
||||
Property 5: Bug Condition — Quality Gate Threshold
|
||||
|
||||
_For any_ model metric snapshot that is between 24h and 48h old, the fixed quality gate SHALL NOT reject it as stale, allowing the system to remain in non-paper mode during the bootstrapping period.
|
||||
|
||||
**Validates: Requirements 2.6**
|
||||
|
||||
Property 6: Preservation — Normal Document Processing
|
||||
|
||||
_For any_ document that enters `parsed` status and is processed within 30 minutes, the fixed `recover_stale_documents()` function SHALL NOT interfere with normal extraction queue processing, preserving the existing pipeline flow.
|
||||
|
||||
**Validates: Requirements 3.1, 3.7**
|
||||
|
||||
Property 7: Preservation — Primary Price Path
|
||||
|
||||
_For any_ prediction snapshot where `fetch_latest_close_price()` returns a valid price, the fixed code SHALL use that price directly without invoking any fallback, preserving the primary price lookup behavior.
|
||||
|
||||
**Validates: Requirements 3.2**
|
||||
|
||||
Property 8: Preservation — Balanced Sentiment
|
||||
|
||||
_For any_ ticker with a balanced sentiment distribution (equal bullish/bearish evidence), the z-score normalization SHALL produce values centered around 0, preserving neutral/mixed recommendation output without artificial skew.
|
||||
|
||||
**Validates: Requirements 3.3**
|
||||
|
||||
Property 9: Preservation — Quality Gate Valid Snapshots
|
||||
|
||||
_For any_ model metric snapshot younger than 48h that meets all threshold criteria, the fixed quality gate SHALL continue to promote recommendations to `live_eligible` mode per existing logic.
|
||||
|
||||
**Validates: Requirements 3.5, 3.6**
|
||||
|
||||
## Fix Implementation
|
||||
|
||||
### Changes Required
|
||||
|
||||
**Bug 1: Stuck Parsed Docs Recovery**
|
||||
|
||||
**File**: `services/scheduler/app.py`
|
||||
|
||||
**Function**: `recover_stale_documents()`
|
||||
|
||||
**Specific Changes**:
|
||||
1. **Lower stale threshold**: Change `STALE_PARSED_THRESHOLD_MINUTES` from `240` to `30` — documents stuck longer than 30 minutes are likely orphaned
|
||||
2. **Increase batch limit**: Change `LIMIT 100` to `LIMIT 500` in the SQL query
|
||||
3. **Update enqueued TTL**: Change `_ENQUEUED_TTL` from `14400` (4h) to `3600` (1h) to match the new threshold
|
||||
|
||||
---
|
||||
|
||||
**Bug 2: Extended Price Fallback**
|
||||
|
||||
**File**: `services/validation/prediction_snapshot.py`
|
||||
|
||||
**Function**: `create_prediction_snapshot()`
|
||||
|
||||
**Specific Changes**:
|
||||
1. **Add market_snapshots time-window fallback**: After the positions fallback fails, query `market_snapshots` for the most recent bar within 24h of the current time for the ticker
|
||||
2. **SQL query**: `SELECT close FROM market_snapshots WHERE ticker = $1 AND timestamp >= NOW() - INTERVAL '24 hours' ORDER BY timestamp DESC LIMIT 1`
|
||||
3. **Log the fallback**: Add info-level logging when the extended fallback is used
|
||||
|
||||
**New File**: `scripts/backfill_snapshot_prices.py`
|
||||
|
||||
**Purpose**: One-time backfill script to populate `price_at_prediction` for existing NULL snapshots using the extended fallback chain.
|
||||
|
||||
**Approach**:
|
||||
1. Query all `prediction_snapshots` where `price_at_prediction IS NULL`
|
||||
2. For each, attempt: `market_snapshots` within 24h of `generated_at`, then `positions` table
|
||||
3. Update the row with the found price
|
||||
4. Report statistics (found via market_snapshots, found via positions, still NULL)
|
||||
|
||||
---
|
||||
|
||||
**Bug 3: Sentiment Z-Score Normalization**
|
||||
|
||||
**File**: `services/aggregation/worker.py` (or new helper in `services/aggregation/scoring.py`)
|
||||
|
||||
**Function**: New function `normalize_impact_scores()` called before `compute_signal_weight()`
|
||||
|
||||
**Specific Changes**:
|
||||
1. **Add normalization function**: Compute 7-day rolling mean and stddev of `impact_score` per ticker from `document_impact_records`
|
||||
2. **Formula**: `normalized = (raw - mean_7d) / max(stddev_7d, 0.1)` — the 0.1 floor prevents division by near-zero stddev for low-activity tickers
|
||||
3. **Integration point**: In the aggregation loop (around line 440 of worker.py), normalize `imp.impact_score` before passing to `compute_signal_weight()` and `WeightedSignal`
|
||||
4. **Fallback**: If fewer than 5 records exist in the 7-day window, use the raw score (insufficient data for meaningful normalization)
|
||||
5. **Query**: `SELECT AVG(impact_score) as mean, STDDEV(impact_score) as stddev FROM document_impact_records WHERE ticker = $1 AND created_at >= NOW() - INTERVAL '7 days'`
|
||||
|
||||
---
|
||||
|
||||
**Bug 4: Signal Engine Scale Down**
|
||||
|
||||
**Files**: `infra/helm/stonks-oracle/values.yaml`, `values-beta.yaml`, `values-paper.yaml`
|
||||
|
||||
**Specific Changes**:
|
||||
1. **values.yaml**: Change `signalEngine.replicas` from `1` to `0`
|
||||
2. **values-beta.yaml**: Add `signalEngine.replicas: 0` under `services:`
|
||||
3. **values-paper.yaml**: Add `signalEngine.replicas: 0` under `services:`
|
||||
|
||||
---
|
||||
|
||||
**Bug 5: Quality Gate Threshold**
|
||||
|
||||
**File**: `services/trading/model_quality_gate.py`
|
||||
|
||||
**Class**: `QualityGateConfig`
|
||||
|
||||
**Specific Changes**:
|
||||
1. **Change default**: `max_snapshot_age_hours: int = 48` (was 24)
|
||||
|
||||
## Testing Strategy
|
||||
|
||||
### Validation Approach
|
||||
|
||||
The testing strategy follows a two-phase approach: first, surface counterexamples that demonstrate the bug on unfixed code, then verify the fix works correctly and preserves existing behavior.
|
||||
|
||||
### Exploratory Bug Condition Checking
|
||||
|
||||
**Goal**: Surface counterexamples that demonstrate the bugs BEFORE implementing the fixes. Confirm or refute the root cause analysis.
|
||||
|
||||
**Test Plan**: Write tests that exercise each bug condition on the unfixed code to observe failures.
|
||||
|
||||
**Test Cases**:
|
||||
1. **Batch Overflow Test**: Create 600 documents in `parsed` status older than threshold, run `recover_stale_documents()`, assert only 100 are processed (will demonstrate Bug 1)
|
||||
2. **Price Fallback Gap Test**: Call `create_prediction_snapshot()` for a ticker with no position and no exact market_snapshots match, assert `price_at_prediction` is NULL (will demonstrate Bug 2)
|
||||
3. **Sentiment Bias Test**: Generate 50 impact records with systematic negative bias (mean=-0.3, stddev=0.1), compute weighted signals, assert directional signal is negative (will demonstrate Bug 3)
|
||||
4. **Quality Gate Staleness Test**: Set most recent metric snapshot to 26h ago, evaluate quality gate, assert it fails (will demonstrate Bug 5)
|
||||
|
||||
**Expected Counterexamples**:
|
||||
- Bug 1: Only 100 of 600 documents recovered per cycle
|
||||
- Bug 2: `price_at_prediction` stored as NULL despite market data existing within 24h
|
||||
- Bug 3: Weighted sentiment average heavily negative despite mixed underlying events
|
||||
- Bug 5: Quality gate returns `passed=False` with reason containing "stale"
|
||||
|
||||
### Fix Checking
|
||||
|
||||
**Goal**: Verify that for all inputs where the bug condition holds, the fixed function produces the expected behavior.
|
||||
|
||||
**Pseudocode:**
|
||||
```
|
||||
FOR ALL input WHERE isBugCondition(input) DO
|
||||
result := fixedFunction(input)
|
||||
ASSERT expectedBehavior(result)
|
||||
END FOR
|
||||
```
|
||||
|
||||
**Per-bug fix checks:**
|
||||
- Bug 1: `recover_stale_documents()` processes up to 500 docs with 30-min threshold
|
||||
- Bug 2: Extended fallback returns a price when `market_snapshots` has data within 24h
|
||||
- Bug 3: Normalized impact scores have mean ≈ 0 and stddev ≈ 1 for active tickers
|
||||
- Bug 4: `kubectl get pods` shows 0 signal-engine pods
|
||||
- Bug 5: Quality gate passes for snapshots 24–48h old that meet metric thresholds
|
||||
|
||||
### Preservation Checking
|
||||
|
||||
**Goal**: Verify that for all inputs where the bug condition does NOT hold, the fixed function produces the same result as the original function.
|
||||
|
||||
**Pseudocode:**
|
||||
```
|
||||
FOR ALL input WHERE NOT isBugCondition(input) DO
|
||||
ASSERT originalFunction(input) = fixedFunction(input)
|
||||
END FOR
|
||||
```
|
||||
|
||||
**Testing Approach**: Property-based testing is recommended for preservation checking because:
|
||||
- It generates many test cases automatically across the input domain
|
||||
- It catches edge cases that manual unit tests might miss
|
||||
- It provides strong guarantees that behavior is unchanged for all non-buggy inputs
|
||||
|
||||
**Test Plan**: Observe behavior on UNFIXED code first for normal inputs, then write property-based tests capturing that behavior.
|
||||
|
||||
**Test Cases**:
|
||||
1. **Normal Doc Processing Preservation**: Documents < 30 min old are never touched by recovery
|
||||
2. **Primary Price Preservation**: When `fetch_latest_close_price()` succeeds, no fallback is invoked
|
||||
3. **Balanced Sentiment Preservation**: Tickers with symmetric impact_score distributions produce neutral signals after normalization
|
||||
4. **Quality Gate Normal Preservation**: Snapshots < 48h old and meeting thresholds still pass
|
||||
5. **Failed Extraction Preservation**: `retry_failed_extractions()` behavior unchanged
|
||||
|
||||
### Unit Tests
|
||||
|
||||
- Test `recover_stale_documents()` with various document counts (0, 50, 500, 1000)
|
||||
- Test extended price fallback with market_snapshots at various time offsets (1h, 12h, 23h, 25h)
|
||||
- Test z-score normalization with known distributions (mean=0, mean=-0.5, stddev=0, stddev=0.05)
|
||||
- Test quality gate with snapshot ages at boundary (23h, 24h, 47h, 48h, 49h)
|
||||
- Test backfill script with mixed NULL/non-NULL snapshots
|
||||
|
||||
### Property-Based Tests
|
||||
|
||||
- Generate random document ages and counts, verify recovery processes correct subset (> 30 min old, up to 500)
|
||||
- Generate random ticker price scenarios, verify fallback chain ordering is preserved (primary → positions → market_snapshots_24h → NULL)
|
||||
- Generate random impact_score distributions per ticker, verify normalized output has bounded variance and zero-centered mean
|
||||
- Generate random snapshot ages, verify quality gate accepts [0, 48h) and rejects [48h, ∞)
|
||||
|
||||
### Integration Tests
|
||||
|
||||
- End-to-end: create documents in `parsed` status, run scheduler cycle, verify extraction queue populated
|
||||
- End-to-end: create prediction snapshot for ticker without position, verify price populated from market_snapshots
|
||||
- End-to-end: run full aggregation cycle with biased NuExtract3 outputs, verify recommendation direction is not systematically biased
|
||||
- Helm template render: verify signal-engine deployment has 0 replicas in all value files
|
||||
@@ -0,0 +1,174 @@
|
||||
# Implementation Plan
|
||||
|
||||
## Overview
|
||||
|
||||
Bugfix implementation for five pipeline health issues: stuck parsed docs, missing price fallback, uncalibrated sentiment scores, idle signal-engine pods, and overly strict quality gate threshold. Tasks follow the exploratory bugfix workflow: explore bugs via tests, preserve existing behavior, implement fixes, validate.
|
||||
|
||||
## Tasks
|
||||
|
||||
- [x] 1. Write bug condition exploration test
|
||||
- **Property 1: Bug Condition** - Pipeline Health Degradation
|
||||
- **CRITICAL**: This test MUST FAIL on unfixed code - failure confirms the bugs exist
|
||||
- **DO NOT attempt to fix the test or the code when it fails**
|
||||
- **NOTE**: This test encodes the expected behavior - it will validate the fix when it passes after implementation
|
||||
- **GOAL**: Surface counterexamples that demonstrate all five bugs exist
|
||||
- **Scoped PBT Approach**: Scope properties to the concrete failing cases for each bug condition
|
||||
- Test file: `tests/test_pbt_pipeline_health_bug_condition.py`
|
||||
- **Bug 1 - Batch Overflow**: Create 600 documents in `parsed` status older than 30 min, run `recover_stale_documents()`, assert up to 500 are recovered per cycle (will FAIL on unfixed code which caps at 100)
|
||||
- **Bug 2 - Price Fallback Gap**: Call `create_prediction_snapshot()` for ticker with no position and no exact market_snapshots match but data within 24h exists, assert `price_at_prediction` is NOT NULL (will FAIL on unfixed code which returns NULL)
|
||||
- **Bug 3 - Sentiment Bias**: Generate 50 impact records with systematic negative bias (mean=-0.3, stddev=0.1), compute weighted signals via aggregation, assert normalized output is zero-centered (will FAIL on unfixed code which passes raw scores)
|
||||
- **Bug 5 - Quality Gate Staleness**: Set most recent metric snapshot to 26h ago, evaluate quality gate, assert it passes (will FAIL on unfixed code which rejects at 24h)
|
||||
- Run tests on UNFIXED code
|
||||
- **EXPECTED OUTCOME**: Tests FAIL (this is correct - it proves the bugs exist)
|
||||
- Document counterexamples: batch capped at 100, price stored as NULL, sentiment heavily negative, quality gate returns `passed=False`
|
||||
- Mark task complete when tests are written, run, and failures are documented
|
||||
- _Requirements: 1.1, 1.2, 1.3, 1.4, 1.6_
|
||||
|
||||
- [x] 2. Write preservation property tests (BEFORE implementing fix)
|
||||
- **Property 2: Preservation** - Pipeline Behavior Unchanged for Non-Bug Inputs
|
||||
- **IMPORTANT**: Follow observation-first methodology
|
||||
- Test file: `tests/test_pbt_pipeline_health_preservation.py`
|
||||
- **Normal Doc Processing**: Observe that documents < 30 min old are never touched by `recover_stale_documents()` on unfixed code. Write property: for all documents with age < 30 min, recovery task does NOT enqueue them.
|
||||
- **Primary Price Path**: Observe that when `fetch_latest_close_price()` returns a valid price, no fallback is invoked. Write property: for all tickers where primary price exists, result equals primary price.
|
||||
- **Balanced Sentiment**: Observe that tickers with symmetric impact_score distributions (mean ≈ 0) produce neutral signals. Write property: for all impact_score sets with mean ≈ 0, normalized output remains centered around 0.
|
||||
- **Quality Gate Normal**: Observe that snapshots < 48h old meeting thresholds pass the quality gate. Write property: for all snapshot ages in [0, 48h) meeting metric criteria, quality gate returns `passed=True`.
|
||||
- **Failed Extraction Independence**: Observe `retry_failed_extractions()` behavior is unaffected. Write property: for all documents in `extraction_failed` status, retry logic unchanged.
|
||||
- Run tests on UNFIXED code
|
||||
- **EXPECTED OUTCOME**: Tests PASS (this confirms baseline behavior to preserve)
|
||||
- Mark task complete when tests are written, run, and passing on unfixed code
|
||||
- _Requirements: 3.1, 3.2, 3.3, 3.5, 3.6, 3.7_
|
||||
|
||||
- [x] 3. Fix: Signal Engine Scale Down (Helm values)
|
||||
|
||||
- [x] 3.1 Set signal-engine replicas to 0 in all Helm values files
|
||||
- In `infra/helm/stonks-oracle/values.yaml`: change `signalEngine.replicas` from `1` to `0`
|
||||
- In `infra/helm/stonks-oracle/values-beta.yaml`: add/set `signalEngine.replicas: 0` under `services:`
|
||||
- In `infra/helm/stonks-oracle/values-paper.yaml`: add/set `signalEngine.replicas: 0` under `services:`
|
||||
- _Bug_Condition: input.signalEngineReplicas > 0 AND input.dualPipelineEnabled == FALSE_
|
||||
- _Expected_Behavior: signalEngine.replicas == 0 when dual pipeline disabled_
|
||||
- _Preservation: Signal-engine architecture remains functional when re-enabled with replicas > 0_
|
||||
- _Requirements: 2.5, 3.4_
|
||||
|
||||
- [x] 4. Fix: Quality Gate Threshold Relaxation
|
||||
|
||||
- [x] 4.1 Change max_snapshot_age_hours default from 24 to 48
|
||||
- File: `services/trading/model_quality_gate.py`
|
||||
- In `QualityGateConfig` class, change `max_snapshot_age_hours: int = 24` to `max_snapshot_age_hours: int = 48`
|
||||
- _Bug_Condition: input.snapshotAgeHours > 24 AND input.snapshotAgeHours <= 48 AND input.maxSnapshotAgeConfig == 24_
|
||||
- _Expected_Behavior: Quality gate accepts snapshots up to 48h old_
|
||||
- _Preservation: Snapshots < 48h meeting criteria continue to promote to live_eligible_
|
||||
- _Requirements: 2.6, 3.5_
|
||||
|
||||
- [x] 5. Fix: Stuck Parsed Docs Recovery
|
||||
|
||||
- [x] 5.1 Lower STALE_PARSED_THRESHOLD_MINUTES from 240 to 30
|
||||
- File: `services/scheduler/app.py`
|
||||
- Change constant: `STALE_PARSED_THRESHOLD_MINUTES = 30`
|
||||
- Documents stuck longer than 30 minutes are likely orphaned
|
||||
- _Requirements: 2.1_
|
||||
|
||||
- [x] 5.2 Increase recovery batch LIMIT from 100 to 500
|
||||
- File: `services/scheduler/app.py`
|
||||
- In `recover_stale_documents()` SQL query, change `LIMIT 100` to `LIMIT 500`
|
||||
- _Requirements: 2.1_
|
||||
|
||||
- [x] 5.3 Update _ENQUEUED_TTL from 14400 to 3600
|
||||
- File: `services/scheduler/app.py`
|
||||
- Change `_ENQUEUED_TTL = 3600` (1 hour, matching the new recovery cadence)
|
||||
- _Bug_Condition: input.stuckParsedDocCount > 100 AND input.recoveryBatchLimit == 100 AND input.docStaleMinutes >= 240_
|
||||
- _Expected_Behavior: Recovery processes up to 500 docs per cycle with 30-min threshold_
|
||||
- _Preservation: Documents < 30 min old left alone for extraction queue consumer_
|
||||
- _Requirements: 2.1, 3.1, 3.7_
|
||||
|
||||
- [x] 6. Fix: Extended Price Fallback
|
||||
|
||||
- [x] 6.1 Add market_snapshots 24h time-window fallback to create_prediction_snapshot()
|
||||
- File: `services/validation/prediction_snapshot.py`
|
||||
- After positions fallback fails, query: `SELECT close FROM market_snapshots WHERE ticker = $1 AND timestamp >= NOW() - INTERVAL '24 hours' ORDER BY timestamp DESC LIMIT 1`
|
||||
- Add info-level logging when extended fallback is used
|
||||
- _Bug_Condition: input.tickerPrice IS NULL AND input.positionPrice IS NULL AND input.marketSnapshotWithin24h IS NOT NULL_
|
||||
- _Expected_Behavior: Use market_snapshots bar close price as price_at_prediction_
|
||||
- _Preservation: Primary fetch_latest_close_price() path unchanged when it returns a valid price_
|
||||
- _Requirements: 2.2, 3.2_
|
||||
|
||||
- [x] 6.2 Create backfill script scripts/backfill_snapshot_prices.py
|
||||
- Query all `prediction_snapshots` where `price_at_prediction IS NULL`
|
||||
- For each, attempt: `market_snapshots` within 24h of `generated_at`, then `positions` table
|
||||
- Update row with found price
|
||||
- Report statistics: found via market_snapshots, found via positions, still NULL
|
||||
- _Requirements: 2.3_
|
||||
|
||||
- [x] 7. Fix: Sentiment Z-Score Normalization
|
||||
|
||||
- [x] 7.1 Add normalize_impact_scores() function
|
||||
- File: `services/aggregation/scoring.py` (new helper function)
|
||||
- Query 7-day mean and stddev per ticker: `SELECT AVG(impact_score) as mean, STDDEV(impact_score) as stddev FROM document_impact_records WHERE ticker = $1 AND created_at >= NOW() - INTERVAL '7 days'`
|
||||
- Formula: `normalized = (raw - mean_7d) / max(stddev_7d, 0.1)`
|
||||
- Fallback: if fewer than 5 records in 7-day window, return raw score unchanged
|
||||
- The 0.1 floor prevents division by near-zero stddev for low-activity tickers
|
||||
- _Requirements: 2.4_
|
||||
|
||||
- [x] 7.2 Integrate normalization into aggregation loop
|
||||
- File: `services/aggregation/worker.py`
|
||||
- Before `compute_signal_weight()` call (around line 440), normalize `imp.impact_score` via `normalize_impact_scores()`
|
||||
- Pass normalized value into `compute_signal_weight()` and `WeightedSignal`
|
||||
- _Bug_Condition: input.impactScoreUsedRaw == TRUE AND input.ticker7dStddev > 0_
|
||||
- _Expected_Behavior: Normalized impact scores with mean ≈ 0, stddev ≈ 1 for active tickers_
|
||||
- _Preservation: Tickers with balanced distributions continue to produce neutral signals_
|
||||
- _Requirements: 2.4, 3.3_
|
||||
|
||||
- [x] 8. Verify fixes pass all tests
|
||||
|
||||
- [x] 8.1 Verify bug condition exploration test now passes
|
||||
- **Property 1: Expected Behavior** - Pipeline Health Bugs Resolved
|
||||
- **IMPORTANT**: Re-run the SAME test from task 1 - do NOT write a new test
|
||||
- The test from task 1 encodes the expected behavior for all five bugs
|
||||
- Run `tests/test_pbt_pipeline_health_bug_condition.py`
|
||||
- **EXPECTED OUTCOME**: Test PASSES (confirms bugs are fixed)
|
||||
- _Requirements: 2.1, 2.2, 2.4, 2.6_
|
||||
|
||||
- [x] 8.2 Verify preservation tests still pass
|
||||
- **Property 2: Preservation** - Pipeline Behavior Unchanged for Non-Bug Inputs
|
||||
- **IMPORTANT**: Re-run the SAME tests from task 2 - do NOT write new tests
|
||||
- Run `tests/test_pbt_pipeline_health_preservation.py`
|
||||
- **EXPECTED OUTCOME**: Tests PASS (confirms no regressions)
|
||||
- Confirm all preservation properties still hold after fixes
|
||||
- _Requirements: 3.1, 3.2, 3.3, 3.5, 3.6, 3.7_
|
||||
|
||||
- [x] 9. Lint and final validation
|
||||
- Run `.venv/bin/ruff check services/` and fix any lint errors
|
||||
- Run `.venv/bin/python -m pytest tests/ -x --tb=short -q` to confirm full test suite passes
|
||||
- Verify Helm template renders correctly with 0 signal-engine replicas
|
||||
- _Requirements: all_
|
||||
|
||||
- [x] 10. Checkpoint - Ensure all tests pass
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
## Task Dependency Graph
|
||||
|
||||
```json
|
||||
{
|
||||
"waves": [
|
||||
{"tasks": ["1", "2"]},
|
||||
{"tasks": ["3", "4"]},
|
||||
{"tasks": ["5"]},
|
||||
{"tasks": ["6"]},
|
||||
{"tasks": ["7"]},
|
||||
{"tasks": ["8"]},
|
||||
{"tasks": ["9"]},
|
||||
{"tasks": ["10"]}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
Tasks 3, 4 are independent config changes (no code deps).
|
||||
Task 5 is independent but task 6 builds on the price fallback concept.
|
||||
Task 7 is the most complex (new function + integration).
|
||||
Tasks 8-10 must run after all fixes are applied.
|
||||
|
||||
## Notes
|
||||
|
||||
- Bug 4 (signal engine) is validated by Helm template rendering, not a unit test
|
||||
- The backfill script (6.2) is a one-time operation, not covered by recurring tests
|
||||
- Preservation tests use Hypothesis with `@settings(max_examples=100)` per project conventions
|
||||
- Test files follow `test_pbt_*` naming convention per project standards
|
||||
@@ -0,0 +1 @@
|
||||
{"specId": "a7e3f1b2-9c4d-4e8a-b5f6-d2a1c3e7f9b0", "workflowType": "requirements-first", "specType": "feature"}
|
||||
@@ -0,0 +1,350 @@
|
||||
# Design Document: Remote vLLM Support
|
||||
|
||||
## Overview
|
||||
|
||||
This design introduces an LLM provider abstraction layer into Stonks Oracle so that both the existing Ollama backend and a new remote vLLM backend can be used interchangeably for document extraction and event classification. The vLLM server at `http://192.168.42.254:8000` runs `RedHatAI/Qwen3.6-35B-A3B-NVFP4` on an NVIDIA RTX 5090 with tensor parallelism and exposes an OpenAI-compatible `/v1/chat/completions` API.
|
||||
|
||||
The design preserves full backward compatibility — existing Ollama deployments work without any configuration changes. Provider selection is driven by the existing `model_provider` column in the `ai_agents` and `agent_variants` database tables, requiring no new migrations.
|
||||
|
||||
## Architecture
|
||||
|
||||
```mermaid
|
||||
graph TD
|
||||
subgraph "Extractor Worker"
|
||||
MAIN[main.py]
|
||||
FACTORY[LLMClientFactory]
|
||||
EXTRACT[Extraction Pipeline]
|
||||
CLASSIFY[Event Classification Pipeline]
|
||||
end
|
||||
|
||||
subgraph "Provider Abstraction"
|
||||
PROTO[LLMClient Protocol]
|
||||
OLLAMA_IMPL[OllamaClient]
|
||||
VLLM_IMPL[VLLMClient]
|
||||
end
|
||||
|
||||
subgraph "Configuration"
|
||||
RESOLVER[AgentConfigResolver]
|
||||
OLLAMA_CFG[OllamaConfig]
|
||||
VLLM_CFG[VLLMConfig]
|
||||
APP_CFG[AppConfig]
|
||||
end
|
||||
|
||||
subgraph "External Services"
|
||||
OLLAMA_SRV[Ollama Server<br/>:11434/api/chat]
|
||||
VLLM_SRV[vLLM Server<br/>:8000/v1/chat/completions]
|
||||
end
|
||||
|
||||
MAIN --> FACTORY
|
||||
FACTORY --> PROTO
|
||||
PROTO --> OLLAMA_IMPL
|
||||
PROTO --> VLLM_IMPL
|
||||
EXTRACT --> PROTO
|
||||
CLASSIFY --> PROTO
|
||||
|
||||
RESOLVER --> FACTORY
|
||||
OLLAMA_CFG --> FACTORY
|
||||
VLLM_CFG --> FACTORY
|
||||
APP_CFG --> OLLAMA_CFG
|
||||
APP_CFG --> VLLM_CFG
|
||||
|
||||
OLLAMA_IMPL --> OLLAMA_SRV
|
||||
VLLM_IMPL --> VLLM_SRV
|
||||
```
|
||||
|
||||
The key architectural decision is to use a Python `Protocol` (structural typing) rather than an ABC for the LLM client interface. This allows the existing `OllamaClient` to satisfy the protocol without inheritance changes, maintaining backward compatibility. The `VLLMClient` is a new class that also satisfies the protocol.
|
||||
|
||||
A factory function in `services/extractor/llm_factory.py` takes a `ResolvedAgentConfig` and the base configs, returning the appropriate client. The extractor worker (`main.py`) uses this factory instead of directly constructing `OllamaClient`.
|
||||
|
||||
## Components and Interfaces
|
||||
|
||||
### 1. LLM Client Protocol (`services/shared/llm_protocol.py`)
|
||||
|
||||
A `typing.Protocol` defining the contract both clients must satisfy:
|
||||
|
||||
```python
|
||||
from typing import Protocol, runtime_checkable
|
||||
|
||||
@runtime_checkable
|
||||
class LLMClient(Protocol):
|
||||
async def call_llm(
|
||||
self,
|
||||
prompts: dict[str, str],
|
||||
json_schema: dict[str, object],
|
||||
document_text: str = "",
|
||||
) -> "ExtractionAttempt": ...
|
||||
|
||||
async def close(self) -> None: ...
|
||||
```
|
||||
|
||||
The `call_llm` method signature matches the existing `OllamaClient._call_ollama()` parameters and return type. The `OllamaClient` gains a public `call_llm` method that delegates to `_call_ollama()`, preserving the private method for internal backward compatibility.
|
||||
|
||||
### 2. VLLMClient (`services/extractor/vllm_client.py`)
|
||||
|
||||
New client implementing the `LLMClient` protocol for the OpenAI-compatible API:
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class VLLMClient:
|
||||
_config: VLLMConfig
|
||||
_http: httpx.AsyncClient
|
||||
_owns_client: bool
|
||||
|
||||
async def call_llm(
|
||||
self,
|
||||
prompts: dict[str, str],
|
||||
json_schema: dict[str, object],
|
||||
document_text: str = "",
|
||||
) -> ExtractionAttempt: ...
|
||||
|
||||
async def close(self) -> None: ...
|
||||
```
|
||||
|
||||
**Request format** (OpenAI-compatible):
|
||||
```json
|
||||
{
|
||||
"model": "RedHatAI/Qwen3.6-35B-A3B-NVFP4",
|
||||
"messages": [
|
||||
{"role": "system", "content": "..."},
|
||||
{"role": "user", "content": "..."}
|
||||
],
|
||||
"max_tokens": 4096,
|
||||
"temperature": 0.7,
|
||||
"response_format": {"type": "json_object"}
|
||||
}
|
||||
```
|
||||
|
||||
**Response parsing**: Extracts `choices[0].message.content`, then applies the same `_strip_markdown_fences()` and `_repair_json()` pipeline as `OllamaClient`.
|
||||
|
||||
**Error handling**: Maps HTTP errors to the same string format as `OllamaClient` (`timeout`, `http_{code}`, `connection_error: {details}`, `empty_model_response`), so the existing `_is_retryable()` function works without modification.
|
||||
|
||||
**Key differences from OllamaClient**:
|
||||
- Endpoint: `/v1/chat/completions` instead of `/api/chat`
|
||||
- No `think: false`, `stream: false`, or `options` block
|
||||
- Uses `max_tokens` instead of `options.num_predict`
|
||||
- Uses `response_format: {"type": "json_object"}` for structured output
|
||||
- Supports `temperature` parameter (Ollama uses model defaults)
|
||||
- Response in `choices[0].message.content` instead of `message.content`
|
||||
|
||||
### 3. VLLMConfig (`services/shared/config.py`)
|
||||
|
||||
New dataclass alongside `OllamaConfig`:
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class VLLMConfig:
|
||||
base_url: str = "http://192.168.42.254:8000"
|
||||
model: str = "RedHatAI/Qwen3.6-35B-A3B-NVFP4"
|
||||
timeout: int = 120
|
||||
max_retries: int = 2
|
||||
retry_base_delay: float = 1.0
|
||||
retry_max_delay: float = 10.0
|
||||
retry_backoff_multiplier: float = 2.0
|
||||
max_tokens: int = 32768
|
||||
temperature: float = 0.7
|
||||
api_key: str = "" # Optional, for authenticated vLLM deployments
|
||||
```
|
||||
|
||||
Loaded from `VLLM_*` environment variables in `load_config()`. Added to `AppConfig` as `vllm: VLLMConfig`.
|
||||
|
||||
### 4. LLM Client Factory (`services/extractor/llm_factory.py`)
|
||||
|
||||
Factory function that replaces the hardcoded `OllamaClient` construction:
|
||||
|
||||
```python
|
||||
def build_llm_client(
|
||||
resolved: ResolvedAgentConfig | None,
|
||||
ollama_config: OllamaConfig,
|
||||
vllm_config: VLLMConfig,
|
||||
http_client: httpx.AsyncClient | None = None,
|
||||
) -> LLMClient:
|
||||
"""Return the appropriate LLM client based on resolved provider."""
|
||||
...
|
||||
|
||||
def build_config_from_resolved(
|
||||
resolved: ResolvedAgentConfig,
|
||||
base_ollama: OllamaConfig,
|
||||
base_vllm: VLLMConfig,
|
||||
) -> OllamaConfig | VLLMConfig:
|
||||
"""Build provider-specific config from resolved agent config."""
|
||||
...
|
||||
```
|
||||
|
||||
Provider routing logic:
|
||||
1. If `resolved` is `None` or `resolved.model_provider` is `"ollama"` or empty → `OllamaClient`
|
||||
2. If `resolved.model_provider` is `"vllm"` → `VLLMClient`
|
||||
3. Unknown provider → log warning, fall back to `OllamaClient`
|
||||
|
||||
### 5. Updated Extractor Worker (`services/extractor/main.py`)
|
||||
|
||||
Changes to `main()`:
|
||||
- Replace `_build_ollama_config_from_resolved()` with `build_llm_client()` from the factory
|
||||
- Store clients as `LLMClient` type instead of `OllamaClient`
|
||||
- On config refresh (every 100 jobs), detect provider changes and swap clients
|
||||
- Log provider switches at INFO level
|
||||
|
||||
Changes to `_process_macro_classification()`:
|
||||
- Accept `LLMClient` instead of `OllamaClient` for the classifier parameter
|
||||
|
||||
### 6. Updated OllamaClient (`services/extractor/client.py`)
|
||||
|
||||
Minimal changes to satisfy the protocol:
|
||||
- Add public `call_llm()` method that delegates to `_call_ollama()`
|
||||
- Keep `_call_ollama()` as-is for backward compatibility
|
||||
- The `extract()` method continues to call `_call_ollama()` internally
|
||||
|
||||
### 7. Updated Event Classifier (`services/extractor/event_classifier.py`)
|
||||
|
||||
Changes to `classify_global_event()`:
|
||||
- Accept `LLMClient` instead of `Any` for the `ollama_client` parameter
|
||||
- Call `client.call_llm()` instead of `ollama_client._call_ollama()`
|
||||
- Set `ModelMetadata.provider` based on the actual client type (inspect `_config` or pass provider string)
|
||||
|
||||
### 8. Helm Values (`infra/helm/stonks-oracle/values.yaml`)
|
||||
|
||||
New config entries:
|
||||
```yaml
|
||||
config:
|
||||
VLLM_BASE_URL: "http://192.168.42.254:8000"
|
||||
VLLM_MODEL: "RedHatAI/Qwen3.6-35B-A3B-NVFP4"
|
||||
VLLM_TIMEOUT: "120"
|
||||
VLLM_MAX_RETRIES: "2"
|
||||
VLLM_TEMPERATURE: "0.7"
|
||||
VLLM_API_KEY: ""
|
||||
```
|
||||
|
||||
### 9. Health Check (`services/extractor/vllm_client.py`)
|
||||
|
||||
Startup validation function:
|
||||
|
||||
```python
|
||||
async def check_vllm_health(base_url: str, timeout: float = 10.0) -> bool:
|
||||
"""GET {base_url}/v1/models to verify vLLM is reachable."""
|
||||
...
|
||||
```
|
||||
|
||||
Called from `main()` when the resolved or default config specifies vLLM. On failure, logs WARNING and falls back to Ollama. On success, logs INFO with server URL and model list.
|
||||
|
||||
## Data Models
|
||||
|
||||
### VLLMConfig Dataclass
|
||||
|
||||
| Field | Type | Default | Env Var |
|
||||
|-------|------|---------|---------|
|
||||
| `base_url` | `str` | `http://192.168.42.254:8000` | `VLLM_BASE_URL` |
|
||||
| `model` | `str` | `RedHatAI/Qwen3.6-35B-A3B-NVFP4` | `VLLM_MODEL` |
|
||||
| `timeout` | `int` | `120` | `VLLM_TIMEOUT` |
|
||||
| `max_retries` | `int` | `2` | `VLLM_MAX_RETRIES` |
|
||||
| `retry_base_delay` | `float` | `1.0` | `VLLM_RETRY_BASE_DELAY` |
|
||||
| `retry_max_delay` | `float` | `10.0` | `VLLM_RETRY_MAX_DELAY` |
|
||||
| `retry_backoff_multiplier` | `float` | `2.0` | `VLLM_RETRY_BACKOFF_MULTIPLIER` |
|
||||
| `max_tokens` | `int` | `32768` | `VLLM_MAX_TOKENS` |
|
||||
| `temperature` | `float` | `0.7` | `VLLM_TEMPERATURE` |
|
||||
| `api_key` | `str` | `""` | `VLLM_API_KEY` |
|
||||
|
||||
### ExtractionAttempt (unchanged)
|
||||
|
||||
The existing `ExtractionAttempt` dataclass is reused as-is for both providers. No changes needed.
|
||||
|
||||
### ModelMetadata (unchanged structure, new values)
|
||||
|
||||
The `provider` field now accepts `"vllm"` in addition to `"ollama"`. No schema change needed.
|
||||
|
||||
## Error Handling
|
||||
|
||||
### Error String Format Parity
|
||||
|
||||
Both clients produce identical error string formats so `_is_retryable()` works unchanged:
|
||||
|
||||
| Condition | Error String | Retryable |
|
||||
|-----------|-------------|-----------|
|
||||
| HTTP timeout | `timeout` | Yes |
|
||||
| HTTP 400/401/403/404/422 | `http_{code}` | No |
|
||||
| HTTP 500/502/503/429 | `http_{code}` | Yes |
|
||||
| Connection refused/reset | `connection_error: {details}` | Yes |
|
||||
| Empty response body | `empty_model_response` | Yes |
|
||||
| Invalid JSON in response | `invalid_response_json` | Yes |
|
||||
|
||||
### Health Check Failure
|
||||
|
||||
If the vLLM health check fails at startup:
|
||||
1. Log WARNING with the error details
|
||||
2. Fall back to `OllamaClient` using `OllamaConfig`
|
||||
3. Continue operation — the system degrades gracefully rather than crashing
|
||||
|
||||
### Provider Switch During Refresh
|
||||
|
||||
When the config refresh (every 100 jobs) detects a provider change:
|
||||
1. Close the old client (`await old_client.close()`)
|
||||
2. Construct the new client via the factory
|
||||
3. Log the switch at INFO level
|
||||
4. If new client construction fails, keep the old client and log ERROR
|
||||
|
||||
## Testing Strategy
|
||||
|
||||
### Property-Based Tests (`tests/test_pbt_llm_provider.py`)
|
||||
|
||||
Property-based tests using Hypothesis to verify the provider abstraction:
|
||||
|
||||
**P1: Provider factory routing property** (Req 3.4, 3.5, 9.5)
|
||||
For all `model_provider` values in `{"ollama", "vllm", "", None}`, the factory returns the correct client type. For `"ollama"`, empty, or `None`, returns `OllamaClient`. For `"vllm"`, returns `VLLMClient`.
|
||||
|
||||
**P2: Error string format consistency property** (Req 5.6)
|
||||
For all HTTP status codes (100-599), both `OllamaClient` and `VLLMClient` produce error strings in the same format (`http_{code}`), and `_is_retryable()` returns the same result for both.
|
||||
|
||||
**P3: VLLMClient request payload structure property** (Req 2.1, 8.1)
|
||||
For all generated prompt dicts (system + user messages of arbitrary text), the VLLMClient produces a request payload that: contains `model`, `messages`, `max_tokens`, `temperature`; does NOT contain `think`, `stream`, `options`, `num_ctx`, `num_predict`.
|
||||
|
||||
**P4: JSON repair idempotence property** (Req 2.4)
|
||||
For all valid JSON strings, `_repair_json(json_str)` returns a string that `json.loads()` can parse, and `_repair_json(_repair_json(json_str)) == _repair_json(json_str)` (idempotence).
|
||||
|
||||
**P5: Markdown fence stripping round-trip property** (Req 2.3)
|
||||
For all strings `s`, `_strip_markdown_fences(f"```json\n{s}\n```")` returns `s` (stripped), and `_strip_markdown_fences(s)` returns `s` when no fences are present (identity).
|
||||
|
||||
**P6: VLLMConfig default construction property** (Req 3.1)
|
||||
For all VLLMConfig instances constructed with default values, `base_url` is non-empty, `timeout > 0`, `max_retries >= 0`, `temperature` is between 0.0 and 2.0, and `max_tokens > 0`.
|
||||
|
||||
### Unit Tests (`tests/test_vllm_client.py`)
|
||||
|
||||
Example-based tests for specific behaviors:
|
||||
|
||||
- VLLMClient sends correct payload to `/v1/chat/completions` (mock httpx)
|
||||
- VLLMClient extracts content from `choices[0].message.content`
|
||||
- VLLMClient handles empty choices array → `empty_model_response`
|
||||
- VLLMClient handles timeout → `timeout` error
|
||||
- VLLMClient handles HTTP 500 → `http_500` error, retryable
|
||||
- VLLMClient handles HTTP 400 → `http_400` error, non-retryable
|
||||
- VLLMClient handles connection refused → `connection_error: ...`
|
||||
- VLLMClient applies markdown fence stripping
|
||||
- VLLMClient applies JSON repair
|
||||
- VLLMClient includes temperature in payload
|
||||
- VLLMClient includes `response_format` in payload
|
||||
- Health check success logs INFO
|
||||
- Health check failure logs WARNING and returns False
|
||||
- Factory returns OllamaClient for provider="ollama"
|
||||
- Factory returns VLLMClient for provider="vllm"
|
||||
- Factory returns OllamaClient for provider="" (default)
|
||||
- Factory returns OllamaClient for unknown provider with warning
|
||||
- VLLMConfig loads from environment variables
|
||||
- AppConfig includes vllm field with defaults
|
||||
- OllamaClient.call_llm() delegates to _call_ollama()
|
||||
|
||||
### Existing Tests (unchanged)
|
||||
|
||||
- `tests/test_ollama_client.py` — continues to pass without modification
|
||||
- All other existing test files — unaffected
|
||||
|
||||
## File Changes Summary
|
||||
|
||||
| File | Change Type | Description |
|
||||
|------|-------------|-------------|
|
||||
| `services/shared/llm_protocol.py` | **New** | `LLMClient` Protocol definition |
|
||||
| `services/extractor/vllm_client.py` | **New** | `VLLMClient` implementation + health check |
|
||||
| `services/extractor/llm_factory.py` | **New** | Factory function for provider routing |
|
||||
| `services/shared/config.py` | **Modified** | Add `VLLMConfig`, update `AppConfig`, update `load_config()` |
|
||||
| `services/extractor/client.py` | **Modified** | Add `call_llm()` public method to `OllamaClient` |
|
||||
| `services/extractor/event_classifier.py` | **Modified** | Use `call_llm()` instead of `_call_ollama()`, accept `LLMClient` type |
|
||||
| `services/extractor/main.py` | **Modified** | Use factory, support provider switching, health check |
|
||||
| `infra/helm/stonks-oracle/values.yaml` | **Modified** | Add `VLLM_*` config entries |
|
||||
| `tests/test_pbt_llm_provider.py` | **New** | Property-based tests for provider abstraction |
|
||||
| `tests/test_vllm_client.py` | **New** | Unit tests for VLLMClient and factory |
|
||||
@@ -0,0 +1,136 @@
|
||||
# Requirements Document
|
||||
|
||||
## Introduction
|
||||
|
||||
Add remote vLLM support to the Stonks Oracle platform. The system currently uses Ollama exclusively for LLM inference via the `/api/chat` endpoint. A remote vLLM server running `RedHatAI/Qwen3.6-35B-A3B-NVFP4` on a 5090 GPU with tensor parallelism is available at `http://192.168.42.254:8000` and exposes an OpenAI-compatible `/v1/chat/completions` API. This feature introduces a provider abstraction layer so that both Ollama and vLLM backends can be used interchangeably, selected per-agent via the existing `model_provider` database column and environment variable configuration. The abstraction preserves all existing behavior (retry logic, JSON repair, audit trail, backoff, context window override) while adapting to the differences between the two API protocols.
|
||||
|
||||
## Glossary
|
||||
|
||||
- **LLM_Client**: An abstract interface defining the contract for sending chat completion requests to any LLM backend. Concrete implementations exist for Ollama and vLLM.
|
||||
- **Ollama_Backend**: The existing Ollama inference server at `ollama.ollama-service.svc.cluster.local:11434` (cluster) or `http://10.1.1.12:2701` (external), using the `/api/chat` endpoint with Ollama-specific payload fields (`think`, `options.num_ctx`, `options.num_predict`).
|
||||
- **VLLM_Backend**: A remote vLLM inference server at `http://192.168.42.254:8000` exposing the OpenAI-compatible `/v1/chat/completions` endpoint. Runs `RedHatAI/Qwen3.6-35B-A3B-NVFP4` on a 5090 GPU with tensor parallelism.
|
||||
- **Provider**: A string identifier (`ollama` or `vllm`) that determines which LLM_Client implementation is used for a given agent. Stored in the `model_provider` column of `ai_agents` and `agent_variants` tables.
|
||||
- **LLM_Config**: A provider-agnostic configuration dataclass containing connection and inference parameters (base_url, model, timeout, retries, max_tokens, context_window) used to construct an LLM_Client.
|
||||
- **Extraction_Pipeline**: The document intelligence extraction workflow in `services/extractor/client.py` that sends documents to an LLM and parses structured JSON responses.
|
||||
- **Event_Classification_Pipeline**: The macro event classification workflow in `services/extractor/event_classifier.py` that classifies global news articles via an LLM.
|
||||
- **Agent_Config_Resolver**: The `AgentConfigResolver` in `services/shared/agent_config.py` that resolves runtime configuration from the `ai_agents` and `agent_variants` database tables, including the `model_provider` field.
|
||||
- **OpenAI_Chat_Format**: The request/response format used by `/v1/chat/completions` — messages array with role/content, `max_tokens`, `temperature`, and response in `choices[0].message.content`.
|
||||
- **JSON_Repair**: The existing `json-repair` library usage that fixes malformed JSON from model output, applied regardless of provider.
|
||||
- **Model_Metadata**: The `ModelMetadata` Pydantic model in `services/shared/schemas.py` that tracks `provider`, `model_name`, `prompt_version`, and `schema_version` for audit.
|
||||
|
||||
## Requirements
|
||||
|
||||
### Requirement 1: Provider Abstraction Layer
|
||||
|
||||
**User Story:** As a developer, I want a provider abstraction layer that decouples LLM inference from any specific backend, so that the extraction and classification pipelines can use either Ollama or vLLM without code changes in the calling services.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE LLM_Client interface SHALL define an async method that accepts a messages list (system and user prompts), a JSON schema hint, and optional document text, and returns an attempt result containing raw output, validation report, error string, duration, and model name.
|
||||
2. THE LLM_Client interface SHALL define an async `close` method for releasing underlying HTTP resources.
|
||||
3. WHEN the Extraction_Pipeline calls the LLM, THE Extraction_Pipeline SHALL use the LLM_Client interface instead of calling Ollama-specific endpoints directly.
|
||||
4. WHEN the Event_Classification_Pipeline calls the LLM, THE Event_Classification_Pipeline SHALL use the LLM_Client interface instead of calling `_call_ollama()` directly.
|
||||
5. THE Ollama_Backend implementation of LLM_Client SHALL preserve the existing `/api/chat` payload structure including `think: false`, `stream: false`, `options.num_predict`, and `options.num_ctx`.
|
||||
6. THE VLLM_Backend implementation of LLM_Client SHALL send requests to `/v1/chat/completions` using the OpenAI_Chat_Format with `model`, `messages`, `max_tokens`, and `temperature` fields.
|
||||
7. FOR ALL valid prompt inputs, sending a prompt through the Ollama_Backend and parsing the response SHALL produce the same ExtractionAttempt structure as the current `_call_ollama()` method (round-trip equivalence with existing behavior).
|
||||
|
||||
### Requirement 2: vLLM Client Implementation
|
||||
|
||||
**User Story:** As a developer, I want a vLLM client that communicates with the remote vLLM server using the OpenAI-compatible API, so that the platform can leverage the 5090 GPU for inference.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE VLLM_Backend SHALL send POST requests to `{base_url}/v1/chat/completions` with a JSON payload containing `model`, `messages` (array of role/content objects), `max_tokens`, and `temperature`.
|
||||
2. THE VLLM_Backend SHALL extract the response content from `choices[0].message.content` in the OpenAI-compatible response format.
|
||||
3. THE VLLM_Backend SHALL apply the same markdown fence stripping logic as the Ollama_Backend to handle model output wrapped in ```json ... ``` blocks.
|
||||
4. THE VLLM_Backend SHALL apply the same JSON_Repair logic as the Ollama_Backend to fix malformed JSON in model output.
|
||||
5. WHEN the vLLM server returns an HTTP timeout, THE VLLM_Backend SHALL report the error as `timeout` in the attempt result, consistent with the Ollama_Backend error format.
|
||||
6. WHEN the vLLM server returns an HTTP error status, THE VLLM_Backend SHALL report the error as `http_{status_code}` in the attempt result, consistent with the Ollama_Backend error format.
|
||||
7. WHEN the vLLM server returns an empty `choices` array or missing `content`, THE VLLM_Backend SHALL report the error as `empty_model_response`.
|
||||
8. IF the vLLM server is unreachable, THEN THE VLLM_Backend SHALL report the error as `connection_error: {details}`, consistent with the Ollama_Backend error format.
|
||||
9. THE VLLM_Backend SHALL use the same `httpx.AsyncClient` timeout configuration as the Ollama_Backend, derived from the LLM_Config timeout value.
|
||||
10. THE VLLM_Backend SHALL support an optional `temperature` parameter from the resolved agent config, defaulting to 0.7 when not specified.
|
||||
|
||||
### Requirement 3: Provider-Aware Configuration
|
||||
|
||||
**User Story:** As an operator, I want to configure the vLLM backend via environment variables and database agent config, so that I can switch providers without code changes.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Configuration SHALL include a `VLLMConfig` dataclass with fields: `base_url` (default `http://192.168.42.254:8000`), `model` (default `RedHatAI/Qwen3.6-35B-A3B-NVFP4`), `timeout` (default 120), `max_retries` (default 2), `retry_base_delay`, `retry_max_delay`, `retry_backoff_multiplier`, `max_tokens` (default 32768), and `temperature` (default 0.7).
|
||||
2. THE Configuration SHALL load VLLMConfig values from environment variables prefixed with `VLLM_` (e.g., `VLLM_BASE_URL`, `VLLM_MODEL`, `VLLM_TIMEOUT`), following the same pattern as OllamaConfig.
|
||||
3. THE AppConfig dataclass SHALL include a `vllm` field of type VLLMConfig alongside the existing `ollama` field.
|
||||
4. WHEN the Agent_Config_Resolver resolves a `model_provider` value of `vllm`, THE service SHALL use the VLLMConfig base_url and construct a VLLM_Backend client instead of an Ollama_Backend client.
|
||||
5. WHEN the Agent_Config_Resolver resolves a `model_provider` value of `ollama` or when no `model_provider` is specified, THE service SHALL continue to use the OllamaConfig and Ollama_Backend client as the default.
|
||||
6. THE `_build_ollama_config_from_resolved` function in `services/extractor/main.py` SHALL be generalized to a provider-aware factory that returns the appropriate config and client type based on the resolved `model_provider`.
|
||||
|
||||
### Requirement 4: Provider Selection in Extractor Worker
|
||||
|
||||
**User Story:** As a developer, I want the extractor worker to select the correct LLM client based on the resolved agent config provider, so that each agent can independently use Ollama or vLLM.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN the extractor worker starts, THE worker SHALL construct the default LLM_Client based on the environment variable configuration (defaulting to Ollama_Backend).
|
||||
2. WHEN the Agent_Config_Resolver returns a resolved config with `model_provider = "vllm"` for the `document-extractor` slug, THE worker SHALL construct a VLLM_Backend client using the VLLMConfig base_url and the resolved model_name.
|
||||
3. WHEN the Agent_Config_Resolver returns a resolved config with `model_provider = "vllm"` for the `event-classifier` slug, THE worker SHALL construct a VLLM_Backend client for the event classification pipeline.
|
||||
4. WHEN the resolved config changes provider during a config refresh cycle (every 100 jobs), THE worker SHALL close the old LLM_Client and construct a new one matching the updated provider.
|
||||
5. WHEN the resolved config changes from `ollama` to `vllm` or vice versa, THE worker SHALL log the provider switch at INFO level including the old and new provider, model name, and variant ID.
|
||||
|
||||
### Requirement 5: Retry and Error Handling Parity
|
||||
|
||||
**User Story:** As a developer, I want the vLLM client to use the same retry logic, backoff strategy, and error classification as the Ollama client, so that reliability behavior is consistent across providers.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE VLLM_Backend SHALL use the same exponential backoff computation as the Ollama_Backend, using `retry_base_delay`, `retry_max_delay`, and `retry_backoff_multiplier` from the LLM_Config.
|
||||
2. THE VLLM_Backend SHALL classify HTTP 400, 401, 403, 404, and 422 errors as non-retryable, consistent with the Ollama_Backend.
|
||||
3. THE VLLM_Backend SHALL classify HTTP 500, 502, 503, 429, timeout, and connection errors as retryable, consistent with the Ollama_Backend.
|
||||
4. WHEN the VLLM_Backend encounters a retryable error, THE Extraction_Pipeline SHALL retry up to `max_retries` times with exponential backoff, preserving each attempt in the audit trail.
|
||||
5. WHEN the VLLM_Backend encounters a non-retryable error, THE Extraction_Pipeline SHALL stop retries immediately and record the attempt as non-retryable.
|
||||
6. FOR ALL error types, the VLLM_Backend error string format SHALL match the Ollama_Backend error string format so that `_is_retryable()` works without modification.
|
||||
|
||||
### Requirement 6: Audit Trail and Model Metadata
|
||||
|
||||
**User Story:** As a developer, I want the audit trail and model metadata to correctly reflect which provider and model were used for each extraction, so that I can trace results back to the specific backend.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN the VLLM_Backend completes an extraction attempt, THE attempt record SHALL include the vLLM model name in the `model` field.
|
||||
2. WHEN an extraction or classification succeeds via the VLLM_Backend, THE Model_Metadata in the result SHALL have `provider` set to `"vllm"` and `model_name` set to the vLLM model identifier.
|
||||
3. WHEN the `agent_performance_log` records an invocation that used the VLLM_Backend, THE log entry SHALL be attributed to the correct agent_id and variant_id, consistent with Ollama_Backend logging.
|
||||
4. THE MinIO prompt and result artifacts persisted by the Event_Classification_Pipeline SHALL include the provider name and model name in the stored JSON, regardless of which backend was used.
|
||||
|
||||
### Requirement 7: Health Check and Connectivity Validation
|
||||
|
||||
**User Story:** As an operator, I want the system to validate connectivity to the vLLM server at startup, so that misconfiguration is detected early rather than failing silently on the first inference request.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN the extractor worker starts and the resolved or default config specifies `model_provider = "vllm"`, THE worker SHALL send a GET request to `{vllm_base_url}/v1/models` to verify the vLLM server is reachable.
|
||||
2. IF the vLLM health check fails at startup, THEN THE worker SHALL log a WARNING and fall back to the Ollama_Backend, continuing operation with degraded capability.
|
||||
3. IF the vLLM health check succeeds, THEN THE worker SHALL log an INFO message confirming the vLLM connection including the server URL and available model name.
|
||||
4. THE health check SHALL use a timeout of 10 seconds to avoid blocking worker startup on an unresponsive server.
|
||||
|
||||
### Requirement 8: Context Window and Token Handling for vLLM
|
||||
|
||||
**User Story:** As a developer, I want the vLLM client to handle context window and token limits appropriately for the vLLM API, so that large documents are processed correctly on the remote GPU.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN the resolved agent config specifies a non-zero `context_window`, THE VLLM_Backend SHALL omit the `num_ctx` Ollama-specific option and instead rely on the vLLM server's model configuration for context window sizing.
|
||||
2. THE VLLM_Backend SHALL pass `max_tokens` in the OpenAI-compatible request payload to control the maximum number of output tokens generated.
|
||||
3. WHEN the resolved agent config specifies a non-zero `input_token_limit`, THE Extraction_Pipeline SHALL truncate the input text before sending it to the VLLM_Backend, using the same truncation logic as for the Ollama_Backend.
|
||||
4. WHEN the resolved agent config specifies a non-zero `token_budget`, THE worker SHALL enforce the same hourly token budget check for vLLM invocations as for Ollama invocations.
|
||||
|
||||
### Requirement 9: Backward Compatibility
|
||||
|
||||
**User Story:** As a developer, I want the vLLM integration to be fully backward compatible, so that existing Ollama-based deployments continue to work without any configuration changes.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN no `VLLM_BASE_URL` environment variable is set and no agent config specifies `model_provider = "vllm"`, THE system SHALL behave identically to the current Ollama-only implementation.
|
||||
2. THE existing `OllamaConfig` dataclass and its environment variable loading SHALL remain unchanged.
|
||||
3. THE existing `OllamaClient` class SHALL continue to function for Ollama-specific usage, with the LLM_Client interface added as a compatible layer on top.
|
||||
4. THE existing test suite in `tests/test_ollama_client.py` SHALL continue to pass without modification.
|
||||
5. WHEN the `model_provider` column in `ai_agents` or `agent_variants` contains `"ollama"` or NULL, THE system SHALL use the Ollama_Backend, preserving current behavior.
|
||||
6. THE database migration for this feature SHALL NOT alter existing table structures; it SHALL only add new columns or tables if needed.
|
||||
@@ -0,0 +1,82 @@
|
||||
# Tasks
|
||||
|
||||
## Task 1: LLM Client Protocol and VLLMConfig
|
||||
|
||||
- [x] 1.1 Create `services/shared/llm_protocol.py` with `LLMClient` Protocol defining `call_llm(prompts, json_schema, document_text) -> ExtractionAttempt` and `close()` methods
|
||||
- [x] 1.2 Add `VLLMConfig` dataclass to `services/shared/config.py` with fields: `base_url`, `model`, `timeout`, `max_retries`, `retry_base_delay`, `retry_max_delay`, `retry_backoff_multiplier`, `max_tokens`, `temperature`, `api_key`
|
||||
- [x] 1.3 Add `vllm: VLLMConfig` field to `AppConfig` dataclass
|
||||
- [x] 1.4 Add `VLLM_*` environment variable loading to `load_config()` function
|
||||
- [x] 1.5 Add public `call_llm()` method to `OllamaClient` in `services/extractor/client.py` that delegates to `_call_ollama()`
|
||||
|
||||
## Task 2: VLLMClient Implementation
|
||||
|
||||
- [x] 2.1 Create `services/extractor/vllm_client.py` with `VLLMClient` class that satisfies the `LLMClient` protocol
|
||||
- [x] 2.2 Implement `call_llm()` method that sends POST to `/v1/chat/completions` with OpenAI-compatible payload (`model`, `messages`, `max_tokens`, `temperature`, `response_format`)
|
||||
- [x] 2.3 Implement response parsing: extract content from `choices[0].message.content`, apply `_strip_markdown_fences()` and `_repair_json()`
|
||||
- [x] 2.4 Implement error handling: map timeout → `timeout`, HTTP errors → `http_{code}`, connection errors → `connection_error: {details}`, empty response → `empty_model_response`
|
||||
- [x] 2.5 Implement `close()` method to release the underlying `httpx.AsyncClient`
|
||||
- [x] 2.6 Implement `check_vllm_health(base_url, timeout=10.0)` async function that GETs `/v1/models` and returns bool
|
||||
|
||||
## Task 3: LLM Client Factory
|
||||
|
||||
- [x] 3.1 Create `services/extractor/llm_factory.py` with `build_llm_client()` function that returns `OllamaClient` or `VLLMClient` based on resolved `model_provider`
|
||||
- [x] 3.2 Implement `build_config_from_resolved()` function that creates provider-specific config from `ResolvedAgentConfig` and base configs
|
||||
- [x] 3.3 Handle unknown provider values: log warning and fall back to `OllamaClient`
|
||||
|
||||
## Task 4: Update Extractor Worker for Provider Abstraction
|
||||
|
||||
- [x] 4.1 Update `services/extractor/main.py` to import and use `build_llm_client()` from the factory instead of directly constructing `OllamaClient`
|
||||
- [x] 4.2 Replace `_build_ollama_config_from_resolved()` usage with the factory's `build_config_from_resolved()` for both extractor and classifier clients
|
||||
- [x] 4.3 Add vLLM health check call at startup when resolved config specifies `model_provider = "vllm"`, with fallback to Ollama on failure
|
||||
- [x] 4.4 Update config refresh logic (every 100 jobs) to detect provider changes, close old client, and construct new client via factory
|
||||
- [x] 4.5 Add INFO-level logging for provider switches including old/new provider, model name, and variant ID
|
||||
|
||||
## Task 5: Update Event Classifier for Provider Abstraction
|
||||
|
||||
- [x] 5.1 Update `classify_global_event()` in `services/extractor/event_classifier.py` to accept `LLMClient` protocol type instead of `Any` for the client parameter
|
||||
- [x] 5.2 Replace `ollama_client._call_ollama()` calls with `client.call_llm()` calls
|
||||
- [x] 5.3 Update `ModelMetadata.provider` assignment to use the actual provider string from the client (detect from config type or pass explicitly)
|
||||
- [x] 5.4 Update retry logic to use client config attributes instead of accessing `ollama_client._base_delay` and `ollama_client._backoff_multiplier` directly
|
||||
|
||||
## Task 6: Helm Configuration
|
||||
|
||||
- [x] 6.1 Add `VLLM_BASE_URL`, `VLLM_MODEL`, `VLLM_TIMEOUT`, `VLLM_MAX_RETRIES`, `VLLM_TEMPERATURE`, and `VLLM_API_KEY` entries to the `config:` section in `infra/helm/stonks-oracle/values.yaml`
|
||||
|
||||
## Task 7: Unit Tests for VLLMClient
|
||||
|
||||
- [x] 7.1 Create `tests/test_vllm_client.py` with test for VLLMClient sending correct payload to `/v1/chat/completions` using mock httpx transport
|
||||
- [x] 7.2 Add test for VLLMClient extracting content from `choices[0].message.content`
|
||||
- [x] 7.3 Add test for VLLMClient handling empty choices array returning `empty_model_response` error
|
||||
- [x] 7.4 Add test for VLLMClient handling HTTP timeout returning `timeout` error
|
||||
- [x] 7.5 Add test for VLLMClient handling HTTP 500 returning `http_500` retryable error
|
||||
- [x] 7.6 Add test for VLLMClient handling HTTP 400 returning `http_400` non-retryable error
|
||||
- [x] 7.7 Add test for VLLMClient handling connection error returning `connection_error: ...`
|
||||
- [x] 7.8 Add test for VLLMClient applying markdown fence stripping and JSON repair to response
|
||||
- [x] 7.9 Add test for VLLMClient including temperature and response_format in payload
|
||||
- [x] 7.10 Add test for health check success returning True and logging INFO
|
||||
- [x] 7.11 Add test for health check failure returning False and logging WARNING
|
||||
- [x] 7.12 Add test for OllamaClient.call_llm() delegating to _call_ollama()
|
||||
- [x] 7.13 Add test for VLLMConfig loading from environment variables
|
||||
- [x] 7.14 Add test for AppConfig including vllm field with correct defaults
|
||||
|
||||
## Task 8: Unit Tests for LLM Factory
|
||||
|
||||
- [x] 8.1 Add tests to `tests/test_vllm_client.py` for factory returning OllamaClient when provider is "ollama"
|
||||
- [x] 8.2 Add test for factory returning VLLMClient when provider is "vllm"
|
||||
- [x] 8.3 Add test for factory returning OllamaClient when provider is empty string (default)
|
||||
- [x] 8.4 Add test for factory returning OllamaClient with warning when provider is unknown value
|
||||
|
||||
## Task 9: Property-Based Tests
|
||||
|
||||
- [x] 9.1 Create `tests/test_pbt_llm_provider.py` with property test for factory routing: for all model_provider in {"ollama", "vllm", "", None}, factory returns correct client type [PBT]
|
||||
- [x] 9.2 Add property test for error string format consistency: for all HTTP status codes (100-599), `_is_retryable()` classifies them consistently [PBT]
|
||||
- [x] 9.3 Add property test for VLLMClient request payload structure: for all generated prompt dicts, payload contains required OpenAI fields and excludes Ollama-specific fields [PBT]
|
||||
- [x] 9.4 Add property test for JSON repair idempotence: for all valid JSON strings, `_repair_json()` is idempotent [PBT]
|
||||
- [x] 9.5 Add property test for markdown fence stripping: for all strings, wrapping in fences then stripping recovers the original [PBT]
|
||||
- [x] 9.6 Add property test for VLLMConfig defaults: for all default-constructed instances, invariants hold (timeout > 0, max_retries >= 0, 0 <= temperature <= 2, max_tokens > 0) [PBT]
|
||||
|
||||
## Task 10: Verification and Backward Compatibility
|
||||
|
||||
- [x] 10.1 Run existing `tests/test_ollama_client.py` to verify no regressions
|
||||
- [x] 10.2 Run `ruff check services/` to verify no lint errors in modified files
|
||||
- [x] 10.3 Run full test suite `python -m pytest tests/ -x --tb=short -q` to verify all tests pass
|
||||
@@ -0,0 +1 @@
|
||||
{"specId": "e6d189b2-5861-4e24-954f-5e254246a910", "workflowType": "requirements-first", "specType": "feature"}
|
||||
@@ -0,0 +1,341 @@
|
||||
# Design Document: Sanitized Pipeline Documentation
|
||||
|
||||
## Overview
|
||||
|
||||
This design specifies the process and structure for producing a sanitized version of the 6-page intelligence pipeline deep dive documentation. The sanitized docs transform the existing `docs/intelligence-pipeline-deep-dive/` content into domain-neutral equivalents stored at `docs/sanitized-pipeline-deep-dive/`, stripping all financial, market, and trading language while preserving every engineering detail — algorithms, formulas, architectural patterns, queue topologies, database schemas, code module references, and Mermaid diagrams.
|
||||
|
||||
The deliverable is a documentation-only transformation. No application code, database schemas, or infrastructure changes are involved. The output is Markdown files and Mermaid diagram files that mirror the original structure with domain-neutral framing.
|
||||
|
||||
**Key design decision**: The sanitization is a manual content transformation guided by a defined terminology map. Each source file is read, transformed according to the mapping rules, and written to the output directory. The original files remain untouched.
|
||||
|
||||
### Source Material
|
||||
|
||||
The source documentation at `docs/intelligence-pipeline-deep-dive/` consists of:
|
||||
|
||||
| File | Content |
|
||||
|------|---------|
|
||||
| `index.md` | Table of contents, introduction, diagram links, related docs |
|
||||
| `01-data-ingestion-and-preparation.md` | Scheduler, ingestion worker, deduplication, parser |
|
||||
| `02-ai-agent-processing-and-extraction.md` | Document extractor, event classifier, JSON repair, validation |
|
||||
| `03-signal-scoring-and-weighted-signals.md` | Composite weight formula, three signal layers, sentiment mapping |
|
||||
| `04-trend-aggregation-and-accumulating-signals.md` | Time windows, trend direction, contradiction, evidence ranking, confidence |
|
||||
| `05-recommendation-generation.md` | Suppression, eligibility, position sizing, thesis, risk classification |
|
||||
| `06-trading-decisions-and-execution.md` | Trading engine, pre-trade checks, circuit breakers, broker adapter |
|
||||
| `diagrams/ingestion-to-extraction-flow.md` | Mermaid flowchart: scheduler → ingestion → parser → extractor |
|
||||
| `diagrams/three-layer-signal-merging.md` | Mermaid flowchart: three signal layers → aggregation |
|
||||
| `diagrams/weighted-signal-computation.md` | Mermaid flowchart: composite weight formula breakdown |
|
||||
| `diagrams/trend-accumulation-escalation.md` | Mermaid flowchart: time windows → escalation path |
|
||||
| `diagrams/recommendation-generation-flow.md` | Mermaid flowchart: suppression → eligibility → thesis → risk |
|
||||
| `diagrams/trading-engine-decision-loop.md` | Mermaid flowchart: pre-trade checks → position sizing → order submission |
|
||||
|
||||
## Architecture
|
||||
|
||||
### Output File Organization
|
||||
|
||||
The sanitized docs mirror the source structure with sanitized filenames:
|
||||
|
||||
```
|
||||
docs/sanitized-pipeline-deep-dive/
|
||||
├── index.md
|
||||
├── 01-data-ingestion-and-preparation.md
|
||||
├── 02-ai-agent-processing-and-extraction.md
|
||||
├── 03-signal-scoring-and-weighted-signals.md
|
||||
├── 04-trend-aggregation-and-accumulating-signals.md
|
||||
├── 05-recommendation-generation.md
|
||||
├── 06-decision-execution.md
|
||||
└── diagrams/
|
||||
├── ingestion-to-extraction-flow.md
|
||||
├── three-layer-signal-merging.md
|
||||
├── weighted-signal-computation.md
|
||||
├── trend-accumulation-escalation.md
|
||||
├── recommendation-generation-flow.md
|
||||
└── decision-engine-loop.md
|
||||
```
|
||||
|
||||
**Filename changes from source:**
|
||||
- `06-trading-decisions-and-execution.md` → `06-decision-execution.md` (removes "trading")
|
||||
- `diagrams/trading-engine-decision-loop.md` → `diagrams/decision-engine-loop.md` (removes "trading")
|
||||
- All other filenames are already domain-neutral and remain unchanged
|
||||
|
||||
### Transformation Process
|
||||
|
||||
The sanitization follows a three-pass approach for each file:
|
||||
|
||||
1. **Terminology pass**: Apply the terminology map to replace all financial/trading terms with domain-neutral equivalents. This covers inline text, headings, table cells, code blocks, and Mermaid diagram labels.
|
||||
2. **Reference pass**: Update all internal cross-references to point to sanitized filenames (e.g., `06-trading-decisions-and-execution.md` → `06-decision-execution.md`, `trading-engine-decision-loop.md` → `decision-engine-loop.md`). Remove or neutralize references to external financial docs (e.g., links to `../llm-to-trade-pipeline.md` become neutral descriptions).
|
||||
3. **Narrative pass**: Reframe example scenarios, inline illustrations, and narrative framing to use domain-neutral language. This pass handles context-dependent replacements that a simple find-and-replace cannot catch — e.g., "a bearish article about AAPL" becomes "a negative-sentiment article about Entity-A".
|
||||
|
||||
### Content Flow
|
||||
|
||||
The sanitized docs preserve the same page-to-page narrative flow as the originals:
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
P1["Page 1\nData Ingestion"] --> P2["Page 2\nAI Extraction"]
|
||||
P2 --> P3["Page 3\nSignal Scoring"]
|
||||
P3 --> P4["Page 4\nTrend Aggregation"]
|
||||
P4 --> P5["Page 5\nRecommendations"]
|
||||
P5 --> P6["Page 6\nDecision Execution"]
|
||||
```
|
||||
|
||||
## Components and Interfaces
|
||||
|
||||
### Terminology Map
|
||||
|
||||
The core of the sanitization is a defined mapping from financial/trading terms to domain-neutral equivalents. The map is applied consistently across all files.
|
||||
|
||||
#### System and Provider Names
|
||||
|
||||
| Source Term | Sanitized Replacement |
|
||||
|-------------|----------------------|
|
||||
| Stonks Oracle / stonks | the platform / the system |
|
||||
| Polygon.io / Polygon | external data provider / data source API |
|
||||
| SEC EDGAR / SEC / EFTS | public records API / regulatory filings source |
|
||||
| Alpaca / AlpacaBrokerAdapter | execution adapter / external execution API |
|
||||
| Wall Street | (removed or reframed) |
|
||||
|
||||
#### Trading and Financial Actions
|
||||
|
||||
| Source Term | Sanitized Replacement |
|
||||
|-------------|----------------------|
|
||||
| buy | act |
|
||||
| sell | defer |
|
||||
| hold | monitor |
|
||||
| watch | observe |
|
||||
| trading engine | decision execution engine |
|
||||
| paper trading / paper_eligible | simulation mode / simulation_eligible |
|
||||
| live trading / live_eligible | live execution mode / production_eligible |
|
||||
| trade / trading (as action) | decision / execution |
|
||||
| order (broker order) | execution request |
|
||||
| pre-trade checks | pre-execution checks |
|
||||
|
||||
#### Financial Concepts
|
||||
|
||||
| Source Term | Sanitized Replacement |
|
||||
|-------------|----------------------|
|
||||
| portfolio | resource pool / allocation pool |
|
||||
| portfolio allocation | resource allocation |
|
||||
| portfolio heat | pool exposure |
|
||||
| portfolio snapshots | pool snapshots |
|
||||
| position sizing | commitment sizing / resource allocation |
|
||||
| position (open position) | commitment / active commitment |
|
||||
| stop-loss | risk threshold / loss limit |
|
||||
| take-profit | gain target |
|
||||
| bullish | positive / favorable |
|
||||
| bearish | negative / unfavorable |
|
||||
| stock ticker / ticker symbol | entity identifier |
|
||||
| stock market | (removed or reframed) |
|
||||
| earnings / earnings call / earnings report | performance report / periodic disclosure |
|
||||
| 10-K / 10-Q / 8-K | regulatory filing types |
|
||||
| SEC filings | regulatory filings |
|
||||
| broker / broker API | execution adapter / execution API |
|
||||
| P&L | gain/loss |
|
||||
| Sharpe ratio | risk-adjusted return ratio |
|
||||
| drawdown | peak-to-trough decline |
|
||||
| win rate | success rate |
|
||||
|
||||
#### Ticker Symbols and Company Names
|
||||
|
||||
| Source Term | Sanitized Replacement |
|
||||
|-------------|----------------------|
|
||||
| AAPL / Apple | Entity-A |
|
||||
| TSLA / Tesla | Entity-B |
|
||||
| NVDA / NVIDIA | Entity-C |
|
||||
| XOM | Entity-D |
|
||||
| META | Entity-E |
|
||||
| Any other ticker | Entity-{letter} or "tracked entity" |
|
||||
|
||||
#### Redis Keys
|
||||
|
||||
| Source Pattern | Sanitized Pattern |
|
||||
|----------------|-------------------|
|
||||
| `stonks:queue:*` | `app:queue:*` |
|
||||
| `stonks:dedupe:*` | `app:dedupe:*` |
|
||||
| `stonks:ratelimit:*` | `app:ratelimit:*` |
|
||||
| `stonks:trading:circuit_breaker:*` | `app:execution:circuit_breaker:*` |
|
||||
| `stonks:dedupe:trading:*` | `app:dedupe:execution:*` |
|
||||
|
||||
#### MinIO Buckets
|
||||
|
||||
| Source Bucket | Sanitized Bucket |
|
||||
|---------------|-----------------|
|
||||
| `stonks-raw-market` | `app-raw-data` |
|
||||
| `stonks-raw-news` | `app-raw-content` |
|
||||
| `stonks-raw-filings` | `app-raw-filings` |
|
||||
| `stonks-normalized` | `app-normalized` |
|
||||
| `stonks-llm-prompts` | `app-llm-prompts` |
|
||||
| `stonks-llm-results` | `app-llm-results` |
|
||||
|
||||
#### Database Tables
|
||||
|
||||
| Source Table | Sanitized Table |
|
||||
|-------------|----------------|
|
||||
| `trading_decisions` | `execution_decisions` |
|
||||
| `portfolio_snapshots` | `pool_snapshots` |
|
||||
| `portfolio_pct` (column) | `allocation_pct` |
|
||||
|
||||
All other table names (`documents`, `document_intelligence`, `trend_windows`, `recommendations`, etc.) are already domain-neutral and remain unchanged.
|
||||
|
||||
#### Adapter and Source Type Names
|
||||
|
||||
| Source Term | Sanitized Replacement |
|
||||
|-------------|----------------------|
|
||||
| `PolygonNewsAdapter` | `ExternalNewsAdapter` |
|
||||
| `PolygonMarketAdapter` | `ExternalDataAdapter` |
|
||||
| `SECEdgarAdapter` | `RegulatoryFilingsAdapter` |
|
||||
| `AlpacaBrokerAdapter` | `ExecutionAdapter` |
|
||||
| `broker` (source_type) | `execution_api` |
|
||||
| `market_api` (source_type) | `data_api` |
|
||||
| `filings_api` (source_type) | `filings_api` (unchanged — already neutral) |
|
||||
|
||||
### Preserved Engineering Terms
|
||||
|
||||
The following terms are explicitly preserved because they describe engineering patterns, not financial concepts:
|
||||
|
||||
- **circuit breaker** — engineering safety pattern for rate limiting and cascading failure prevention
|
||||
- **exponential backoff** — retry pattern
|
||||
- **adapter pattern** — software design pattern (only the domain-specific adapter *names* are sanitized)
|
||||
- **signal** — used in signal processing and scoring context
|
||||
- **trend**, **sentiment**, **confidence**, **contradiction**, **evidence** — data analysis terms
|
||||
- **recency decay**, **credibility weight**, **novelty bonus** — scoring algorithm terms
|
||||
- **weighted sentiment average** — mathematical computation term
|
||||
|
||||
### Preserved Technical Content
|
||||
|
||||
All of the following are preserved verbatim (with only the terminology map applied to embedded financial terms):
|
||||
|
||||
- Composite signal scoring formula: `combined = gate × recency × credibility × (1 + novelty_bonus) × market_context_multiplier`
|
||||
- Confidence computation formula with log₂ scaling and four components
|
||||
- Weighted sentiment average formula
|
||||
- All threshold values, configuration parameters, and numeric constants
|
||||
- All Markdown table structures containing technical parameters
|
||||
- All code module path references (e.g., `services/aggregation/scoring.py`)
|
||||
- Three-layer signal architecture with weight ratios (1.0, 0.3, 0.2)
|
||||
- Contradiction detection algorithm and evidence ranking methodology
|
||||
- All PostgreSQL table structures and column descriptions (with sanitized names where needed)
|
||||
- All Redis queue patterns and operations (`rpush`/`lpop`/`blpop`)
|
||||
- All MinIO storage patterns (with sanitized bucket names)
|
||||
- Ollama as the LLM inference provider
|
||||
|
||||
### Index Page Reframing
|
||||
|
||||
The sanitized `index.md` describes the system as an "AI-driven intelligence-to-decision pipeline" that:
|
||||
1. Ingests data from multiple external data sources
|
||||
2. Extracts structured intelligence via NLP/LLM
|
||||
3. Scores and weights signals
|
||||
4. Aggregates trends across time windows
|
||||
5. Generates recommendations with quality gates
|
||||
6. Executes decisions autonomously with safety mechanisms
|
||||
|
||||
References to "Stonks Oracle" are replaced with "the platform" or "the system". References to financial-specific APIs (Polygon.io, SEC EDGAR) are replaced with neutral descriptions. The "Related Documentation" section links are updated to use neutral descriptions or removed if they reference financial-specific content.
|
||||
|
||||
### Page 06 Reframing
|
||||
|
||||
Page 06 undergoes the most extensive reframing since it covers the trading engine. Key changes:
|
||||
- Title: "Decision Execution" instead of "Trading Decisions and Execution"
|
||||
- "Trading engine" → "decision execution engine"
|
||||
- "Pre-trade checks" → "pre-execution checks"
|
||||
- "Broker adapter" / "Alpaca" → "execution adapter" / "external execution API"
|
||||
- "Paper trading" → "simulation mode"
|
||||
- "Live trading" → "live execution mode"
|
||||
- "Portfolio" → "resource pool" / "allocation pool"
|
||||
- "Position" → "commitment" / "active commitment"
|
||||
- "Stop-loss" → "risk threshold"
|
||||
- "Take-profit" → "gain target"
|
||||
- All order submission language reframed as "execution request submission"
|
||||
|
||||
### Diagram Sanitization
|
||||
|
||||
Each Mermaid diagram file receives the same terminology map treatment:
|
||||
- Node labels containing financial terms are replaced
|
||||
- Queue name labels (`stonks:queue:*` → `app:queue:*`)
|
||||
- Bucket name labels (`stonks-raw-market` → `app-raw-data`)
|
||||
- Table name labels (`trading_decisions` → `execution_decisions`)
|
||||
- Adapter names in node labels
|
||||
- Subgraph titles containing financial terms
|
||||
- The `trading-engine-decision-loop.md` diagram is renamed to `decision-engine-loop.md`
|
||||
|
||||
Mermaid syntax, node relationships, subgraph structures, and flow directions are preserved exactly.
|
||||
|
||||
## Data Models
|
||||
|
||||
This feature produces only documentation files. There are no new data models, database tables, or schema changes.
|
||||
|
||||
The sanitized narrative pages reference the same data models as the originals, with terminology-mapped names where applicable:
|
||||
|
||||
- **`WeightedSignal`** — document reference + composite weight + sentiment + impact (unchanged)
|
||||
- **`SignalWeight`** — breakdown of recency, credibility, novelty, confidence gate, market context multiplier (unchanged)
|
||||
- **`TrendSummary`** — rolling trend for an entity across a time window (unchanged)
|
||||
- **`Recommendation`** — actionable decision recommendation (reframed from "trade recommendation")
|
||||
- **`execution_decisions`** table — audit record of every decision evaluation (sanitized from `trading_decisions`)
|
||||
- **`pool_snapshots`** table — resource pool state snapshots (sanitized from `portfolio_snapshots`)
|
||||
|
||||
|
||||
## Correctness Properties
|
||||
|
||||
*A property is a characteristic or behavior that should hold true across all valid executions of a system — essentially, a formal statement about what the system should do. Properties serve as the bridge between human-readable specifications and machine-verifiable correctness guarantees.*
|
||||
|
||||
The sanitized documentation set has one key universal property: the complete absence of financial/trading terminology across all output files. This is well-suited to property-based testing because the property must hold for *every* file in the output set, and the banned term list is large enough that systematic checking across all files provides high-value coverage.
|
||||
|
||||
### Property 1: Banned Financial Terminology Exclusion
|
||||
|
||||
*For any* file in the sanitized documentation set (`docs/sanitized-pipeline-deep-dive/`), the file content shall not contain any term from the comprehensive banned financial terminology list. The banned list includes: stock ticker symbols (AAPL, TSLA, NVDA, XOM, META, and all 50 tracked tickers), company names used as financial examples (Apple, Tesla, NVIDIA), trading action labels (buy, sell, hold, watch as action labels — BUY, SELL, HOLD, WATCH in uppercase), financial system terms (trading engine, paper trading, live trading, paper_eligible, live_eligible, portfolio, portfolio allocation, portfolio heat, portfolio snapshots, broker, Alpaca, broker adapter, broker API, stock market, Wall Street, bullish, bearish, position sizing, stop-loss), financial event terms (SEC EDGAR, SEC filings, 10-K, 10-Q, 8-K, earnings, earnings call, earnings report), provider names (Polygon.io, Polygon), system names (Stonks Oracle, stonks), and infrastructure patterns containing financial terms (stonks: prefix in Redis keys, stonks- prefix in MinIO buckets, trading_decisions table name, portfolio_snapshots table name).
|
||||
|
||||
**Validates: Requirements 3.1, 3.2, 3.3, 3.4, 3.5, 3.6, 3.7, 3.8, 3.9, 3.10, 6.2, 7.1, 7.2, 7.3, 8.1, 8.2**
|
||||
|
||||
## Error Handling
|
||||
|
||||
Since this is a documentation-only deliverable, there is no runtime error handling to design. The primary quality concerns are:
|
||||
|
||||
### Accuracy of Terminology Replacement
|
||||
|
||||
Every financial/trading term must be replaced with its domain-neutral equivalent. Missing a single instance of "stonks" in a Redis key pattern or "AAPL" in an example scenario would violate the sanitization requirements. The terminology map defined in the Components section serves as the authoritative reference.
|
||||
|
||||
### Preservation of Technical Content
|
||||
|
||||
The sanitization must not accidentally remove or alter engineering content. Key risks:
|
||||
- **Formula corruption**: The composite weight formula contains `market_context_multiplier` — the word "market" must not be blindly replaced since it's part of a technical variable name
|
||||
- **Code path corruption**: Module paths like `services/trading/engine.py` contain "trading" — these paths reference actual files and must be preserved as-is (the code files are not being renamed)
|
||||
- **Table name corruption**: Database table names like `trading_decisions` need sanitization in narrative text but the actual SQL/code references to the original table names should be handled carefully
|
||||
|
||||
**Design decision**: Code module paths (e.g., `services/trading/engine.py`) are preserved exactly as they appear in the source, since they reference actual files in the repository. Only narrative references to concepts (e.g., "the trading engine") are sanitized. Variable names within formulas and code blocks are preserved. Database table names are sanitized in narrative descriptions and table listings, but inline code references note the sanitized name.
|
||||
|
||||
### Cross-Reference Integrity
|
||||
|
||||
All internal links must resolve to files that exist in the sanitized output:
|
||||
- Page-to-page links must use sanitized filenames
|
||||
- Diagram links must use sanitized diagram filenames
|
||||
- No links should point back to the source `docs/intelligence-pipeline-deep-dive/` directory
|
||||
|
||||
## Testing Strategy
|
||||
|
||||
### Why Limited PBT Applies
|
||||
|
||||
This is a documentation-only deliverable — the output is static Markdown files, not executable code with functions and data transformations. However, one universal property (banned term exclusion) is well-suited to property-based testing because it must hold across all files and involves checking a large set of terms against file content.
|
||||
|
||||
Most other requirements (structural checks, content preservation, narrative reframing) are better verified through example-based tests and manual review.
|
||||
|
||||
### Property-Based Tests
|
||||
|
||||
- **Library**: Hypothesis (Python, already in the project)
|
||||
- **Configuration**: `@settings(max_examples=100)`
|
||||
- **Property 1 implementation**: Generate random selections from the banned term list and random file selections from the sanitized docs, verify the term does not appear in the file content. Alternatively, exhaustively check all banned terms against all files (since the file set is small and fixed, this is more practical as an exhaustive example-based test).
|
||||
|
||||
**Practical note**: Given the small, fixed file set (14 files), the banned term exclusion property is most practically implemented as an exhaustive check — iterate all files × all banned terms — rather than a randomized property test. This provides complete coverage rather than probabilistic coverage.
|
||||
|
||||
### Example-Based Tests
|
||||
|
||||
1. **File structure verification**: Verify all expected files exist at the correct paths
|
||||
2. **Cross-reference integrity**: Parse all sanitized files, extract markdown links, verify they resolve to existing sanitized files
|
||||
3. **Mermaid syntax validation**: Verify each diagram file contains valid Mermaid `flowchart` declarations
|
||||
4. **Technical content preservation**: Spot-check that key formulas, threshold values, and code module paths are present in the sanitized docs
|
||||
5. **Terminology replacement verification**: Spot-check that key replacements appear (e.g., "decision execution engine" replaces "trading engine")
|
||||
6. **Index page framing**: Verify the index describes the system as an "AI-driven intelligence-to-decision pipeline"
|
||||
7. **Database table sanitization**: Verify `execution_decisions` appears where `trading_decisions` was, and `pool_snapshots` where `portfolio_snapshots` was
|
||||
|
||||
### Manual Review
|
||||
|
||||
- Narrative coherence and readability of the sanitized content
|
||||
- Consistency of domain-neutral framing across all pages
|
||||
- Quality of example scenario replacements (e.g., "bearish article about AAPL" → "negative-sentiment article about Entity-A")
|
||||
- Preservation of page-to-page transition flow
|
||||
@@ -0,0 +1,202 @@
|
||||
# Requirements Document
|
||||
|
||||
## Introduction
|
||||
|
||||
This feature produces a sanitized version of the existing 6-page intelligence pipeline deep dive documentation (`docs/intelligence-pipeline-deep-dive/`) for use in a work presentation. The sanitized version strips all financial, market, and trading language — stock tickers, buy/sell/hold actions, portfolio allocation, broker APIs, and domain-specific framing — and reframes the content as a general-purpose AI decision intelligence pipeline. The sanitized docs are stored as a separate doc group under `docs/sanitized-pipeline-deep-dive/`, preserving the original documents untouched. All engineering depth — algorithms, formulas, architectural patterns, queue topologies, database schemas, code module references, and Mermaid diagrams — is preserved. Only the domain-specific framing changes.
|
||||
|
||||
## Glossary
|
||||
|
||||
- **Source_Docs**: The original 6-page documentation set at `docs/intelligence-pipeline-deep-dive/`, including `index.md`, pages `01` through `06`, and the `diagrams/` subdirectory containing 6 Mermaid diagram files.
|
||||
- **Sanitized_Docs**: The output documentation set at `docs/sanitized-pipeline-deep-dive/`, mirroring the structure of Source_Docs with all financial/market/trading language replaced by domain-neutral equivalents.
|
||||
- **Sanitization_Engine**: The process (manual or automated) that transforms Source_Docs into Sanitized_Docs by applying the terminology mapping and content reframing rules defined in this document.
|
||||
- **Terminology_Map**: The defined set of financial/market/trading terms and their domain-neutral replacements used by the Sanitization_Engine.
|
||||
- **Entity_Identifier**: The domain-neutral replacement for stock ticker symbols (e.g., AAPL, TSLA) in Sanitized_Docs.
|
||||
- **Decision_Term**: A domain-neutral action term (act, defer, monitor, observe) that replaces trading actions (buy, sell, hold, watch) in Sanitized_Docs.
|
||||
- **Decision_Execution_Engine**: The domain-neutral name for the trading engine in Sanitized_Docs.
|
||||
- **Execution_Adapter**: The domain-neutral name for broker adapters and broker API references in Sanitized_Docs.
|
||||
- **Allocation_Pool**: The domain-neutral name for portfolio references in Sanitized_Docs.
|
||||
- **Commitment_Sizing**: The domain-neutral name for position sizing in Sanitized_Docs.
|
||||
|
||||
---
|
||||
|
||||
## Requirements
|
||||
|
||||
### Requirement 1: Separate Output Directory
|
||||
|
||||
**User Story:** As a presenter, I want the sanitized docs stored in a separate directory from the originals, so that the original documentation remains untouched and both versions coexist.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Sanitization_Engine SHALL write all output files to `docs/sanitized-pipeline-deep-dive/`.
|
||||
2. THE Sanitization_Engine SHALL NOT modify, overwrite, or delete any file under `docs/intelligence-pipeline-deep-dive/`.
|
||||
3. THE Sanitized_Docs SHALL contain an `index.md` file at the root of `docs/sanitized-pipeline-deep-dive/`.
|
||||
4. THE Sanitized_Docs SHALL contain a `diagrams/` subdirectory under `docs/sanitized-pipeline-deep-dive/`.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 2: Mirror the 6-Page Structure
|
||||
|
||||
**User Story:** As a presenter, I want the sanitized docs to mirror the same 6-page structure as the originals, so that readers familiar with the original can navigate the sanitized version identically.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Sanitized_Docs SHALL contain exactly 6 numbered page files matching the naming pattern of Source_Docs: `01-*.md` through `06-*.md`.
|
||||
2. THE Sanitized_Docs SHALL contain an `index.md` with a table of contents linking to all 6 pages and all diagrams, mirroring the structure of the Source_Docs index.
|
||||
3. THE Sanitized_Docs SHALL contain one Mermaid diagram file in `diagrams/` for each diagram file present in `docs/intelligence-pipeline-deep-dive/diagrams/`.
|
||||
4. WHEN a Source_Docs page contains internal cross-references to other pages or diagrams, THE Sanitized_Docs equivalent page SHALL contain corresponding cross-references pointing to the Sanitized_Docs versions of those pages and diagrams.
|
||||
5. THE Sanitized_Docs page filenames SHALL use sanitized titles (e.g., `06-decision-execution.md` instead of `06-trading-decisions-and-execution.md`).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 3: Strip Financial and Trading Terminology
|
||||
|
||||
**User Story:** As a presenter, I want all financial, market, and trading language removed from the sanitized docs, so that the presentation focuses on engineering without revealing the financial domain.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Sanitized_Docs SHALL NOT contain any stock ticker symbols (e.g., AAPL, TSLA, NVDA, XOM, META).
|
||||
2. THE Sanitized_Docs SHALL NOT contain the trading action terms "buy", "sell", "hold", or "watch" when used as system action labels or decision outputs.
|
||||
3. THE Sanitized_Docs SHALL NOT contain the terms "trading engine", "paper trading", "live trading", "paper_eligible", or "live_eligible".
|
||||
4. THE Sanitized_Docs SHALL NOT contain the terms "portfolio", "portfolio allocation", "portfolio heat", or "portfolio snapshots" when referring to the resource management domain concept.
|
||||
5. THE Sanitized_Docs SHALL NOT contain references to "broker", "Alpaca", "broker adapter", or "broker API".
|
||||
6. THE Sanitized_Docs SHALL NOT contain the terms "stock market", "Wall Street", "bullish", "bearish", "position sizing" (as a financial concept label), or "stop-loss" (as a financial concept label).
|
||||
7. THE Sanitized_Docs SHALL NOT contain company names used as financial examples (e.g., "Apple", "Tesla", "NVIDIA" when used in a stock/market context).
|
||||
8. THE Sanitized_Docs SHALL NOT contain the terms "SEC EDGAR", "SEC filings", "10-K", "10-Q", "8-K", "earnings", "earnings call", or "earnings report" as domain-specific financial references.
|
||||
9. THE Sanitized_Docs SHALL NOT contain references to "Polygon.io" or "Polygon" as a financial data provider name.
|
||||
10. THE Sanitized_Docs SHALL NOT contain the term "Stonks Oracle" or "stonks" as a system name.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 4: Apply Domain-Neutral Terminology Mapping
|
||||
|
||||
**User Story:** As a presenter, I want consistent domain-neutral replacements for all stripped terms, so that the sanitized docs read coherently as a general-purpose AI decision intelligence pipeline.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN the Source_Docs use "stock ticker" or specific ticker symbols, THE Sanitized_Docs SHALL use "entity identifier" or "tracked entity".
|
||||
2. WHEN the Source_Docs use "buy/sell/hold/watch" as action labels, THE Sanitized_Docs SHALL use "act/defer/monitor/observe" or equivalent neutral decision terms.
|
||||
3. WHEN the Source_Docs use "trading engine", THE Sanitized_Docs SHALL use "decision execution engine" or "action engine".
|
||||
4. WHEN the Source_Docs use "portfolio", THE Sanitized_Docs SHALL use "resource pool" or "allocation pool".
|
||||
5. WHEN the Source_Docs use "broker" or "Alpaca", THE Sanitized_Docs SHALL use "execution adapter" or "external execution API".
|
||||
6. WHEN the Source_Docs use "paper trading", THE Sanitized_Docs SHALL use "simulation mode" or "dry-run mode".
|
||||
7. WHEN the Source_Docs use "live trading", THE Sanitized_Docs SHALL use "live execution mode" or "production mode".
|
||||
8. WHEN the Source_Docs use "bullish" or "bearish", THE Sanitized_Docs SHALL use "positive" or "negative" (or "favorable"/"unfavorable").
|
||||
9. WHEN the Source_Docs use "position sizing", THE Sanitized_Docs SHALL use "resource allocation" or "commitment sizing".
|
||||
10. WHEN the Source_Docs use "stop-loss", THE Sanitized_Docs SHALL use "risk threshold" or "loss limit".
|
||||
11. WHEN the Source_Docs use "Stonks Oracle" or "stonks", THE Sanitized_Docs SHALL use a neutral system name such as "the platform" or "the system".
|
||||
12. WHEN the Source_Docs use "SEC EDGAR" or "SEC filings", THE Sanitized_Docs SHALL use "regulatory filings source" or "public records API".
|
||||
13. WHEN the Source_Docs use "Polygon.io" or "Polygon", THE Sanitized_Docs SHALL use "external data provider" or "data source API".
|
||||
14. WHEN the Source_Docs use "earnings" as a catalyst type or event, THE Sanitized_Docs SHALL use "performance report" or "periodic disclosure".
|
||||
15. THE Sanitized_Docs SHALL apply the Terminology_Map consistently across all 6 pages, the index, and all diagram files.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 5: Preserve Engineering and Technical Depth
|
||||
|
||||
**User Story:** As a presenter, I want all engineering concepts, algorithms, formulas, and architectural details preserved, so that the sanitized docs demonstrate the technical sophistication of the system.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Sanitized_Docs SHALL preserve all references to Redis queue patterns, including queue names and `rpush`/`lpop`/`blpop` operations.
|
||||
2. THE Sanitized_Docs SHALL preserve all references to PostgreSQL tables, including table names and column descriptions.
|
||||
3. THE Sanitized_Docs SHALL preserve all references to MinIO buckets and storage patterns.
|
||||
4. THE Sanitized_Docs SHALL preserve all references to Ollama as the LLM inference provider.
|
||||
5. THE Sanitized_Docs SHALL preserve the composite signal scoring formula: `combined = gate × recency × credibility × (1 + novelty_bonus) × market_context_multiplier`.
|
||||
6. THE Sanitized_Docs SHALL preserve the confidence computation formula with log₂ scaling and its four components (unique source count, average extraction credibility, signal agreement with sample-size dampening, contradiction penalty).
|
||||
7. THE Sanitized_Docs SHALL preserve the weighted sentiment average formula: `weighted_avg = Σ(combined_weight × impact_score × sentiment_value) / Σ(combined_weight × impact_score)`.
|
||||
8. THE Sanitized_Docs SHALL preserve all code module path references (e.g., `services/aggregation/scoring.py`, `services/recommendation/eligibility.py`).
|
||||
9. THE Sanitized_Docs SHALL preserve the three-layer signal architecture, renaming the layers with domain-neutral labels (e.g., "Entity-Specific Signals", "Environmental Signals", "Relational Signals") while retaining the weight ratios (1.0, 0.3, 0.2).
|
||||
10. THE Sanitized_Docs SHALL preserve all threshold values, configuration parameters, and numeric constants (e.g., confidence gate of 0.2, recency half-lives per window, eligibility thresholds).
|
||||
11. THE Sanitized_Docs SHALL preserve all Markdown table structures containing technical parameters and thresholds.
|
||||
12. THE Sanitized_Docs SHALL preserve the contradiction detection algorithm, evidence ranking methodology, and trend projection computation.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 6: Sanitize Mermaid Diagrams
|
||||
|
||||
**User Story:** As a presenter, I want the Mermaid diagrams sanitized with the same terminology mapping as the narrative pages, so that diagrams and text are consistent.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Sanitized_Docs SHALL contain one sanitized Mermaid diagram file for each of the 6 diagram files in Source_Docs.
|
||||
2. WHEN a Source_Docs diagram contains financial/trading terminology (e.g., "trading engine", "buy/sell", "paper_eligible", "bullish/bearish", ticker symbols), THE corresponding Sanitized_Docs diagram SHALL use the same domain-neutral replacements defined in the Terminology_Map.
|
||||
3. THE Sanitized_Docs diagrams SHALL preserve all Mermaid syntax, node relationships, subgraph structures, and flow directions from the Source_Docs diagrams.
|
||||
4. THE Sanitized_Docs diagrams SHALL preserve all code module path references and service names within diagram nodes.
|
||||
5. THE Sanitized_Docs diagram filenames SHALL use sanitized names where the original names contain financial terms (e.g., `decision-engine-loop.md` instead of `trading-engine-decision-loop.md`).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 7: Sanitize Redis Key and Queue Name References
|
||||
|
||||
**User Story:** As a presenter, I want Redis key patterns and queue names sanitized where they contain financial terms, so that even infrastructure-level references are domain-neutral.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN a Source_Docs Redis queue name contains "stonks" (e.g., `stonks:queue:ingestion`), THE Sanitized_Docs SHALL replace "stonks" with a neutral prefix (e.g., `app:queue:ingestion`).
|
||||
2. WHEN a Source_Docs Redis key pattern contains "trading" (e.g., `stonks:queue:broker_orders`, `stonks:trading:circuit_breaker:*`), THE Sanitized_Docs SHALL replace the trading-specific segment with a neutral equivalent (e.g., `app:queue:execution_orders`, `app:execution:circuit_breaker:*`).
|
||||
3. THE Sanitized_Docs SHALL apply Redis key sanitization consistently across all narrative pages and diagram files.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 8: Sanitize MinIO Bucket Name References
|
||||
|
||||
**User Story:** As a presenter, I want MinIO bucket names sanitized where they contain financial terms, so that storage references are domain-neutral.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN a Source_Docs MinIO bucket name contains "stonks" (e.g., `stonks-raw-market`, `stonks-raw-news`, `stonks-normalized`), THE Sanitized_Docs SHALL replace "stonks" with a neutral prefix (e.g., `app-raw-data`, `app-raw-content`, `app-normalized`).
|
||||
2. THE Sanitized_Docs SHALL apply MinIO bucket name sanitization consistently across all narrative pages and diagram files.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 9: Sanitize Database Table and Column References Where Needed
|
||||
|
||||
**User Story:** As a presenter, I want database table and column names that contain obvious financial terms sanitized, while preserving the overall schema structure.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN a Source_Docs database table name contains "trading" (e.g., `trading_decisions`), THE Sanitized_Docs SHALL use a neutral equivalent (e.g., `execution_decisions`).
|
||||
2. WHEN a Source_Docs database table or column references "portfolio" (e.g., `portfolio_snapshots`, `portfolio_pct`), THE Sanitized_Docs SHALL use a neutral equivalent (e.g., `pool_snapshots`, `allocation_pct`).
|
||||
3. THE Sanitized_Docs SHALL preserve all other database table names that do not contain financial-specific terms (e.g., `documents`, `document_intelligence`, `trend_windows`, `recommendations`).
|
||||
4. THE Sanitized_Docs SHALL apply database reference sanitization consistently across all narrative pages.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 10: Sanitize Example Scenarios and Inline References
|
||||
|
||||
**User Story:** As a presenter, I want all inline examples, scenario walkthroughs, and narrative references sanitized, so that no financial context leaks through illustrative content.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN a Source_Docs page uses a specific company name or ticker in an example scenario (e.g., "a bearish article about AAPL"), THE Sanitized_Docs SHALL replace the reference with a generic entity (e.g., "a negative-sentiment article about Entity-A").
|
||||
2. WHEN a Source_Docs page describes a financial event as an example (e.g., "earnings miss", "tariff announcement affecting XOM"), THE Sanitized_Docs SHALL reframe the example using domain-neutral language (e.g., "a negative performance disclosure", "a regulatory policy change affecting Entity-B").
|
||||
3. WHEN a Source_Docs page references market-specific concepts in narrative flow (e.g., "markets move fast", "trading volume", "intraday swings"), THE Sanitized_Docs SHALL reframe using neutral language (e.g., "conditions change rapidly", "activity volume", "short-term fluctuations").
|
||||
4. THE Sanitized_Docs SHALL preserve the logical structure and teaching purpose of all example scenarios while removing the financial framing.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 11: Preserve Acceptable Engineering Terms
|
||||
|
||||
**User Story:** As a presenter, I want general engineering terms that happen to overlap with financial language preserved when they describe engineering patterns, so that the technical accuracy is maintained.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Sanitized_Docs SHALL preserve the term "circuit breaker" when it describes the engineering safety pattern (rate limiting, cascading failure prevention).
|
||||
2. THE Sanitized_Docs SHALL preserve the term "exponential backoff" and all retry/backoff patterns.
|
||||
3. THE Sanitized_Docs SHALL preserve all adapter pattern references (the software design pattern), renaming only the domain-specific adapter names (e.g., "AlpacaBrokerAdapter" becomes a neutral name).
|
||||
4. THE Sanitized_Docs SHALL preserve the term "signal" as used in the signal processing and scoring context.
|
||||
5. THE Sanitized_Docs SHALL preserve the terms "trend", "sentiment", "confidence", "contradiction", and "evidence" as used in the data analysis context.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 12: Reframe the System Narrative
|
||||
|
||||
**User Story:** As a presenter, I want the overall system narrative reframed as a general-purpose AI decision intelligence pipeline, so that the presentation tells a coherent story without financial context.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Sanitized_Docs index page SHALL describe the system as an "AI-driven intelligence-to-decision pipeline" that ingests data from multiple sources, extracts structured intelligence via NLP/LLM, scores and weights signals, aggregates trends across time windows, generates recommendations with quality gates, and executes decisions autonomously with safety mechanisms.
|
||||
2. THE Sanitized_Docs page 01 SHALL describe data ingestion from "multiple external data sources" rather than from financial-specific APIs.
|
||||
3. THE Sanitized_Docs page 06 SHALL describe "autonomous decision execution with safety mechanisms" rather than "trading decisions and execution".
|
||||
4. WHEN the Source_Docs conclusion references the "intelligence-to-decision pipeline in Stonks Oracle", THE Sanitized_Docs conclusion SHALL reference the "intelligence-to-decision pipeline" without a financial system name.
|
||||
5. THE Sanitized_Docs SHALL maintain the narrative flow where each page ends with a transition to the next page, preserving the end-to-end story structure.
|
||||
@@ -0,0 +1,47 @@
|
||||
# Tasks — Sanitized Pipeline Documentation
|
||||
|
||||
## Task 1: Create Output Directory and Index Page
|
||||
|
||||
- [x] 1.1 Create the `docs/sanitized-pipeline-deep-dive/` directory and `diagrams/` subdirectory
|
||||
- [x] 1.2 Create `docs/sanitized-pipeline-deep-dive/index.md` with sanitized content: replace "Stonks Oracle" with "the platform", replace Polygon.io/SEC EDGAR references with neutral descriptions, update all page links to use sanitized filenames (e.g., `06-decision-execution.md`), update diagram links to use sanitized names (e.g., `decision-engine-loop.md`), describe the system as an "AI-driven intelligence-to-decision pipeline", and update or remove the Related Documentation section to use neutral descriptions
|
||||
|
||||
## Task 2: Sanitize Page 01 — Data Ingestion and Preparation
|
||||
|
||||
- [x] 2.1 Create `docs/sanitized-pipeline-deep-dive/01-data-ingestion-and-preparation.md` by transforming the source page: replace "Stonks Oracle" with "the platform", replace "Polygon.io" with "external data provider", replace "SEC EDGAR"/"EFTS" with "public records API"/"regulatory filings source", replace "AlpacaBrokerAdapter" with "ExecutionAdapter", replace adapter class names (PolygonNewsAdapter → ExternalNewsAdapter, PolygonMarketAdapter → ExternalDataAdapter, SECEdgarAdapter → RegulatoryFilingsAdapter), replace all `stonks:` Redis key prefixes with `app:`, replace MinIO bucket names (stonks-raw-market → app-raw-data, stonks-raw-news → app-raw-content, stonks-raw-filings → app-raw-filings, stonks-normalized → app-normalized), replace ticker symbols (AAPL → Entity-A, etc.) and company names with generic entities, replace "broker" source_type with "execution_api", replace "SEC" references with "regulatory filings", replace "10-K"/"10-Q"/"8-K" with "regulatory filing types", replace "earnings" with "performance report", sanitize example paths (e.g., `news_api/AAPL/...` → `news_api/Entity-A/...`), update cross-references to use sanitized filenames, and preserve all engineering content (queue operations, table structures, quality scoring formula, code module paths)
|
||||
|
||||
## Task 3: Sanitize Page 02 — AI Agent Processing and Extraction
|
||||
|
||||
- [x] 3.1 Create `docs/sanitized-pipeline-deep-dive/02-ai-agent-processing-and-extraction.md` by transforming the source page: replace "Stonks Oracle" references, replace financial document type references (SEC filings → regulatory filings, earnings transcripts → performance transcripts), replace "financial document analyst" role description with "document analyst", replace ticker symbols and company names in examples (AAPL, TSLA, NVDA, XOM, META → Entity-A through Entity-E), replace "bearish"/"bullish" with "negative"/"positive", replace "earnings" catalyst type references with "performance_report", replace "stock ticker" with "entity identifier", replace "market implications" with neutral language, replace `stonks:queue:*` Redis keys with `app:queue:*`, replace MinIO bucket names (stonks-llm-prompts → app-llm-prompts, stonks-llm-results → app-llm-results, stonks-normalized → app-normalized), replace "tariff announcement affecting XOM" example with neutral equivalent, update cross-references, and preserve all engineering content (JSON repair pipeline, validation logic, AgentConfigResolver, Ollama references, code module paths, schema field descriptions)
|
||||
|
||||
## Task 4: Sanitize Page 03 — Signal Scoring and Weighted Signals
|
||||
|
||||
- [x] 4.1 Create `docs/sanitized-pipeline-deep-dive/03-signal-scoring-and-weighted-signals.md` by transforming the source page: replace "bullish"/"bearish" with "positive"/"negative" throughout, replace "trading recommendations" with "decision recommendations", replace ticker examples (AAPL, NVDA) with Entity-A/Entity-C, replace "market context" variable references carefully (preserve `market_context_multiplier` as a technical variable name but sanitize narrative references to "market conditions" → "environmental conditions"), replace "trading volume" with "activity volume", replace `stonks:queue:*` Redis keys with `app:queue:*`, replace "bullish_pct > bearish_pct" with "positive_pct > negative_pct" in signal propagation description, update cross-references, and preserve all engineering content (composite weight formula, recency decay formula, half-life tables, credibility weight computation, novelty bonus formula, weighted sentiment average formula, three-layer architecture with weight ratios 1.0/0.3/0.2, all threshold values and configuration parameters)
|
||||
|
||||
## Task 5: Sanitize Page 04 — Trend Aggregation and Accumulating Signals
|
||||
|
||||
- [x] 5.1 Create `docs/sanitized-pipeline-deep-dive/04-trend-aggregation-and-accumulating-signals.md` by transforming the source page: replace "bullish"/"bearish" with "positive"/"negative" in trend direction descriptions and TrendDirection enum values, replace "trading recommendations" with "decision recommendations", replace "BULLISH_THRESHOLD"/"BEARISH_THRESHOLD" with "POSITIVE_THRESHOLD"/"NEGATIVE_THRESHOLD", replace "paper_eligible"/"live_eligible" with "simulation_eligible"/"production_eligible", replace "paper trading"/"live trading" with "simulation mode"/"live execution mode", replace "buy"/"sell"/"hold"/"watch" action labels with "act"/"defer"/"monitor"/"observe", replace "trading_decisions" table with "execution_decisions", replace "portfolio" references, replace ticker examples (AAPL) with Entity-A, replace "earnings miss" example with "negative performance disclosure", replace `stonks:queue:*` Redis keys with `app:queue:*`, update cross-references, and preserve all engineering content (five time windows, trend direction derivation thresholds, contradiction detection algorithm, evidence ranking, confidence computation formula with log₂ scaling, trend projection computation, all persistence tables)
|
||||
|
||||
## Task 6: Sanitize Page 05 — Recommendation Generation
|
||||
|
||||
- [x] 6.1 Create `docs/sanitized-pipeline-deep-dive/05-recommendation-generation.md` by transforming the source page: replace "buy"/"sell"/"hold"/"watch" action labels with "act"/"defer"/"monitor"/"observe", replace "BUY"/"SELL"/"HOLD"/"WATCH" with "ACT"/"DEFER"/"MONITOR"/"OBSERVE", replace "paper_eligible"/"live_eligible" with "simulation_eligible"/"production_eligible", replace "paper trading"/"live trading" with "simulation mode"/"live execution mode", replace "trading engine" with "decision execution engine", replace "portfolio" with "resource pool"/"allocation pool", replace "portfolio_pct" with "allocation_pct", replace "position sizing" with "commitment sizing", replace "position" (as financial position) with "commitment", replace "stop-loss" with "risk threshold", replace "trading-eligible" with "execution-eligible", replace "trade" (as noun/verb) with "decision"/"execution", replace ticker examples (AAPL) with Entity-A, replace "earnings" catalyst references with "performance_report", replace `stonks:queue:*` Redis keys with `app:queue:*`, replace "broker adapter" with "execution adapter", replace "Alpaca" with "external execution API", update cross-references to use sanitized filenames (06-decision-execution.md), and preserve all engineering content (suppression thresholds, eligibility gates, position sizing formulas, thesis generation logic, risk classification computation, all persistence tables)
|
||||
|
||||
## Task 7: Sanitize Page 06 — Decision Execution
|
||||
|
||||
- [x] 7.1 Create `docs/sanitized-pipeline-deep-dive/06-decision-execution.md` by transforming the source page: change title to "Decision Execution", replace "trading engine" with "decision execution engine" throughout, replace "TradingEngine" class references with "DecisionEngine" in narrative (preserve code module path `services/trading/engine.py`), replace "trade"/"trading" with "decision"/"execution" in narrative, replace "pre-trade checks" with "pre-execution checks", replace "buy"/"sell" action labels with "act"/"defer", replace "paper trading"/"paper_eligible" with "simulation mode"/"simulation_eligible", replace "live trading"/"live_eligible" with "live execution mode"/"production_eligible", replace "broker"/"Alpaca" with "execution adapter"/"external execution API", replace "AlpacaBrokerAdapter" with "ExecutionAdapter" in narrative, replace "portfolio" with "resource pool"/"allocation pool", replace "portfolio heat" with "pool exposure", replace "portfolio_snapshots" with "pool_snapshots", replace "position"/"positions" (financial) with "commitment"/"commitments", replace "position sizing"/"PositionSizer" with "commitment sizing" in narrative, replace "stop-loss" with "risk threshold", replace "take-profit" with "gain target", replace "P&L" with "gain/loss", replace "Sharpe ratio" with "risk-adjusted return ratio", replace "win rate" with "success rate", replace "drawdown" with "peak-to-trough decline", replace "trading_decisions" table with "execution_decisions", replace `stonks:queue:broker_orders` with `app:queue:execution_orders`, replace `stonks:trading:circuit_breaker:*` with `app:execution:circuit_breaker:*`, replace `stonks:dedupe:trading:*` with `app:dedupe:execution:*`, replace all other `stonks:` Redis key prefixes with `app:`, replace "paper-api.alpaca.markets" with "execution-api.example.com", replace "Polygon API" with "data source API", replace ticker examples with Entity-{letter}, replace "earnings" references with "performance report"/"periodic disclosure", update cross-references to use sanitized filenames, update the Conclusion section to remove "Stonks Oracle" and financial framing, and preserve all engineering content (5 concurrent async tasks, circuit breaker algorithm, reserve pool logic, risk tier parameters table, position sizing pipeline, order submission flow, all code module paths, all threshold values)
|
||||
|
||||
## Task 8: Sanitize Mermaid Diagrams
|
||||
|
||||
- [x] 8.1 Create `docs/sanitized-pipeline-deep-dive/diagrams/ingestion-to-extraction-flow.md` by transforming the source diagram: replace `stonks:queue:*` with `app:queue:*`, replace MinIO bucket names (stonks-raw-market → app-raw-data, stonks-raw-news → app-raw-content, stonks-raw-filings → app-raw-filings, stonks-normalized → app-normalized), replace adapter names in node labels (PolygonMarketAdapter → ExternalDataAdapter, PolygonNewsAdapter → ExternalNewsAdapter, SECEdgarAdapter → RegulatoryFilingsAdapter, MacroNewsAdapter unchanged, WebScrapeAdapter unchanged), replace "AlpacaBrokerAdapter" if present, and preserve all Mermaid syntax, node relationships, subgraph structures, flow directions, and code module paths
|
||||
- [x] 8.2 Create `docs/sanitized-pipeline-deep-dive/diagrams/three-layer-signal-merging.md` by transforming the source diagram: replace `stonks:queue:*` with `app:queue:*`, replace "bullish_pct > bearish_pct" if present, and preserve all Mermaid syntax and structure
|
||||
- [x] 8.3 Create `docs/sanitized-pipeline-deep-dive/diagrams/weighted-signal-computation.md` by copying the source diagram with minimal changes (content is already domain-neutral — only replace any `stonks:` references if present), preserving all Mermaid syntax and structure
|
||||
- [x] 8.4 Create `docs/sanitized-pipeline-deep-dive/diagrams/trend-accumulation-escalation.md` by transforming the source diagram: replace "BULLISH"/"BEARISH" with "POSITIVE"/"NEGATIVE", replace "BUY / SELL" with "ACT / DEFER", replace "paper_eligible"/"live_eligible" if present, and preserve all Mermaid syntax and structure
|
||||
- [x] 8.5 Create `docs/sanitized-pipeline-deep-dive/diagrams/recommendation-generation-flow.md` by transforming the source diagram: replace `stonks:queue:*` with `app:queue:*`, replace "BUY"/"SELL"/"HOLD"/"WATCH" with "ACT"/"DEFER"/"MONITOR"/"OBSERVE", replace "paper_eligible"/"live_eligible" with "simulation_eligible"/"production_eligible", replace "portfolio" with "allocation pool", and preserve all Mermaid syntax and structure
|
||||
- [x] 8.6 Create `docs/sanitized-pipeline-deep-dive/diagrams/decision-engine-loop.md` (renamed from trading-engine-decision-loop.md) by transforming the source diagram: replace "Trading Engine" with "Decision Execution Engine", replace `stonks:queue:broker_orders` with `app:queue:execution_orders`, replace `stonks:dedupe:trading:*` with `app:dedupe:execution:*`, replace `stonks:trading:circuit_breaker:*` with `app:execution:circuit_breaker:*`, replace "buy, sell" with "act, defer", replace "paper_eligible, live_eligible" with "simulation_eligible, production_eligible", replace "Alpaca paper trading" with "external execution API (simulation)", replace "portfolio" references with "resource pool"/"allocation pool", replace "Portfolio heat" with "Pool exposure", replace "portfolio_snapshots" with "pool_snapshots", replace "trading_decisions" with "execution_decisions", replace "Sharpe ratio" with "risk-adjusted return ratio", replace "drawdown" with "peak-to-trough decline", replace "win rate" with "success rate", replace "P&L" with "gain/loss", and preserve all Mermaid syntax, node relationships, subgraph structures, flow directions, and code module paths
|
||||
|
||||
## Task 9: Verification and Cross-Reference Integrity
|
||||
|
||||
- [x] 9.1 Verify all sanitized files exist at the expected paths: index.md, 6 numbered pages (01-06), and 6 diagram files in diagrams/
|
||||
- [x] 9.2 Verify no sanitized file contains any banned financial term: scan all files for ticker symbols (AAPL, TSLA, NVDA, XOM, META), company names (Apple, Tesla, NVIDIA as financial references), system names (Stonks Oracle, stonks), provider names (Polygon.io, Polygon, SEC EDGAR, Alpaca), financial terms (trading engine, paper trading, live trading, paper_eligible, live_eligible, portfolio, broker, bullish, bearish, position sizing, stop-loss, stock market, Wall Street, earnings, 10-K, 10-Q, 8-K), and infrastructure patterns (stonks: prefix, stonks- prefix, trading_decisions, portfolio_snapshots)
|
||||
- [x] 9.3 Verify all internal cross-references resolve: parse all markdown links in sanitized files, confirm each link target exists in the sanitized output directory
|
||||
- [x] 9.4 Verify key engineering content is preserved: check that the composite weight formula, confidence computation formula, weighted sentiment average formula, three-layer weight ratios (1.0, 0.3, 0.2), and key threshold values (confidence gate 0.2, eligibility confidence 0.35) appear in the sanitized docs
|
||||
- [x] 9.5 Verify source files are unmodified: confirm that no files under `docs/intelligence-pipeline-deep-dive/` were changed
|
||||
@@ -0,0 +1 @@
|
||||
{"specId": "b595d834-7e72-4fab-87a9-65c92115a069", "workflowType": "requirements-first", "specType": "feature"}
|
||||
@@ -0,0 +1,732 @@
|
||||
# Design Document — Signal Math Upgrade
|
||||
|
||||
## Overview
|
||||
|
||||
This design upgrades the Stonks Oracle signal processing pipeline from deterministic heuristic formulas to a probabilistic, regime-aware, and adaptive mathematical framework. The upgrade spans all pipeline stages — signal scoring, trend assembly, macro impact, competitive signals, trend projection, and recommendation generation — while preserving the existing `WeightedSignal` abstraction, three-layer architecture, database schema, and dataclass interfaces.
|
||||
|
||||
The core transformation replaces:
|
||||
- **Binary confidence gate** → smooth sigmoid transition
|
||||
- **Weighted sentiment average** → Bayesian log-likelihood accumulation with Beta posterior
|
||||
- **Fixed recency decay** → adaptive event-specific half-lives
|
||||
- **Linear macro exposure** → multiplicative compounding exposure
|
||||
- **Additive macro integration** → conditional multiplicative modifiers
|
||||
- **Simple contradiction ratio** → weighted disagreement entropy
|
||||
- **Heuristic trend confidence** → Bayesian posterior variance
|
||||
- **Threshold-based direction** → entropy-based mixed signal detection
|
||||
- **Simple momentum** → exponentially weighted momentum with volatility scaling
|
||||
- **Confidence/strength gates** → expected value recommendation gate
|
||||
- **Fixed relationship transfer** → graph-distance attenuated competitive signals
|
||||
|
||||
All changes are gated behind a `probabilistic_scoring_enabled` feature flag in `risk_configs`, allowing incremental rollout with instant rollback. New outputs (P_bull, α, β, entropy, regime, EV) are stored in existing JSONB columns — no database migrations required.
|
||||
|
||||
### Design Rationale
|
||||
|
||||
Markets are fundamentally probabilistic and regime-dependent. The current pipeline collapses rich evidence into binary sentiment labels and fixed-weight averages, losing uncertainty structure. A Bayesian framework preserves the full posterior distribution, enabling the system to distinguish between "strongly bullish" and "weakly bullish with high uncertainty" — a distinction that directly impacts position sizing and risk management.
|
||||
|
||||
The regime detector adapts scoring thresholds to market conditions (panic vs. trending vs. mean-reverting), and the expected value gate ensures recommendations only proceed when the risk-adjusted outcome is positive. Together, these changes transform the pipeline from a sentiment aggregator into a probabilistic forecasting engine.
|
||||
|
||||
---
|
||||
|
||||
## Architecture
|
||||
|
||||
### High-Level Pipeline Flow
|
||||
|
||||
The upgraded pipeline maintains the existing three-layer architecture but introduces new computation stages within each layer. The feature flag controls which computation path is taken at each stage.
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
subgraph "Layer 1: Company Signals"
|
||||
A[Document Intelligence Records] --> B[Signal Scorer]
|
||||
B --> |"probabilistic=false"| C1[Binary Gate + Fixed Decay]
|
||||
B --> |"probabilistic=true"| C2[Sigmoid Gate + Adaptive Decay<br/>+ Info Gain + Source Accuracy]
|
||||
C1 --> D[WeightedSignal list]
|
||||
C2 --> D
|
||||
end
|
||||
|
||||
subgraph "Layer 2: Macro Signals"
|
||||
E[Global Events] --> F[Macro Scorer]
|
||||
F --> |"probabilistic=false"| G1[Linear Weighted Sum]
|
||||
F --> |"probabilistic=true"| G2[Multiplicative Exposure]
|
||||
G1 --> H[Macro WeightedSignals]
|
||||
G2 --> H
|
||||
end
|
||||
|
||||
subgraph "Layer 3: Competitive Signals"
|
||||
I[Pattern Matcher] --> J[Signal Propagation]
|
||||
J --> |"probabilistic=false"| K1[Flat Transfer Strength]
|
||||
J --> |"probabilistic=true"| K2[Graph-Distance Attenuation]
|
||||
K1 --> L[Competitive WeightedSignals]
|
||||
K2 --> L
|
||||
end
|
||||
|
||||
subgraph "Regime Detection (new)"
|
||||
M[Market Data] --> N[Regime Detector]
|
||||
N --> O{Regime Classification}
|
||||
O --> P[trend-following / panic / mean-reversion / uncertainty]
|
||||
end
|
||||
|
||||
subgraph "Trend Assembly"
|
||||
D --> Q[Merge Signals]
|
||||
H --> |"probabilistic=false"| Q
|
||||
H --> |"probabilistic=true"| R[Conditional Macro Modifier]
|
||||
R --> Q
|
||||
L --> Q
|
||||
Q --> S[Trend Assembler]
|
||||
S --> |"probabilistic=false"| T1[Heuristic Confidence + Threshold Direction]
|
||||
S --> |"probabilistic=true"| T2[Bayesian Posterior + Entropy Direction<br/>+ Regime-Adjusted Thresholds]
|
||||
P --> T2
|
||||
T1 --> U[TrendSummary]
|
||||
T2 --> U
|
||||
end
|
||||
|
||||
subgraph "Projection"
|
||||
U --> V[Projection Engine]
|
||||
V --> |"probabilistic=false"| W1[Simple Momentum]
|
||||
V --> |"probabilistic=true"| W2[EW Momentum + Vol Scaling]
|
||||
W1 --> X[TrendProjection]
|
||||
W2 --> X
|
||||
end
|
||||
|
||||
subgraph "Recommendation"
|
||||
U --> Y[Recommendation Engine]
|
||||
X --> Y
|
||||
Y --> |"probabilistic=false"| Z1[Confidence + Strength Gates]
|
||||
Y --> |"probabilistic=true"| Z2[EV Gate + Existing Gates]
|
||||
Z1 --> AA[Recommendation]
|
||||
Z2 --> AA
|
||||
end
|
||||
```
|
||||
|
||||
### Feature Flag Control Flow
|
||||
|
||||
The feature flag `probabilistic_scoring_enabled` is read from the `risk_configs` table's `config` JSONB column at the start of each aggregation cycle. It propagates through all pipeline stages via the existing `AggregationConfig` dataclass.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant W as Worker (aggregate_company)
|
||||
participant DB as PostgreSQL (risk_configs)
|
||||
participant S as Signal Scorer
|
||||
participant T as Trend Assembler
|
||||
participant R as Recommendation Engine
|
||||
|
||||
W->>DB: SELECT config FROM risk_configs WHERE active=TRUE
|
||||
DB-->>W: {"macro_enabled": true, "competitive_enabled": true, "probabilistic_scoring_enabled": false}
|
||||
W->>W: Log pipeline mode (heuristic or probabilistic)
|
||||
W->>S: compute_signal_weight(..., probabilistic=flag)
|
||||
S-->>W: WeightedSignal (with or without Bayesian fields)
|
||||
W->>T: assemble_trend_summary(..., probabilistic=flag)
|
||||
T-->>W: TrendSummary (with or without entropy/regime)
|
||||
W->>R: evaluate_eligibility(..., probabilistic=flag)
|
||||
R-->>W: Recommendation (with or without EV gate)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Components and Interfaces
|
||||
|
||||
### New Modules
|
||||
|
||||
| Module | File | Responsibility |
|
||||
|--------|------|----------------|
|
||||
| Bayesian Accumulator | `services/aggregation/bayesian.py` | Log-likelihood accumulation, Beta posterior, P_bull, Bayesian confidence |
|
||||
| Regime Detector | `services/aggregation/regime.py` | EMA computation, volatility ratio, regime classification, threshold adjustment |
|
||||
| Adaptive Decay | integrated into `scoring.py` | Event-specific half-life computation from impact, surprise, market reaction |
|
||||
| Information Gain | integrated into `scoring.py` | Surprise weighting from event type base rates |
|
||||
| Source Accuracy | `services/aggregation/source_accuracy.py` | Historical prediction accuracy tracking per source |
|
||||
| Entropy Detector | integrated into `bayesian.py` | Shannon entropy for mixed signal detection |
|
||||
| EV Gate | integrated into `eligibility.py` | Expected value computation for recommendation eligibility |
|
||||
|
||||
### Modified Modules
|
||||
|
||||
| Module | File | Changes |
|
||||
|--------|------|---------|
|
||||
| Signal Scorer | `services/aggregation/scoring.py` | Sigmoid gate, info gain factor, adaptive decay, regime multiplier, source accuracy factor |
|
||||
| Trend Assembler | `services/aggregation/worker.py` | Bayesian confidence, entropy-based direction, regime-adjusted thresholds, entropy-based contradiction |
|
||||
| Contradiction | `services/aggregation/contradiction.py` | Weighted disagreement entropy replacing minority/majority ratio |
|
||||
| Macro Scorer | `services/aggregation/interpolation.py` | Multiplicative exposure formula, conditional integration mode |
|
||||
| Competitive Scorer | `services/aggregation/signal_propagation.py` | Graph-distance attenuation with historical correlation |
|
||||
| Projection Engine | `services/aggregation/projection.py` | Exponentially weighted momentum, volatility scaling |
|
||||
| Recommendation | `services/recommendation/eligibility.py` | EV gate, P_bull-based position sizing adjustments |
|
||||
| Config | `services/shared/config.py` | New probabilistic config parameters |
|
||||
| Schemas | `services/shared/schemas.py` | Optional new fields on TrendSummary, Recommendation |
|
||||
|
||||
### Component Interface Details
|
||||
|
||||
#### 1. Bayesian Accumulator (`services/aggregation/bayesian.py`)
|
||||
|
||||
```python
|
||||
@dataclass(frozen=True)
|
||||
class BayesianPosterior:
|
||||
"""Bayesian posterior state from signal accumulation."""
|
||||
p_bull: float # σ(L_t), bullish probability [0, 1]
|
||||
alpha: float # Beta distribution α parameter (≥ 1.0)
|
||||
beta: float # Beta distribution β parameter (≥ 1.0)
|
||||
log_likelihood: float # Raw log-likelihood accumulation L_t
|
||||
bayesian_confidence: float # 1 - 4αβ/(α+β)², [0, 1]
|
||||
entropy: float # Shannon entropy H, [0, 1]
|
||||
signal_count: int # Number of signals processed
|
||||
|
||||
# Uninformative prior (no evidence)
|
||||
PRIOR = BayesianPosterior(
|
||||
p_bull=0.5, alpha=1.0, beta=1.0,
|
||||
log_likelihood=0.0, bayesian_confidence=0.0,
|
||||
entropy=1.0, signal_count=0,
|
||||
)
|
||||
|
||||
|
||||
def compute_bayesian_posterior(
|
||||
signals: list[WeightedSignal],
|
||||
) -> BayesianPosterior:
|
||||
"""Accumulate weighted signals into a Bayesian posterior.
|
||||
|
||||
Computes:
|
||||
- Log-likelihood: L_t = Σ(w_i · s_i)
|
||||
- Bullish probability: P_bull = σ(L_t)
|
||||
- Beta posterior: α = 1 + W_bull, β = 1 + W_bear
|
||||
- Bayesian confidence: C = 1 - 4αβ/(α+β)²
|
||||
- Shannon entropy: H = -p·log₂(p) - (1-p)·log₂(1-p)
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
def compute_entropy(p_bull: float) -> float:
|
||||
"""Shannon entropy H = -p·log₂(p) - (1-p)·log₂(1-p).
|
||||
|
||||
Returns value in [0, 1]. Maximum at p=0.5, zero at p=0 or p=1.
|
||||
Handles edge cases p=0 and p=1 by returning 0.0.
|
||||
"""
|
||||
...
|
||||
```
|
||||
|
||||
#### 2. Regime Detector (`services/aggregation/regime.py`)
|
||||
|
||||
```python
|
||||
class MarketRegime(str, Enum):
|
||||
TREND_FOLLOWING = "trend_following"
|
||||
PANIC = "panic"
|
||||
MEAN_REVERSION = "mean_reversion"
|
||||
UNCERTAINTY = "uncertainty"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RegimeClassification:
|
||||
"""Result of regime detection for a ticker."""
|
||||
regime: MarketRegime
|
||||
trend_indicator: float # R = sign(EMA_20 - EMA_100)
|
||||
volatility_ratio: float # V_r = σ_20 / σ_100
|
||||
bullish_threshold: float # Adjusted ±threshold for direction
|
||||
bearish_threshold: float
|
||||
contradiction_penalty_multiplier: float # 0.4 default, 0.6 for uncertainty
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RegimeConfig:
|
||||
ema_short_period: int = 20
|
||||
ema_long_period: int = 100
|
||||
vol_short_period: int = 20
|
||||
vol_long_period: int = 100
|
||||
panic_vol_ratio: float = 1.5
|
||||
trend_vol_ratio: float = 1.2
|
||||
mean_reversion_vol_ratio: float = 1.0
|
||||
default_threshold: float = 0.15
|
||||
panic_threshold: float = 0.10
|
||||
mean_reversion_threshold: float = 0.20
|
||||
uncertainty_contradiction_multiplier: float = 0.6
|
||||
|
||||
|
||||
def classify_regime(
|
||||
closing_prices: list[float],
|
||||
returns: list[float],
|
||||
config: RegimeConfig = RegimeConfig(),
|
||||
) -> RegimeClassification:
|
||||
"""Classify market regime from price and return history.
|
||||
|
||||
Requires at least 100 days of price history for EMA_100.
|
||||
Falls back to UNCERTAINTY when data is insufficient.
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
def compute_ema(values: list[float], period: int) -> float:
|
||||
"""Compute exponential moving average over the last `period` values."""
|
||||
...
|
||||
```
|
||||
|
||||
#### 3. Source Accuracy Tracker (`services/aggregation/source_accuracy.py`)
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class SourceAccuracy:
|
||||
"""Per-source historical prediction accuracy."""
|
||||
source_id: str
|
||||
accuracy_ratio: float # [0, 1] fraction of correct directional calls
|
||||
sample_count: int # Number of signals with known outcomes
|
||||
last_updated: datetime
|
||||
|
||||
@property
|
||||
def accuracy_factor(self) -> float:
|
||||
"""Multiplicative factor for credibility weight.
|
||||
|
||||
Returns 1.0 (neutral) when sample_count < 10.
|
||||
Otherwise scales linearly from 0.5 (0% accuracy) to 1.5 (100% accuracy).
|
||||
"""
|
||||
if self.sample_count < 10:
|
||||
return 1.0
|
||||
return 0.5 + self.accuracy_ratio
|
||||
|
||||
|
||||
async def fetch_source_accuracy(
|
||||
pool: asyncpg.Pool,
|
||||
source_ids: list[str],
|
||||
) -> dict[str, SourceAccuracy]:
|
||||
"""Fetch accuracy metrics for a batch of sources."""
|
||||
...
|
||||
|
||||
|
||||
async def update_source_accuracy(
|
||||
pool: asyncpg.Pool,
|
||||
source_id: str,
|
||||
realized_outcomes: list[tuple[str, float]], # (predicted_direction, actual_7d_return)
|
||||
) -> None:
|
||||
"""Update accuracy metrics for a source based on realized price data."""
|
||||
...
|
||||
```
|
||||
|
||||
#### 4. Extended ScoringConfig
|
||||
|
||||
New fields added to the existing `ScoringConfig` dataclass in `scoring.py`:
|
||||
|
||||
```python
|
||||
@dataclass(frozen=True)
|
||||
class ScoringConfig:
|
||||
# ... existing fields preserved ...
|
||||
|
||||
# Probabilistic scoring toggle (mirrors feature flag for local use)
|
||||
probabilistic: bool = False
|
||||
|
||||
# Sigmoid gate parameters
|
||||
sigmoid_steepness: float = 5.0 # k in σ(k·(x - midpoint))
|
||||
sigmoid_midpoint: float = 0.5 # midpoint of sigmoid transition
|
||||
|
||||
# Information gain parameters
|
||||
info_gain_lambda: float = 0.3 # scaling parameter λ
|
||||
info_gain_max: float = 3.0 # maximum clamp for info gain factor
|
||||
default_base_rate: float = 0.1 # fallback when event type rate unknown
|
||||
|
||||
# Adaptive decay parameters (β scaling factors)
|
||||
adaptive_decay_impact_scale: float = 1.0 # max β_impact
|
||||
adaptive_decay_surprise_scale: float = 1.0 # max β_surprise at r=3.0
|
||||
adaptive_decay_market_scale: float = 0.5 # max β_market_reaction
|
||||
|
||||
# Regime multiplier parameters
|
||||
regime_return_weight: float = 0.15 # coefficient for |z_r|
|
||||
regime_volume_weight: float = 0.10 # coefficient for |z_v|
|
||||
regime_multiplier_max: float = 2.5 # M_regime ceiling
|
||||
```
|
||||
|
||||
#### 5. Extended WeightedSignal
|
||||
|
||||
The existing `WeightedSignal` dataclass gains optional fields:
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class WeightedSignal:
|
||||
"""A document intelligence reference paired with its computed weight."""
|
||||
document_id: str
|
||||
weight: SignalWeight
|
||||
sentiment_value: float
|
||||
impact_score: float
|
||||
|
||||
# New optional fields for probabilistic mode
|
||||
info_gain_factor: float = 1.0 # r = 1 + λ·(-log₂ P(event_type))
|
||||
source_accuracy_factor: float = 1.0 # [0.5, 1.5] from historical accuracy
|
||||
adaptive_half_life: float | None = None # τ_i when adaptive decay is active
|
||||
```
|
||||
|
||||
#### 6. Extended SignalWeight
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class SignalWeight:
|
||||
"""Breakdown of a document's aggregation weight."""
|
||||
recency: float
|
||||
credibility: float
|
||||
novelty_bonus: float
|
||||
confidence_gate: float
|
||||
market_ctx_multiplier: float
|
||||
combined: float
|
||||
|
||||
# New optional fields for probabilistic mode
|
||||
sigmoid_gate: float | None = None # Smooth gate value [0, 1]
|
||||
info_gain_factor: float = 1.0 # Surprise multiplier
|
||||
source_accuracy_factor: float = 1.0 # Historical accuracy multiplier
|
||||
regime_multiplier: float | None = None # M_regime replacing M_context
|
||||
```
|
||||
|
||||
#### 7. Extended TrendSummary
|
||||
|
||||
New optional fields on the existing Pydantic model:
|
||||
|
||||
```python
|
||||
class TrendSummary(BaseModel):
|
||||
# ... all existing fields preserved ...
|
||||
|
||||
# New optional fields for probabilistic mode
|
||||
p_bull: float | None = None # Bayesian bullish probability
|
||||
alpha: float | None = None # Beta posterior α
|
||||
beta_param: float | None = None # Beta posterior β (named to avoid shadowing)
|
||||
bayesian_confidence: float | None = None # 1 - 4αβ/(α+β)²
|
||||
entropy: float | None = None # Shannon entropy H
|
||||
regime: str | None = None # Market regime classification
|
||||
pipeline_mode: str = "heuristic" # "heuristic" or "probabilistic"
|
||||
```
|
||||
|
||||
#### 8. Extended Recommendation
|
||||
|
||||
```python
|
||||
class Recommendation(BaseModel):
|
||||
# ... all existing fields preserved ...
|
||||
|
||||
# New optional fields for probabilistic mode
|
||||
expected_value: float | None = None # EV = P_bull·R_up - P_bear·R_down
|
||||
p_bull: float | None = None # Bayesian bullish probability used
|
||||
pipeline_mode: str = "heuristic" # "heuristic" or "probabilistic"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Data Models
|
||||
|
||||
### Database Storage Strategy
|
||||
|
||||
All new mathematical outputs are stored in existing JSONB columns. No new database migrations are required.
|
||||
|
||||
#### trend_windows table
|
||||
|
||||
The `market_context` JSONB column (currently stores volatility/volume data) is extended to include probabilistic outputs:
|
||||
|
||||
```json
|
||||
{
|
||||
"volatility": 1.23,
|
||||
"volume_change_pct": 45.2,
|
||||
"price_change_pct": -2.1,
|
||||
"probabilistic": {
|
||||
"p_bull": 0.72,
|
||||
"alpha": 8.3,
|
||||
"beta": 3.1,
|
||||
"log_likelihood": 0.94,
|
||||
"bayesian_confidence": 0.61,
|
||||
"entropy": 0.42,
|
||||
"regime": "trend_following",
|
||||
"regime_volatility_ratio": 0.85,
|
||||
"pipeline_mode": "probabilistic",
|
||||
"contradiction_entropy": 0.31,
|
||||
"macro_modifier": 1.15
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
#### recommendations table
|
||||
|
||||
The existing `invalidation_conditions` JSONB column stores recommendation-level data. The new EV and probabilistic fields are stored in a new key within the existing decision trace flow. Since recommendations don't have a dedicated metadata JSONB column, we add the probabilistic fields to the thesis text and store structured data in the `risk_checks` JSONB column of the `recommendation_evaluations` table:
|
||||
|
||||
```json
|
||||
{
|
||||
"ev": 0.0082,
|
||||
"p_bull": 0.72,
|
||||
"r_up": 0.034,
|
||||
"r_down": 0.012,
|
||||
"pipeline_mode": "probabilistic",
|
||||
"ev_threshold": 0.005
|
||||
}
|
||||
```
|
||||
|
||||
#### risk_configs table
|
||||
|
||||
The `config` JSONB column gains the new feature flag:
|
||||
|
||||
```json
|
||||
{
|
||||
"macro_enabled": true,
|
||||
"competitive_enabled": true,
|
||||
"probabilistic_scoring_enabled": false
|
||||
}
|
||||
```
|
||||
|
||||
#### source_accuracy table (new — Requirement 4)
|
||||
|
||||
This is the one new database table required, stored via a migration:
|
||||
|
||||
```sql
|
||||
CREATE TABLE IF NOT EXISTS source_accuracy (
|
||||
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||
source_id VARCHAR(200) NOT NULL,
|
||||
accuracy_ratio FLOAT NOT NULL DEFAULT 0.5,
|
||||
sample_count INTEGER NOT NULL DEFAULT 0,
|
||||
last_updated TIMESTAMPTZ NOT NULL DEFAULT NOW(),
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
|
||||
UNIQUE(source_id)
|
||||
);
|
||||
CREATE INDEX idx_source_accuracy_source ON source_accuracy(source_id);
|
||||
```
|
||||
|
||||
Note: This is the only schema addition. All other new outputs use existing JSONB columns.
|
||||
|
||||
### Event Type Base Rates
|
||||
|
||||
Information gain computation requires empirical base rates for event types. These are stored as a configuration constant (not in the database) and can be tuned over time:
|
||||
|
||||
```python
|
||||
EVENT_TYPE_BASE_RATES: dict[str, float] = {
|
||||
"earnings": 0.25, # Quarterly, common
|
||||
"product_launch": 0.10, # Moderately rare
|
||||
"regulatory": 0.08, # Somewhat rare
|
||||
"legal": 0.05, # Rare
|
||||
"m_and_a": 0.03, # Very rare
|
||||
"management_change": 0.06,
|
||||
"partnership": 0.12,
|
||||
"market_expansion": 0.09,
|
||||
"restructuring": 0.04,
|
||||
"dividend": 0.15,
|
||||
}
|
||||
DEFAULT_BASE_RATE = 0.1 # For unknown event types
|
||||
```
|
||||
|
||||
### Configuration Hierarchy
|
||||
|
||||
```
|
||||
risk_configs.config (DB, runtime)
|
||||
└── probabilistic_scoring_enabled: bool
|
||||
└── AggregationConfig.probabilistic: bool (in-memory)
|
||||
└── ScoringConfig.probabilistic: bool (per-cycle)
|
||||
├── scoring.py: sigmoid vs binary gate
|
||||
├── scoring.py: adaptive vs fixed decay
|
||||
├── scoring.py: info gain factor
|
||||
├── scoring.py: regime multiplier vs market context
|
||||
├── worker.py: Bayesian vs heuristic confidence
|
||||
├── worker.py: entropy vs threshold direction
|
||||
├── contradiction.py: entropy vs ratio
|
||||
├── interpolation.py: multiplicative vs linear
|
||||
├── signal_propagation.py: graph-distance vs flat
|
||||
├── projection.py: EW momentum vs simple
|
||||
└── eligibility.py: EV gate vs threshold-only
|
||||
```
|
||||
|
||||
|
||||
---
|
||||
|
||||
## Correctness Properties
|
||||
|
||||
*A property is a characteristic or behavior that should hold true across all valid executions of a system — essentially, a formal statement about what the system should do. Properties serve as the bridge between human-readable specifications and machine-verifiable correctness guarantees.*
|
||||
|
||||
The following properties were derived from the acceptance criteria through systematic prework analysis. Each property is universally quantified and maps to specific requirements. Redundant properties were consolidated during reflection (e.g., requirements 17.1–17.7 duplicate properties already stated in requirements 1–15).
|
||||
|
||||
### Property 1: Sigmoid Gate Monotonicity
|
||||
|
||||
*For any* two extraction confidence values x₁, x₂ ∈ [0.0, 1.0] where x₁ ≤ x₂, the sigmoid gate σ(5·(x₁ - 0.5)) SHALL be less than or equal to σ(5·(x₂ - 0.5)). Higher confidence always produces equal or higher gate values.
|
||||
|
||||
**Validates: Requirements 2.6, 17.1**
|
||||
|
||||
### Property 2: Beta Posterior Evidence Accumulation
|
||||
|
||||
*For any* sequence of weighted signal sets where each successive set contains one additional signal, the sum α + β of the Beta posterior parameters SHALL increase monotonically. Evidence always accumulates — adding a signal never reduces the total evidence mass.
|
||||
|
||||
**Validates: Requirements 1.3, 17.2**
|
||||
|
||||
### Property 3: Bayesian Confidence Symmetry and Divergence
|
||||
|
||||
*For any* Beta posterior with parameters α, β ≥ 1.0, the Bayesian confidence C = 1 - 4αβ/(α+β)² SHALL equal 0.0 when α = β (maximum uncertainty) and SHALL increase monotonically as the ratio max(α/β, β/α) increases. Confidence reflects evidence concentration, not evidence volume.
|
||||
|
||||
**Validates: Requirements 1.4, 17.3**
|
||||
|
||||
### Property 4: Bayesian Posterior Round-Trip Consistency
|
||||
|
||||
*For any* set of weighted signals with uniform weights, computing the Beta posterior and extracting the mean P_bull = α/(α+β) SHALL produce a value within 0.05 of σ(L_t) where L_t is the log-likelihood accumulation. The two probabilistic representations are consistent.
|
||||
|
||||
**Validates: Requirements 1.7, 17.7**
|
||||
|
||||
### Property 5: Adaptive Decay Lower Bound
|
||||
|
||||
*For any* valid combination of impact_score ∈ [0, 1], information gain factor r ∈ [1.0, 3.0], and market context multiplier ∈ [1.0, 1.45], the adaptive half-life τ_i SHALL be greater than or equal to the base half-life τ_base. Adaptive decay is always slower or equal to fixed decay, never faster.
|
||||
|
||||
**Validates: Requirements 5.7, 17.4**
|
||||
|
||||
### Property 6: Information Gain Monotonicity
|
||||
|
||||
*For any* two event type base rates p₁, p₂ ∈ (0, 1] where p₁ < p₂, the information gain factor r(p₁) SHALL be greater than or equal to r(p₂). Rarer events always receive higher surprise weight.
|
||||
|
||||
**Validates: Requirements 3.5**
|
||||
|
||||
### Property 7: Multiplicative Macro Exposure Monotonicity
|
||||
|
||||
*For any* overlap configuration (O_geo, O_supply, O_commodity, O_sector) and any dimension k where O_k = 0, setting O_k to any positive value SHALL increase the total macro impact score. Multi-dimensional exposure always compounds — it never reduces impact.
|
||||
|
||||
**Validates: Requirements 10.7, 17.5**
|
||||
|
||||
### Property 8: Shannon Entropy Range and Maximum
|
||||
|
||||
*For any* bullish probability P_bull ∈ (0, 1), the Shannon entropy H = -P_bull·log₂(P_bull) - (1-P_bull)·log₂(1-P_bull) SHALL be in the range (0, 1], with the maximum value of 1.0 occurring at P_bull = 0.5.
|
||||
|
||||
**Validates: Requirements 9.7**
|
||||
|
||||
### Property 9: Contradiction Entropy Monotonicity
|
||||
|
||||
*For any* set of weighted signals containing both positive and negative sentiment signals, the contradiction entropy score SHALL increase monotonically as the weight distribution f_pos approaches 0.5 (equal split). More balanced disagreement always produces higher contradiction.
|
||||
|
||||
**Validates: Requirements 15.7**
|
||||
|
||||
### Property 10: Exponentially Weighted Momentum Direction
|
||||
|
||||
*For any* sequence of monotonically increasing signed trend strengths (each ΔS_{t-k} > 0), the exponentially weighted momentum M_t SHALL be positive. Consistently strengthening bullish trends always produce positive momentum.
|
||||
|
||||
**Validates: Requirements 13.6, 17.6**
|
||||
|
||||
### Property 11: Competitive Signal Distance Attenuation
|
||||
|
||||
*For any* source-target company pair with fixed source signal strength S_source and historical correlation ρ_historical, the transfer strength S_transfer SHALL decrease monotonically with increasing graph distance d_network. Closer competitors always receive stronger signal transfer.
|
||||
|
||||
**Validates: Requirements 12.7**
|
||||
|
||||
### Property 12: Expected Value Directional Consistency
|
||||
|
||||
*For any* Bayesian bullish probability P_bull > 0.5 and estimated returns where R_up > R_down, the expected value EV = P_bull · R_up - (1 - P_bull) · R_down SHALL be positive. When the model is bullish and upside exceeds downside, EV is always positive.
|
||||
|
||||
**Validates: Requirements 17.8**
|
||||
|
||||
### Property 13: Bayesian Confidence Monotonic with Agreeing Signals
|
||||
|
||||
*For any* set of weighted signals where all signals agree on direction (all positive or all negative), adding one more agreeing signal SHALL increase the Bayesian confidence C. More agreeing evidence always increases confidence.
|
||||
|
||||
**Validates: Requirements 8.6**
|
||||
|
||||
### Property 14: Numerical Stability Across All Formulas
|
||||
|
||||
*For any* valid input combination to any formula in the probabilistic pipeline (sigmoid gate, Beta posterior, Bayesian confidence, adaptive decay, regime multiplier, Shannon entropy, multiplicative exposure, EW momentum, expected value), the output SHALL be a finite float (not NaN, not infinity) within the documented range for that formula. This includes regime multiplier M_regime ∈ [1.0, 2.5], entropy H ∈ [0, 1], P_bull ∈ [0, 1], confidence ∈ [0, 1], and M_adj ∈ [-2.0, 2.0].
|
||||
|
||||
**Validates: Requirements 17.9, 6.4**
|
||||
|
||||
---
|
||||
|
||||
## Error Handling
|
||||
|
||||
### Numerical Edge Cases
|
||||
|
||||
| Scenario | Handling |
|
||||
|----------|----------|
|
||||
| P_bull = 0.0 or 1.0 (entropy undefined) | Return H = 0.0 (no uncertainty at extremes) |
|
||||
| σ_20 = 0.0 (zero volatility for momentum scaling) | Use floor max(σ_20, 0.01) per Req 13.4 |
|
||||
| σ_20 = 0.0 or σ_100 = 0.0 (volatility ratio) | Default to uncertainty regime |
|
||||
| log₂(0) in entropy computation | Guard with `if p <= 0 or p >= 1: return 0.0` |
|
||||
| log₂(0) in information gain (base_rate = 0) | Base rates must be > 0; use default 0.1 for unknown |
|
||||
| Division by zero in z-score (σ = 0) | Use M_regime = 1.0 when σ = 0 |
|
||||
| Empty signal list | Return uninformative prior (P_bull=0.5, α=1, β=1, C=0) |
|
||||
| All neutral signals (no positive or negative) | Contradiction = 0.0, direction = neutral |
|
||||
| Extremely large weights (overflow risk) | Python floats handle up to ~1.8e308; clamp combined weight if needed |
|
||||
| NaN from upstream data | Validate inputs; skip signals with NaN weight or sentiment |
|
||||
|
||||
### Feature Flag Failure Modes
|
||||
|
||||
| Failure | Behavior |
|
||||
|---------|----------|
|
||||
| `risk_configs` table unreachable | Default to `probabilistic_scoring_enabled = false` (heuristic mode) |
|
||||
| `config` JSONB missing the key | Default to `false` |
|
||||
| Invalid value type for flag | Default to `false`, log warning |
|
||||
| Flag changes mid-cycle | Flag is read once at cycle start; change takes effect next cycle |
|
||||
|
||||
### Source Accuracy Failures
|
||||
|
||||
| Failure | Behavior |
|
||||
|---------|----------|
|
||||
| `source_accuracy` table unreachable | Use neutral factor 1.0 for all sources |
|
||||
| Accuracy update fails | Log error, continue with stale accuracy data |
|
||||
| Corrupted accuracy data (ratio > 1.0 or < 0.0) | Clamp to [0.0, 1.0] |
|
||||
|
||||
### Regime Detection Failures
|
||||
|
||||
| Failure | Behavior |
|
||||
|---------|----------|
|
||||
| Market data unavailable | Default to uncertainty regime with default thresholds |
|
||||
| Insufficient price history (< 100 days) | Default to uncertainty regime |
|
||||
| Price data contains gaps | Use available data; EMA computation handles gaps gracefully |
|
||||
|
||||
---
|
||||
|
||||
## Testing Strategy
|
||||
|
||||
### Dual Testing Approach
|
||||
|
||||
The signal math upgrade requires both property-based tests (for mathematical correctness) and example-based unit tests (for specific behaviors and integration points). Property-based testing is highly appropriate here because the feature consists primarily of pure mathematical functions with clear input/output behavior, universal properties that hold across wide input spaces, and well-defined range invariants.
|
||||
|
||||
### Property-Based Testing
|
||||
|
||||
**Library:** Hypothesis (already in use per `.hypothesis/` directory and project conventions)
|
||||
|
||||
**Configuration:**
|
||||
- Minimum 100 iterations per property: `@settings(max_examples=100)`
|
||||
- File naming: `test_pbt_signal_math.py` (or split by module)
|
||||
- Tag format: `# Feature: signal-math-upgrade, Property N: <title>`
|
||||
|
||||
**Property tests to implement (one test per correctness property):**
|
||||
|
||||
| Property | Test File | Key Generators |
|
||||
|----------|-----------|----------------|
|
||||
| 1: Sigmoid monotonicity | `test_pbt_signal_math.py` | `st.floats(0.0, 1.0)` pairs |
|
||||
| 2: Evidence accumulation | `test_pbt_signal_math.py` | `st.lists(weighted_signal_strategy)` |
|
||||
| 3: Confidence symmetry/divergence | `test_pbt_signal_math.py` | `st.floats(1.0, 100.0)` for α, β |
|
||||
| 4: Posterior round-trip | `test_pbt_signal_math.py` | `st.lists(uniform_weight_signal_strategy)` |
|
||||
| 5: Adaptive decay lower bound | `test_pbt_signal_math.py` | `st.floats` for impact, surprise, market |
|
||||
| 6: Info gain monotonicity | `test_pbt_signal_math.py` | `st.floats(0.001, 1.0)` pairs |
|
||||
| 7: Macro exposure monotonicity | `test_pbt_signal_math.py` | `st.floats(0.0, 1.0)` for overlaps |
|
||||
| 8: Entropy range/maximum | `test_pbt_signal_math.py` | `st.floats(0.001, 0.999)` for P_bull |
|
||||
| 9: Contradiction monotonicity | `test_pbt_signal_math.py` | Signal sets with varying weight splits |
|
||||
| 10: EW momentum direction | `test_pbt_signal_math.py` | `st.lists(st.floats)` monotonic sequences |
|
||||
| 11: Distance attenuation | `test_pbt_signal_math.py` | `st.integers(1, 3)` for distance |
|
||||
| 12: EV directional consistency | `test_pbt_signal_math.py` | `st.floats(0.5, 1.0)` for P_bull |
|
||||
| 13: Confidence with agreeing signals | `test_pbt_signal_math.py` | Growing lists of same-direction signals |
|
||||
| 14: Numerical stability | `test_pbt_signal_math.py` | Broad `st.floats` for all formula inputs |
|
||||
|
||||
### Example-Based Unit Tests
|
||||
|
||||
**File:** `test_signal_math_unit.py`
|
||||
|
||||
| Test Area | Examples |
|
||||
|-----------|----------|
|
||||
| Sigmoid gate specific values | x=0.5→0.5, x=0.2→<0.05, x=0.8→>0.95 |
|
||||
| Uninformative prior | Empty signals → P_bull=0.5, α=1, β=1, C=0 |
|
||||
| Default base rate | Unknown event type → base_rate=0.1 |
|
||||
| Info gain clamp | Very rare event → factor ≤ 3.0 |
|
||||
| Source accuracy threshold | sample_count < 10 → factor=1.0 |
|
||||
| Adaptive decay edge cases | All zeros → τ_base, all max → 6×τ_base |
|
||||
| Regime classification | Specific (R, V_r) → expected regime |
|
||||
| Regime thresholds | panic→0.10, mean_reversion→0.20, etc. |
|
||||
| Entropy direction mapping | H>0.9→mixed, P_bull>0.65→bullish, etc. |
|
||||
| Zero overlap → zero impact | All overlaps zero → S_macro=0 |
|
||||
| Max overlap value | All overlaps 1.0 → ≈severity×0.724 |
|
||||
| Macro fallback behaviors | Only macro → additive, only company → no modifier |
|
||||
| Graph distance cutoff | d>3 → no propagation |
|
||||
| Momentum fallback | <2 cycles → heuristic fallback |
|
||||
| EV threshold behavior | EV>0.005→proceed, EV≤0.005→informational |
|
||||
| Feature flag behaviors | flag=false→heuristic, flag=true→probabilistic |
|
||||
| Heuristic equivalence | flag=false produces identical outputs to current system |
|
||||
|
||||
### Integration Tests
|
||||
|
||||
| Test Area | Scope |
|
||||
|-----------|-------|
|
||||
| Source accuracy persistence | Write/read from source_accuracy table |
|
||||
| Regime persistence | Store/retrieve regime in JSONB |
|
||||
| EV persistence | Store/retrieve EV in recommendation_evaluations |
|
||||
| Feature flag reading | Read probabilistic_scoring_enabled from risk_configs |
|
||||
| End-to-end pipeline | Full aggregation cycle with probabilistic=true |
|
||||
|
||||
### Test Organization
|
||||
|
||||
```
|
||||
tests/
|
||||
├── test_pbt_signal_math.py # All 14 property-based tests
|
||||
├── test_signal_math_unit.py # Example-based unit tests
|
||||
├── test_bayesian.py # Bayesian accumulator unit tests
|
||||
├── test_regime.py # Regime detector unit tests
|
||||
├── test_source_accuracy.py # Source accuracy tracker tests
|
||||
└── test_signal_math_integration.py # Integration tests (DB required)
|
||||
```
|
||||
@@ -0,0 +1,293 @@
|
||||
# Requirements Document — Signal Math Upgrade
|
||||
|
||||
## Introduction
|
||||
|
||||
The Stonks Oracle platform uses a three-layer signal aggregation engine (company-specific, macro, competitive) to produce market intelligence and drive paper-trading decisions. The current mathematical models are structurally too deterministic and too linear for a market system that is fundamentally probabilistic, regime-dependent, and nonlinear. The pipeline behaves as weighted sentiment aggregation with heuristics rather than a probabilistic forecasting engine.
|
||||
|
||||
This feature upgrades the signal processing mathematics across all pipeline stages — from signal scoring through trend assembly, macro impact, competitive signals, trend projection, and recommendation generation — to replace heuristic formulas with probabilistic, regime-aware, and adaptive alternatives. The goal is to transform prediction quality while preserving the existing `WeightedSignal` abstraction, three-layer architecture, and database schema compatibility.
|
||||
|
||||
## Glossary
|
||||
|
||||
- **Aggregation_Engine**: The core pipeline in `services/aggregation/worker.py` that merges signals from all three layers and computes `TrendSummary` objects across five time windows.
|
||||
- **Signal_Scorer**: The scoring module in `services/aggregation/scoring.py` that transforms raw intelligence records into `WeightedSignal` objects with composite aggregation weights.
|
||||
- **Trend_Assembler**: The component in `services/aggregation/worker.py` that derives trend direction, strength, confidence, and contradiction from merged weighted signals.
|
||||
- **Macro_Scorer**: The macro impact scoring module in `services/aggregation/interpolation.py` that computes per-company impact from global events using overlap-based exposure profiles.
|
||||
- **Competitive_Scorer**: The competitive signal modules in `services/aggregation/pattern_matcher.py` and `services/aggregation/signal_propagation.py` that mine historical patterns and propagate cross-company signals.
|
||||
- **Projection_Engine**: The trend projection module in `services/aggregation/projection.py` that computes forward-looking trend estimates from momentum and macro decay.
|
||||
- **Recommendation_Engine**: The recommendation pipeline in `services/recommendation/` that translates trend assessments into actionable buy/sell/hold/watch decisions with position sizing.
|
||||
- **WeightedSignal**: The core data abstraction pairing a document reference with a composite aggregation weight, sentiment value, and impact score.
|
||||
- **Beta_Distribution**: A probability distribution on [0, 1] parameterized by α and β, used to model the posterior probability of bullish vs bearish sentiment.
|
||||
- **Regime_Detector**: A new component that classifies the current market regime (trend-following, panic, mean-reversion, uncertainty) from price and volume statistics.
|
||||
- **Sigmoid_Function**: The logistic function σ(x) = 1/(1+e^(-x)) used to convert log-likelihood accumulations into probabilities.
|
||||
- **Adaptive_Decay**: A recency decay mechanism where the half-life varies per signal based on event impact, surprise, and market reaction rather than using a fixed constant per window.
|
||||
- **Information_Gain**: A measure of how surprising an event is relative to its base rate, computed as -log P(event_type), used to weight novel signals more heavily.
|
||||
- **Entropy**: Shannon entropy H = -p·log(p) - (1-p)·log(1-p), used to detect mixed sentiment states where the probability distribution is spread rather than concentrated.
|
||||
- **EMA**: Exponential Moving Average, a weighted moving average giving more weight to recent observations, used for trend and volatility regime detection.
|
||||
|
||||
---
|
||||
|
||||
## Requirements
|
||||
|
||||
### Requirement 1: Probabilistic Sentiment Accumulation via Bayesian Evidence
|
||||
|
||||
**User Story:** As a quantitative analyst, I want the signal scoring layer to accumulate sentiment evidence probabilistically using Bayesian methods, so that the system captures uncertainty structure instead of collapsing sentiment into binary ±1 labels.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN a set of weighted signals is provided for a ticker and window, THE Signal_Scorer SHALL compute a log-likelihood accumulation L_t = Σ(w_i · s_i) where w_i is the combined signal weight and s_i is the sentiment value.
|
||||
2. WHEN the log-likelihood L_t has been computed, THE Signal_Scorer SHALL convert the accumulation to a bullish probability using the Sigmoid_Function: P_bull = σ(L_t) = 1/(1+e^(-L_t)).
|
||||
3. WHEN weighted signals are provided, THE Signal_Scorer SHALL maintain a Beta_Distribution posterior with parameters α_t = α_0 + W_bull and β_t = β_0 + W_bear, where W_bull is the sum of combined weights for positive signals and W_bear is the sum for negative signals, and α_0 = β_0 = 1.0 as uninformative priors.
|
||||
4. THE Signal_Scorer SHALL compute Bayesian confidence from the Beta_Distribution posterior variance as C = 1 - 4αβ/(α+β)², where C ranges from 0.0 (maximum uncertainty at α=β) to approaching 1.0 (strong evidence concentration).
|
||||
5. WHEN no signals exist for a ticker and window, THE Signal_Scorer SHALL return P_bull = 0.5, α = 1.0, β = 1.0, and C = 0.0, representing the uninformative prior state.
|
||||
6. THE Signal_Scorer SHALL preserve the existing `WeightedSignal` dataclass interface, adding the Bayesian posterior fields (P_bull, α, β, Bayesian confidence) as additional output alongside the existing weighted sentiment average.
|
||||
7. FOR ALL valid sets of weighted signals, computing the Beta posterior then extracting P_bull SHALL produce a value within 0.05 of σ(L_t) when signal weights are uniform (round-trip consistency between the two probabilistic representations).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 2: Sigmoid Confidence Gate Replacing Binary Gate
|
||||
|
||||
**User Story:** As a quantitative analyst, I want the binary confidence gate replaced with a smooth sigmoid transition, so that marginally confident signals contribute proportionally rather than being completely discarded or fully included.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN a document signal has extraction confidence x, THE Signal_Scorer SHALL compute a soft gate value p = σ(5·(x - 0.5)) = 1/(1+e^(-5·(x-0.5))) instead of the current binary 0/1 gate.
|
||||
2. WHEN extraction confidence is 0.5, THE Signal_Scorer SHALL produce a gate value of 0.5 (the sigmoid midpoint).
|
||||
3. WHEN extraction confidence is below 0.2, THE Signal_Scorer SHALL produce a gate value below 0.05, preserving near-zero weight for very low confidence signals.
|
||||
4. WHEN extraction confidence is above 0.8, THE Signal_Scorer SHALL produce a gate value above 0.95, preserving near-full weight for high confidence signals.
|
||||
5. THE Signal_Scorer SHALL use the sigmoid gate value as a multiplicative factor in the combined weight formula in place of the current binary G_conf.
|
||||
6. FOR ALL extraction confidence values in [0.0, 1.0], THE Signal_Scorer SHALL produce gate values that are monotonically increasing (higher confidence always produces equal or higher gate values).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 3: Information Gain Surprise Weighting
|
||||
|
||||
**User Story:** As a quantitative analyst, I want signals weighted by their information gain (surprise factor), so that rare and unexpected events receive proportionally higher influence than routine signals.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN a signal has a known event type (e.g., earnings, product_launch, regulatory, legal, m_and_a), THE Signal_Scorer SHALL compute an information gain factor r = 1 + λ·(-log₂ P(event_type)), where P(event_type) is the empirical base rate of that event type and λ is a configurable scaling parameter with default 0.3.
|
||||
2. WHEN the event type base rate is not available, THE Signal_Scorer SHALL use a default base rate of 0.1 (treating the event as moderately rare).
|
||||
3. THE Signal_Scorer SHALL multiply the information gain factor r into the combined weight formula as an additional multiplicative component.
|
||||
4. THE Signal_Scorer SHALL clamp the information gain factor to a maximum of 3.0 to prevent extremely rare events from dominating the aggregation.
|
||||
5. FOR ALL event types with base rate in (0, 1], THE Signal_Scorer SHALL produce information gain factors that are monotonically decreasing with increasing base rate (rarer events always receive higher surprise weight).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 4: Historical Source Accuracy Tracking
|
||||
|
||||
**User Story:** As a quantitative analyst, I want source credibility to incorporate historical prediction accuracy, so that sources with a track record of correct directional calls receive higher weight.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Signal_Scorer SHALL maintain a per-source accuracy metric computed as the fraction of past signals from that source where the predicted direction matched the subsequent 7-day price movement direction.
|
||||
2. WHEN a source has at least 10 historical signals with known outcomes, THE Signal_Scorer SHALL incorporate the source accuracy as a multiplicative factor on the credibility weight, scaled linearly from 0.5 (0% accuracy) to 1.5 (100% accuracy).
|
||||
3. WHEN a source has fewer than 10 historical signals, THE Signal_Scorer SHALL use a neutral accuracy factor of 1.0 (no adjustment).
|
||||
4. THE Signal_Scorer SHALL update source accuracy metrics asynchronously after each aggregation cycle, using realized price data from the market data tables.
|
||||
5. THE Signal_Scorer SHALL store source accuracy metrics in a database table with columns for source identifier, accuracy ratio, sample count, and last updated timestamp.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 5: Adaptive Recency Decay with Event-Specific Half-Lives
|
||||
|
||||
**User Story:** As a quantitative analyst, I want recency decay half-lives to adapt based on event characteristics, so that high-impact events persist longer in the aggregation while routine signals decay faster.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN computing recency decay for a signal, THE Signal_Scorer SHALL use an adaptive half-life τ_i = τ_base · (1 + β_impact) · (1 + β_surprise) · (1 + β_market_reaction), where τ_base is the current fixed half-life for the window.
|
||||
2. THE Signal_Scorer SHALL compute β_impact from the signal's impact score, scaled linearly from 0.0 (impact_score = 0) to 1.0 (impact_score = 1.0).
|
||||
3. THE Signal_Scorer SHALL compute β_surprise from the information gain factor (Requirement 3), scaled linearly from 0.0 (r = 1.0, no surprise) to 1.0 (r = 3.0, maximum surprise).
|
||||
4. THE Signal_Scorer SHALL compute β_market_reaction from the market context multiplier, scaled linearly from 0.0 (multiplier = 1.0, no market reaction) to 0.5 (multiplier = 1.45, maximum market reaction).
|
||||
5. WHEN all three β factors are at their maximum, THE Signal_Scorer SHALL produce an adaptive half-life of at most 6× the base half-life (τ_base · 2.0 · 2.0 · 1.5 = 6.0 · τ_base).
|
||||
6. WHEN all three β factors are zero (routine, unsurprising signal in calm market), THE Signal_Scorer SHALL produce the same half-life as the current fixed system (τ_base).
|
||||
7. FOR ALL combinations of impact, surprise, and market reaction values, THE Signal_Scorer SHALL produce adaptive half-lives that are greater than or equal to τ_base (adaptive decay is always slower or equal to the base decay, never faster).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 6: Volatility-Adjusted Normalization (Regime-Aware Scoring)
|
||||
|
||||
**User Story:** As a quantitative analyst, I want signal weights normalized by current market volatility and volume conditions, so that the same signal magnitude is interpreted differently in calm vs volatile markets.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN market data is available for a ticker, THE Signal_Scorer SHALL compute a return z-score z_r = (r_t - μ_20) / σ_20, where r_t is the current return, μ_20 is the 20-day mean return, and σ_20 is the 20-day return standard deviation.
|
||||
2. WHEN market data is available for a ticker, THE Signal_Scorer SHALL compute a volume z-score z_v = (log(V_t) - μ_V) / σ_V, where V_t is the current volume, μ_V is the 20-day mean of log-volume, and σ_V is the 20-day standard deviation of log-volume.
|
||||
3. THE Signal_Scorer SHALL compute a regime multiplier M_regime = 1 + 0.15·|z_r| + 0.10·|z_v|, which amplifies signal weights during abnormal market conditions.
|
||||
4. THE Signal_Scorer SHALL clamp M_regime to the range [1.0, 2.5] to prevent extreme z-scores from producing runaway weight amplification.
|
||||
5. WHEN market data is not available for a ticker, THE Signal_Scorer SHALL use M_regime = 1.0 (no regime adjustment).
|
||||
6. THE Signal_Scorer SHALL replace the current market context multiplier (M_context) with M_regime in the combined weight formula.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 7: Regime Detection and Classification
|
||||
|
||||
**User Story:** As a quantitative analyst, I want the system to detect and classify the current market regime for each ticker, so that scoring thresholds and behavior adapt to whether the market is trending, panicking, mean-reverting, or uncertain.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN market data is available, THE Regime_Detector SHALL compute a trend indicator R = sign(EMA_20 - EMA_100), where EMA_20 and EMA_100 are exponential moving averages of closing prices over 20 and 100 days respectively.
|
||||
2. WHEN market data is available, THE Regime_Detector SHALL compute a volatility ratio V_r = σ_20 / σ_100, where σ_20 and σ_100 are the 20-day and 100-day return standard deviations.
|
||||
3. THE Regime_Detector SHALL classify the market regime into one of four categories based on R and V_r: trend-following (R ≠ 0 AND V_r < 1.2), panic (V_r > 1.5), mean-reversion (R = 0 AND V_r < 1.0), uncertainty (all other cases).
|
||||
4. WHEN the regime is classified as panic, THE Aggregation_Engine SHALL reduce the bullish/bearish threshold from ±0.15 to ±0.10 (making the system more sensitive to directional signals during high-volatility periods).
|
||||
5. WHEN the regime is classified as mean-reversion, THE Aggregation_Engine SHALL increase the bullish/bearish threshold from ±0.15 to ±0.20 (requiring stronger evidence for directional calls in range-bound markets).
|
||||
6. WHEN the regime is classified as trend-following, THE Aggregation_Engine SHALL use the default thresholds of ±0.15.
|
||||
7. WHEN the regime is classified as uncertainty, THE Aggregation_Engine SHALL use the default thresholds of ±0.15 and increase the contradiction penalty multiplier from 0.4 to 0.6.
|
||||
8. THE Regime_Detector SHALL persist the current regime classification per ticker to the database for auditability and dashboard display.
|
||||
9. WHEN market data is insufficient to compute EMA_100 (fewer than 100 days of price history), THE Regime_Detector SHALL default to the uncertainty regime.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 8: Bayesian Posterior Confidence Replacing Heuristic Confidence
|
||||
|
||||
**User Story:** As a quantitative analyst, I want trend confidence derived from the Bayesian posterior distribution rather than the current heuristic weighted formula, so that confidence reflects actual evidence concentration rather than an ad-hoc combination of factors.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN computing trend confidence, THE Trend_Assembler SHALL use the Bayesian confidence C = 1 - 4αβ/(α+β)² from the Beta_Distribution posterior (Requirement 1) as the primary confidence component with weight 0.5.
|
||||
2. THE Trend_Assembler SHALL retain the source count factor (min(N_unique/15, 0.8)) as a secondary confidence component with weight 0.25, rewarding evidence breadth.
|
||||
3. THE Trend_Assembler SHALL retain the contradiction penalty (contradiction_score × 0.4) as a confidence reduction.
|
||||
4. THE Trend_Assembler SHALL compute the combined confidence as: confidence = 0.5 × C_bayesian + 0.25 × F_count + 0.25 × C_avg_credibility - P_contradiction, clamped to [0.0, 1.0].
|
||||
5. THE Trend_Assembler SHALL preserve the existing confidence thresholds for recommendation eligibility (0.35 minimum, 0.50 paper, 0.70 live) without modification.
|
||||
6. FOR ALL signal sets where all signals agree on direction, THE Trend_Assembler SHALL produce Bayesian confidence that increases monotonically with the number of agreeing signals.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 9: Entropy-Based Mixed Signal Detection
|
||||
|
||||
**User Story:** As a quantitative analyst, I want mixed trend detection based on Shannon entropy rather than simple contradiction thresholds, so that the system can distinguish between genuine uncertainty (high entropy) and weak signal (low total weight).
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN the bullish probability P_bull has been computed from the Bayesian posterior, THE Trend_Assembler SHALL compute Shannon entropy H = -P_bull·log₂(P_bull) - (1-P_bull)·log₂(1-P_bull).
|
||||
2. WHEN H > 0.9 (entropy close to maximum of 1.0, indicating near-equal probability of bullish and bearish), THE Trend_Assembler SHALL classify the trend direction as mixed, regardless of the weighted sentiment average.
|
||||
3. WHEN H ≤ 0.9 AND P_bull > 0.65, THE Trend_Assembler SHALL classify the trend direction as bullish.
|
||||
4. WHEN H ≤ 0.9 AND P_bull < 0.35, THE Trend_Assembler SHALL classify the trend direction as bearish.
|
||||
5. WHEN H ≤ 0.9 AND 0.35 ≤ P_bull ≤ 0.65, THE Trend_Assembler SHALL classify the trend direction as neutral.
|
||||
6. THE Trend_Assembler SHALL persist the entropy value H alongside the trend summary for auditability.
|
||||
7. FOR ALL P_bull values in (0, 1), THE Trend_Assembler SHALL compute entropy values in (0, 1], with maximum entropy of 1.0 occurring at P_bull = 0.5.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 10: Multiplicative Macro Exposure Scoring
|
||||
|
||||
**User Story:** As a quantitative analyst, I want macro impact computed using multiplicative exposure rather than linear weighted sums, so that a company exposed across multiple dimensions receives compounding impact rather than simple addition.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN computing macro impact for a company, THE Macro_Scorer SHALL use the multiplicative exposure formula S_macro = severity · (1 - Π_k(1 - w_k · O_k)), where O_k are the overlap components (geographic, supply chain, commodity, sector) and w_k are their respective weights.
|
||||
2. THE Macro_Scorer SHALL use the following overlap weights: w_geo = 0.35, w_supply = 0.25, w_commodity = 0.25, w_sector = 0.15 (matching the current linear weight distribution).
|
||||
3. WHEN a company has zero overlap across all dimensions, THE Macro_Scorer SHALL produce S_macro = 0.0 (no impact).
|
||||
4. WHEN a company has maximum overlap across all dimensions (all O_k = 1.0), THE Macro_Scorer SHALL produce S_macro = severity · (1 - (1-0.35)·(1-0.25)·(1-0.25)·(1-0.15)), which is approximately severity · 0.724.
|
||||
5. THE Macro_Scorer SHALL preserve the existing severity weight mapping (critical=1.0, high=0.75, moderate=0.5, low=0.25).
|
||||
6. THE Macro_Scorer SHALL preserve the existing resilience modifier (R_tier) applied after the multiplicative exposure computation.
|
||||
7. FOR ALL overlap configurations, THE Macro_Scorer SHALL produce impact scores where adding a non-zero overlap in any dimension increases the total impact (monotonicity property).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 11: Conditional Macro Signal Integration
|
||||
|
||||
**User Story:** As a quantitative analyst, I want macro signals treated as conditional modifiers on company signals rather than additive contributions, so that macro context amplifies or dampens existing company-level evidence rather than independently shifting the trend.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN both company signals and macro signals exist for a ticker, THE Aggregation_Engine SHALL apply macro impact as a multiplicative modifier on the company signal strength: S_adjusted = S_company · (1 + M_macro · sign_alignment), where M_macro is the normalized macro impact and sign_alignment is +1 when macro and company signals agree in direction, -1 when they disagree.
|
||||
2. THE Aggregation_Engine SHALL clamp the macro modifier (1 + M_macro · sign_alignment) to the range [0.5, 1.5] to prevent macro signals from inverting or excessively amplifying company signals.
|
||||
3. WHEN only macro signals exist (no company signals), THE Aggregation_Engine SHALL fall back to the current additive behavior with the existing macro weight of 0.3, preserving the macro-only suppression safety mechanism.
|
||||
4. WHEN only company signals exist (macro layer disabled or no macro events), THE Aggregation_Engine SHALL use company signals without modification (modifier = 1.0).
|
||||
5. THE Aggregation_Engine SHALL log the macro modifier value applied to each ticker for auditability.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 12: Graph-Distance Competitive Signal Attenuation
|
||||
|
||||
**User Story:** As a quantitative analyst, I want competitive signal transfer attenuated by network graph distance and historical correlation, so that signals propagate more strongly to closely related competitors and decay for distant relationships.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN propagating a signal from a source company to a target company, THE Competitive_Scorer SHALL compute transfer strength as S_transfer = S_source · ρ_historical · e^(-d_network), where S_source is the source signal strength, ρ_historical is the historical price correlation between the two companies, and d_network is the graph distance in the competitor relationship network.
|
||||
2. THE Competitive_Scorer SHALL compute graph distance d_network as the shortest path length in the competitor relationship graph, where direct competitors have distance 1, competitors-of-competitors have distance 2, and so on.
|
||||
3. WHEN the graph distance exceeds 3, THE Competitive_Scorer SHALL not propagate the signal (e^(-3) ≈ 0.05, below meaningful contribution).
|
||||
4. THE Competitive_Scorer SHALL compute ρ_historical as the 90-day rolling Pearson correlation of daily returns between the source and target companies.
|
||||
5. WHEN historical correlation data is insufficient (fewer than 30 trading days of overlapping data), THE Competitive_Scorer SHALL use a default correlation of 0.3 for same-sector companies and 0.1 for cross-sector companies.
|
||||
6. THE Competitive_Scorer SHALL preserve the existing relationship strength threshold (R_relationship ≥ 0.2) as a pre-filter before applying the graph-distance attenuation.
|
||||
7. FOR ALL source-target pairs, THE Competitive_Scorer SHALL produce transfer strengths that decrease monotonically with increasing graph distance (closer competitors always receive stronger signal transfer).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 13: Exponentially Weighted Momentum
|
||||
|
||||
**User Story:** As a quantitative analyst, I want trend momentum computed using exponentially weighted historical changes rather than a simple current-minus-previous difference, so that the momentum estimate is smoother and less sensitive to single-cycle noise.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN computing trend momentum, THE Projection_Engine SHALL use an exponentially weighted sum M_t = Σ_{k=0}^{K-1} λ^k · ΔS_{t-k}, where ΔS_{t-k} is the signed strength change at lag k, λ = 0.7 is the decay factor, and K is the number of available historical cycles (up to 10).
|
||||
2. THE Projection_Engine SHALL normalize the momentum by dividing by the geometric series sum Σ λ^k to produce a value in [-1, 1].
|
||||
3. WHEN fewer than 2 historical cycles are available, THE Projection_Engine SHALL fall back to the current heuristic (momentum = direction_sign × strength × 0.5).
|
||||
4. THE Projection_Engine SHALL compute volatility-scaled momentum M_adj = M_t / max(σ_20, 0.01), where σ_20 is the 20-day return standard deviation, to normalize momentum relative to the ticker's typical price movement.
|
||||
5. THE Projection_Engine SHALL clamp M_adj to [-2.0, 2.0] to prevent division by very small σ_20 from producing extreme values.
|
||||
6. FOR ALL sequences of monotonically increasing signed strengths, THE Projection_Engine SHALL produce positive momentum values (correctly detecting strengthening bullish trends).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 14: Expected Value Recommendation Gate
|
||||
|
||||
**User Story:** As a quantitative analyst, I want recommendation eligibility based on expected value rather than simple confidence and strength thresholds, so that the system only recommends trades with positive risk-adjusted expected outcomes.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN evaluating recommendation eligibility, THE Recommendation_Engine SHALL compute expected value EV = P_bull · R_up - P_bear · R_down, where P_bull is the Bayesian bullish probability, P_bear = 1 - P_bull, R_up is the estimated upside return, and R_down is the estimated downside return.
|
||||
2. THE Recommendation_Engine SHALL estimate R_up and R_down from the trend strength and the ticker's 20-day historical volatility: R_up = strength · σ_20 · √(horizon_days) and R_down = (1 - strength) · σ_20 · √(horizon_days), where horizon_days corresponds to the trend window duration.
|
||||
3. WHEN EV is positive and exceeds a configurable threshold (default 0.005, representing 0.5% expected return), THE Recommendation_Engine SHALL allow the recommendation to proceed through the existing eligibility gates.
|
||||
4. WHEN EV is negative or below the threshold, THE Recommendation_Engine SHALL force the recommendation to informational mode regardless of confidence and strength.
|
||||
5. THE Recommendation_Engine SHALL persist the computed EV alongside the recommendation for auditability.
|
||||
6. THE Recommendation_Engine SHALL preserve all existing eligibility gates (confidence ≥ 0.35, strength ≥ 0.10, contradiction ≤ 0.60, evidence ≥ 2, direction ≠ neutral) as additional requirements beyond the EV gate.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 15: Contradiction Handling via Weighted Disagreement Entropy
|
||||
|
||||
**User Story:** As a quantitative analyst, I want contradiction detection to use weighted disagreement entropy rather than a simple minority/majority ratio, so that the system better distinguishes between a few strong dissenting signals and many weak ones.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN computing contradiction, THE Trend_Assembler SHALL compute weighted disagreement entropy using the effective weight distribution across positive and negative signal groups.
|
||||
2. THE Trend_Assembler SHALL compute the positive weight fraction f_pos = W_positive / (W_positive + W_negative) and negative weight fraction f_neg = W_negative / (W_positive + W_negative), where W_positive and W_negative are the sums of effective weights (combined_weight × impact_score) for each sentiment group.
|
||||
3. THE Trend_Assembler SHALL compute contradiction entropy as H_contradiction = -f_pos·log₂(f_pos) - f_neg·log₂(f_neg), normalized to [0, 1] (maximum at f_pos = f_neg = 0.5).
|
||||
4. THE Trend_Assembler SHALL weight the contradiction entropy by the total evidence mass: contradiction_score = H_contradiction · min(1.0, (W_positive + W_negative) / W_threshold), where W_threshold is a configurable parameter (default 5.0) representing the evidence mass at which contradiction becomes fully significant.
|
||||
5. WHEN only positive or only negative signals exist (no disagreement), THE Trend_Assembler SHALL produce a contradiction score of 0.0.
|
||||
6. THE Trend_Assembler SHALL preserve the existing `ContradictionResult` interface, populating the overall score with the entropy-based value and retaining the `DisagreementDetail` objects for catalyst-level analysis.
|
||||
7. FOR ALL signal sets with both positive and negative signals, THE Trend_Assembler SHALL produce contradiction scores that increase monotonically as the weight distribution approaches equal split (f_pos → 0.5).
|
||||
|
||||
---
|
||||
|
||||
### Requirement 16: Backward Compatibility and Migration
|
||||
|
||||
**User Story:** As a platform operator, I want the mathematical upgrades to be backward-compatible with the existing database schema and deployable incrementally, so that the upgrade does not require downtime or data migration.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Aggregation_Engine SHALL preserve the existing `WeightedSignal`, `SignalWeight`, `TrendSummary`, and `Recommendation` dataclass interfaces, adding new fields as optional attributes with default values.
|
||||
2. THE Aggregation_Engine SHALL store new mathematical outputs (P_bull, α, β, entropy, regime, EV) in the existing JSONB metadata fields of `trend_windows` and `recommendations` tables rather than requiring new columns.
|
||||
3. THE Aggregation_Engine SHALL support a feature flag `probabilistic_scoring_enabled` in `risk_configs` that toggles between the current heuristic pipeline and the new probabilistic pipeline, defaultable to `false` for safe rollout.
|
||||
4. WHEN `probabilistic_scoring_enabled` is false, THE Aggregation_Engine SHALL produce identical outputs to the current system (no behavioral change).
|
||||
5. WHEN `probabilistic_scoring_enabled` is true, THE Aggregation_Engine SHALL use the new Bayesian, regime-aware, and adaptive formulas for all pipeline stages.
|
||||
6. IF the feature flag toggle fails to read from the database, THEN THE Aggregation_Engine SHALL default to the current heuristic pipeline (fail-safe behavior).
|
||||
7. THE Aggregation_Engine SHALL log which pipeline mode (heuristic or probabilistic) is active at the start of each aggregation cycle.
|
||||
|
||||
---
|
||||
|
||||
### Requirement 17: Property-Based Testing for Mathematical Correctness
|
||||
|
||||
**User Story:** As a developer, I want comprehensive property-based tests validating the mathematical correctness of all new formulas, so that edge cases and numerical stability issues are caught before deployment.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE test suite SHALL include property-based tests using Hypothesis for the sigmoid confidence gate verifying monotonicity (higher confidence input always produces higher or equal gate output) across all float inputs in [0.0, 1.0].
|
||||
2. THE test suite SHALL include property-based tests for the Beta_Distribution posterior verifying that α + β increases monotonically with the number of signals processed (evidence always accumulates).
|
||||
3. THE test suite SHALL include property-based tests for the Bayesian confidence formula verifying that confidence is 0.0 when α = β (maximum uncertainty) and approaches 1.0 as the ratio α/β or β/α increases.
|
||||
4. THE test suite SHALL include property-based tests for the adaptive decay verifying that the adaptive half-life is always greater than or equal to the base half-life for all valid input combinations.
|
||||
5. THE test suite SHALL include property-based tests for the multiplicative macro exposure verifying monotonicity (adding non-zero overlap in any dimension increases total impact).
|
||||
6. THE test suite SHALL include property-based tests for the exponentially weighted momentum verifying that monotonically increasing strength sequences produce positive momentum.
|
||||
7. THE test suite SHALL include a round-trip property test verifying that computing the Beta posterior from signals, extracting P_bull, then reconstructing approximate signal weights produces values consistent with the original inputs.
|
||||
8. THE test suite SHALL include property-based tests for the expected value computation verifying that EV is positive when P_bull > 0.5 and R_up > R_down (basic directional consistency).
|
||||
9. THE test suite SHALL include property-based tests for numerical stability verifying that no formula produces NaN, infinity, or values outside documented ranges for any valid input combination.
|
||||
10. THE test suite SHALL use `@settings(max_examples=100)` and follow the project convention of `test_pbt_*` file naming.
|
||||
@@ -0,0 +1,349 @@
|
||||
# Implementation Plan: Signal Math Upgrade
|
||||
|
||||
## Overview
|
||||
|
||||
Upgrade the Stonks Oracle signal processing pipeline from deterministic heuristic formulas to a probabilistic, regime-aware, and adaptive mathematical framework. Implementation proceeds in layers: foundations (config, schemas, new modules), then each pipeline stage (scoring → trend assembly → macro → competitive → projection → recommendation), then integration wiring, and finally testing. All changes are gated behind the `probabilistic_scoring_enabled` feature flag.
|
||||
|
||||
## Tasks
|
||||
|
||||
- [ ] 1. Foundation: Configuration and schema extensions
|
||||
- [x] 1.1 Extend `ScoringConfig` with probabilistic parameters in `services/aggregation/scoring.py`
|
||||
- Add `probabilistic: bool = False` toggle field
|
||||
- Add sigmoid gate parameters: `sigmoid_steepness`, `sigmoid_midpoint`
|
||||
- Add information gain parameters: `info_gain_lambda`, `info_gain_max`, `default_base_rate`
|
||||
- Add adaptive decay parameters: `adaptive_decay_impact_scale`, `adaptive_decay_surprise_scale`, `adaptive_decay_market_scale`
|
||||
- Add regime multiplier parameters: `regime_return_weight`, `regime_volume_weight`, `regime_multiplier_max`
|
||||
- All new fields must have defaults matching the design document values
|
||||
- _Requirements: 2.5, 3.1, 5.1, 6.3, 16.1_
|
||||
|
||||
- [x] 1.2 Extend `SignalWeight` and `WeightedSignal` dataclasses in `services/aggregation/scoring.py`
|
||||
- Add optional fields to `SignalWeight`: `sigmoid_gate`, `info_gain_factor`, `source_accuracy_factor`, `regime_multiplier`
|
||||
- Add optional fields to `WeightedSignal`: `info_gain_factor`, `source_accuracy_factor`, `adaptive_half_life`
|
||||
- All new fields must have defaults (None or 1.0) for backward compatibility
|
||||
- _Requirements: 16.1, 2.5, 3.3, 4.2_
|
||||
|
||||
- [x] 1.3 Extend `TrendSummary` Pydantic model in `services/shared/schemas.py`
|
||||
- Add optional fields: `p_bull`, `alpha`, `beta_param`, `bayesian_confidence`, `entropy`, `regime`, `pipeline_mode`
|
||||
- `pipeline_mode` defaults to `"heuristic"`; all others default to `None`
|
||||
- _Requirements: 16.1, 1.6, 9.6_
|
||||
|
||||
- [x] 1.4 Extend `Recommendation` model in `services/shared/schemas.py` (or `services/recommendation/eligibility.py`)
|
||||
- Add optional fields: `expected_value`, `p_bull`, `pipeline_mode`
|
||||
- `pipeline_mode` defaults to `"heuristic"`; all others default to `None`
|
||||
- _Requirements: 16.1, 14.5_
|
||||
|
||||
- [x] 1.5 Add `probabilistic_scoring_enabled` feature flag support in `services/shared/config.py`
|
||||
- Read `probabilistic_scoring_enabled` from `risk_configs.config` JSONB
|
||||
- Default to `False` when key is missing, value is invalid, or DB is unreachable
|
||||
- Propagate flag through `AggregationConfig` dataclass
|
||||
- Log which pipeline mode is active at cycle start
|
||||
- _Requirements: 16.3, 16.4, 16.5, 16.6, 16.7_
|
||||
|
||||
- [x] 1.6 Create database migration `infra/migrations/034_source_accuracy.sql`
|
||||
- Create `source_accuracy` table with columns: `id UUID PRIMARY KEY DEFAULT gen_random_uuid()`, `source_id VARCHAR(200) NOT NULL`, `accuracy_ratio FLOAT NOT NULL DEFAULT 0.5`, `sample_count INTEGER NOT NULL DEFAULT 0`, `last_updated TIMESTAMPTZ`, `created_at TIMESTAMPTZ`
|
||||
- Add `UNIQUE(source_id)` constraint and `idx_source_accuracy_source` index
|
||||
- _Requirements: 4.5_
|
||||
|
||||
- [x] 2. Checkpoint — Verify foundation compiles and existing tests pass
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [ ] 3. New module: Bayesian Accumulator (`services/aggregation/bayesian.py`)
|
||||
- [x] 3.1 Implement `BayesianPosterior` dataclass and `compute_bayesian_posterior` function
|
||||
- Create frozen dataclass with fields: `p_bull`, `alpha`, `beta`, `log_likelihood`, `bayesian_confidence`, `entropy`, `signal_count`
|
||||
- Define `PRIOR` class-level constant for uninformative prior (p_bull=0.5, α=1.0, β=1.0, C=0.0, H=1.0)
|
||||
- Implement log-likelihood accumulation: `L_t = Σ(w_i · s_i)` using `weight.combined * sentiment_value`
|
||||
- Compute `P_bull = σ(L_t)` via sigmoid function
|
||||
- Compute Beta posterior: `α = 1 + W_bull`, `β = 1 + W_bear` from positive/negative weight sums
|
||||
- Compute Bayesian confidence: `C = 1 - 4αβ/(α+β)²`
|
||||
- Compute Shannon entropy via `compute_entropy`
|
||||
- Return `PRIOR` for empty signal lists
|
||||
- Skip signals with NaN weight or sentiment
|
||||
- _Requirements: 1.1, 1.2, 1.3, 1.4, 1.5, 1.6_
|
||||
|
||||
- [x] 3.2 Implement `compute_entropy` function
|
||||
- Shannon entropy: `H = -p·log₂(p) - (1-p)·log₂(1-p)`
|
||||
- Return 0.0 for p ≤ 0 or p ≥ 1 (edge cases)
|
||||
- Return value in [0, 1] with maximum 1.0 at p=0.5
|
||||
- _Requirements: 9.1, 9.7_
|
||||
|
||||
- [x] 3.3 Write property test for sigmoid gate monotonicity
|
||||
- **Property 1: Sigmoid Gate Monotonicity**
|
||||
- **Validates: Requirements 2.6, 17.1**
|
||||
|
||||
- [x] 3.4 Write property test for Beta posterior evidence accumulation
|
||||
- **Property 2: Beta Posterior Evidence Accumulation**
|
||||
- **Validates: Requirements 1.3, 17.2**
|
||||
|
||||
- [x] 3.5 Write property test for Bayesian confidence symmetry and divergence
|
||||
- **Property 3: Bayesian Confidence Symmetry and Divergence**
|
||||
- **Validates: Requirements 1.4, 17.3**
|
||||
|
||||
- [x] 3.6 Write property test for Bayesian posterior round-trip consistency
|
||||
- **Property 4: Bayesian Posterior Round-Trip Consistency**
|
||||
- **Validates: Requirements 1.7, 17.7**
|
||||
|
||||
- [x] 3.7 Write property test for Shannon entropy range and maximum
|
||||
- **Property 8: Shannon Entropy Range and Maximum**
|
||||
- **Validates: Requirements 9.7**
|
||||
|
||||
- [x] 3.8 Write property test for Bayesian confidence monotonic with agreeing signals
|
||||
- **Property 13: Bayesian Confidence Monotonic with Agreeing Signals**
|
||||
- **Validates: Requirements 8.6**
|
||||
|
||||
- [ ] 4. New module: Regime Detector (`services/aggregation/regime.py`)
|
||||
- [x] 4.1 Implement `MarketRegime` enum, `RegimeClassification` and `RegimeConfig` dataclasses
|
||||
- `MarketRegime`: `TREND_FOLLOWING`, `PANIC`, `MEAN_REVERSION`, `UNCERTAINTY`
|
||||
- `RegimeClassification`: `regime`, `trend_indicator`, `volatility_ratio`, `bullish_threshold`, `bearish_threshold`, `contradiction_penalty_multiplier`
|
||||
- `RegimeConfig`: all configurable parameters with defaults from design
|
||||
- _Requirements: 7.3_
|
||||
|
||||
- [x] 4.2 Implement `compute_ema` and `classify_regime` functions
|
||||
- `compute_ema`: exponential moving average over last N values
|
||||
- `classify_regime`: compute trend indicator `R = sign(EMA_20 - EMA_100)` and volatility ratio `V_r = σ_20 / σ_100`
|
||||
- Classification rules: trend-following (R≠0 AND V_r<1.2), panic (V_r>1.5), mean-reversion (R=0 AND V_r<1.0), uncertainty (all other)
|
||||
- Adjust thresholds per regime: panic→±0.10, mean-reversion→±0.20, trend-following→±0.15, uncertainty→±0.15 with contradiction multiplier 0.6
|
||||
- Default to uncertainty when data is insufficient (<100 days) or σ values are zero
|
||||
- _Requirements: 7.1, 7.2, 7.3, 7.4, 7.5, 7.6, 7.7, 7.9_
|
||||
|
||||
- [ ] 5. New module: Source Accuracy Tracker (`services/aggregation/source_accuracy.py`)
|
||||
- [x] 5.1 Implement `SourceAccuracy` dataclass and database functions
|
||||
- `SourceAccuracy` dataclass with `source_id`, `accuracy_ratio`, `sample_count`, `last_updated`
|
||||
- `accuracy_factor` property: return 1.0 when sample_count < 10, else `0.5 + accuracy_ratio`
|
||||
- `fetch_source_accuracy`: batch fetch from `source_accuracy` table via asyncpg
|
||||
- `update_source_accuracy`: update accuracy metrics from realized price outcomes
|
||||
- Handle DB unreachable: return neutral factor 1.0 for all sources
|
||||
- Clamp corrupted accuracy_ratio to [0.0, 1.0]
|
||||
- _Requirements: 4.1, 4.2, 4.3, 4.4, 4.5_
|
||||
|
||||
- [x] 6. Checkpoint — Verify new modules compile and unit tests pass
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [ ] 7. Signal Scorer upgrades (`services/aggregation/scoring.py`)
|
||||
- [x] 7.1 Implement sigmoid confidence gate
|
||||
- Add `sigmoid_gate(x, steepness, midpoint)` function: `σ(k·(x - midpoint))`
|
||||
- When `probabilistic=True`, replace binary gate with sigmoid gate in `compute_signal_weight`
|
||||
- When `probabilistic=False`, preserve existing binary gate behavior
|
||||
- _Requirements: 2.1, 2.2, 2.3, 2.4, 2.5_
|
||||
|
||||
- [x] 7.2 Implement information gain surprise weighting
|
||||
- Add `EVENT_TYPE_BASE_RATES` constant dict and `DEFAULT_BASE_RATE = 0.1`
|
||||
- Add `compute_info_gain(event_type, lambda_param, max_gain, default_base_rate)` function: `r = 1 + λ·(-log₂ P(event_type))`, clamped to max 3.0
|
||||
- Integrate as multiplicative factor in combined weight when `probabilistic=True`
|
||||
- _Requirements: 3.1, 3.2, 3.3, 3.4, 3.5_
|
||||
|
||||
- [x] 7.3 Implement adaptive recency decay
|
||||
- Add `compute_adaptive_half_life(base_half_life, impact_score, info_gain_factor, market_multiplier, config)` function
|
||||
- Compute `β_impact`, `β_surprise`, `β_market_reaction` scaling factors per design
|
||||
- `τ_i = τ_base · (1 + β_impact) · (1 + β_surprise) · (1 + β_market_reaction)`
|
||||
- When `probabilistic=True`, use adaptive half-life in `recency_weight`; otherwise use fixed
|
||||
- _Requirements: 5.1, 5.2, 5.3, 5.4, 5.5, 5.6, 5.7_
|
||||
|
||||
- [x] 7.4 Implement regime multiplier replacing market context multiplier
|
||||
- Add `compute_regime_multiplier(returns, volumes, config)` function
|
||||
- Compute z-scores for return and volume, then `M_regime = 1 + 0.15·|z_r| + 0.10·|z_v|`
|
||||
- Clamp to [1.0, 2.5]; default to 1.0 when data unavailable or σ=0
|
||||
- When `probabilistic=True`, use `M_regime` instead of `M_context` in combined weight
|
||||
- _Requirements: 6.1, 6.2, 6.3, 6.4, 6.5_
|
||||
|
||||
- [x] 7.5 Integrate source accuracy factor into `compute_signal_weight`
|
||||
- Accept optional `source_accuracy_factor` parameter
|
||||
- When `probabilistic=True`, multiply into combined weight formula
|
||||
- When `probabilistic=False`, ignore (factor = 1.0)
|
||||
- _Requirements: 4.2, 4.3_
|
||||
|
||||
- [x] 7.6 Update `compute_signal_weight` to branch on `probabilistic` flag
|
||||
- When `probabilistic=True`: use sigmoid gate × recency (adaptive) × credibility × (1 + novelty) × info_gain × source_accuracy × regime_multiplier
|
||||
- When `probabilistic=False`: preserve exact current formula (binary gate × recency × credibility × (1 + novelty) × market_context)
|
||||
- Populate all new optional fields on `SignalWeight` and `WeightedSignal`
|
||||
- _Requirements: 16.4, 16.5_
|
||||
|
||||
- [x] 7.7 Write property test for information gain monotonicity
|
||||
- **Property 6: Information Gain Monotonicity**
|
||||
- **Validates: Requirements 3.5**
|
||||
|
||||
- [x] 7.8 Write property test for adaptive decay lower bound
|
||||
- **Property 5: Adaptive Decay Lower Bound**
|
||||
- **Validates: Requirements 5.7, 17.4**
|
||||
|
||||
- [ ] 8. Contradiction upgrade (`services/aggregation/contradiction.py`)
|
||||
- [x] 8.1 Implement weighted disagreement entropy contradiction
|
||||
- Compute `f_pos = W_positive / (W_positive + W_negative)` and `f_neg = 1 - f_pos`
|
||||
- Compute `H_contradiction = -f_pos·log₂(f_pos) - f_neg·log₂(f_neg)`
|
||||
- Weight by evidence mass: `contradiction_score = H_contradiction · min(1.0, (W_pos + W_neg) / W_threshold)`
|
||||
- Return 0.0 when only one direction exists
|
||||
- Preserve existing `ContradictionResult` interface
|
||||
- When `probabilistic=False`, preserve existing minority/majority ratio behavior
|
||||
- _Requirements: 15.1, 15.2, 15.3, 15.4, 15.5, 15.6, 15.7_
|
||||
|
||||
- [x] 8.2 Write property test for contradiction entropy monotonicity
|
||||
- **Property 9: Contradiction Entropy Monotonicity**
|
||||
- **Validates: Requirements 15.7**
|
||||
|
||||
- [ ] 9. Trend Assembly upgrades (`services/aggregation/worker.py`)
|
||||
- [x] 9.1 Integrate Bayesian posterior into trend assembly
|
||||
- When `probabilistic=True`, call `compute_bayesian_posterior` on merged signals
|
||||
- Use Bayesian confidence formula for trend confidence: `0.5 × C_bayesian + 0.25 × F_count + 0.25 × C_avg_credibility - P_contradiction`
|
||||
- Use entropy-based direction: H>0.9→mixed, P_bull>0.65→bullish, P_bull<0.35→bearish, else neutral
|
||||
- Apply regime-adjusted thresholds from `RegimeClassification`
|
||||
- Populate new `TrendSummary` fields: `p_bull`, `alpha`, `beta_param`, `bayesian_confidence`, `entropy`, `regime`, `pipeline_mode`
|
||||
- Store probabilistic outputs in `market_context` JSONB under `"probabilistic"` key
|
||||
- When `probabilistic=False`, preserve exact current heuristic behavior
|
||||
- _Requirements: 1.1, 1.2, 8.1, 8.2, 8.3, 8.4, 8.5, 9.1, 9.2, 9.3, 9.4, 9.5, 9.6, 7.8, 16.4, 16.5_
|
||||
|
||||
- [x] 9.2 Wire regime detection into the aggregation cycle
|
||||
- Call `classify_regime` with closing prices and returns for each ticker
|
||||
- Pass `RegimeClassification` to trend assembly for threshold adjustment
|
||||
- Default to uncertainty regime when market data is unavailable
|
||||
- Persist regime classification in JSONB for auditability
|
||||
- _Requirements: 7.1, 7.2, 7.3, 7.8, 7.9_
|
||||
|
||||
- [ ] 10. Macro scoring upgrade (`services/aggregation/interpolation.py`)
|
||||
- [x] 10.1 Implement multiplicative macro exposure formula
|
||||
- When `probabilistic=True`, compute `S_macro = severity · (1 - Π_k(1 - w_k · O_k))` instead of linear weighted sum
|
||||
- Preserve overlap weights: w_geo=0.35, w_supply=0.25, w_commodity=0.25, w_sector=0.15
|
||||
- Preserve severity mapping and resilience modifier
|
||||
- When `probabilistic=False`, preserve exact current linear formula
|
||||
- _Requirements: 10.1, 10.2, 10.3, 10.4, 10.5, 10.6_
|
||||
|
||||
- [x] 10.2 Implement conditional macro signal integration
|
||||
- When `probabilistic=True` and both company and macro signals exist, apply macro as multiplicative modifier: `S_adjusted = S_company · clamp(1 + M_macro · sign_alignment, 0.5, 1.5)`
|
||||
- When only macro signals exist, fall back to additive behavior with weight 0.3
|
||||
- When only company signals exist, use modifier = 1.0
|
||||
- Log macro modifier value per ticker
|
||||
- When `probabilistic=False`, preserve current additive merge behavior
|
||||
- _Requirements: 11.1, 11.2, 11.3, 11.4, 11.5_
|
||||
|
||||
- [x] 10.3 Write property test for multiplicative macro exposure monotonicity
|
||||
- **Property 7: Multiplicative Macro Exposure Monotonicity**
|
||||
- **Validates: Requirements 10.7, 17.5**
|
||||
|
||||
- [ ] 11. Competitive signal upgrade (`services/aggregation/signal_propagation.py`)
|
||||
- [x] 11.1 Implement graph-distance attenuation for competitive signals
|
||||
- When `probabilistic=True`, compute `S_transfer = S_source · ρ_historical · e^(-d_network)` instead of flat transfer
|
||||
- Compute graph distance as shortest path in competitor relationship graph (cap at 3)
|
||||
- Use 90-day rolling Pearson correlation for `ρ_historical`; default to 0.3 (same-sector) or 0.1 (cross-sector) when insufficient data (<30 days)
|
||||
- Preserve existing relationship strength threshold (R ≥ 0.2) as pre-filter
|
||||
- When `probabilistic=False`, preserve exact current flat transfer behavior
|
||||
- _Requirements: 12.1, 12.2, 12.3, 12.4, 12.5, 12.6, 12.7_
|
||||
|
||||
- [x] 11.2 Write property test for competitive signal distance attenuation
|
||||
- **Property 11: Competitive Signal Distance Attenuation**
|
||||
- **Validates: Requirements 12.7**
|
||||
|
||||
- [ ] 12. Projection upgrade (`services/aggregation/projection.py`)
|
||||
- [x] 12.1 Implement exponentially weighted momentum
|
||||
- When `probabilistic=True`, compute `M_t = Σ_{k=0}^{K-1} λ^k · ΔS_{t-k}` with λ=0.7, K up to 10
|
||||
- Normalize by geometric series sum to produce value in [-1, 1]
|
||||
- Fall back to current heuristic when fewer than 2 historical cycles available
|
||||
- Compute volatility-scaled momentum: `M_adj = M_t / max(σ_20, 0.01)`, clamped to [-2.0, 2.0]
|
||||
- When `probabilistic=False`, preserve exact current simple momentum behavior
|
||||
- _Requirements: 13.1, 13.2, 13.3, 13.4, 13.5, 13.6_
|
||||
|
||||
- [x] 12.2 Write property test for exponentially weighted momentum direction
|
||||
- **Property 10: Exponentially Weighted Momentum Direction**
|
||||
- **Validates: Requirements 13.6, 17.6**
|
||||
|
||||
- [ ] 13. Recommendation upgrade (`services/recommendation/eligibility.py`)
|
||||
- [x] 13.1 Implement expected value recommendation gate
|
||||
- When `probabilistic=True`, compute `EV = P_bull · R_up - P_bear · R_down`
|
||||
- Estimate `R_up = strength · σ_20 · √(horizon_days)` and `R_down = (1 - strength) · σ_20 · √(horizon_days)`
|
||||
- When EV > threshold (default 0.005), allow recommendation through existing gates
|
||||
- When EV ≤ threshold, force recommendation to informational mode
|
||||
- Persist EV in `risk_checks` JSONB of `recommendation_evaluations`
|
||||
- Populate `expected_value`, `p_bull`, `pipeline_mode` on Recommendation model
|
||||
- Preserve all existing eligibility gates as additional requirements
|
||||
- When `probabilistic=False`, skip EV gate entirely
|
||||
- _Requirements: 14.1, 14.2, 14.3, 14.4, 14.5, 14.6_
|
||||
|
||||
- [x] 13.2 Write property test for expected value directional consistency
|
||||
- **Property 12: Expected Value Directional Consistency**
|
||||
- **Validates: Requirements 17.8**
|
||||
|
||||
- [x] 14. Checkpoint — Verify all pipeline stages compile and existing tests still pass
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
- [ ] 15. Integration wiring and feature flag plumbing
|
||||
- [x] 15.1 Wire feature flag through the aggregation worker entry point
|
||||
- Read `probabilistic_scoring_enabled` from `risk_configs` at cycle start in `services/aggregation/worker.py`
|
||||
- Pass flag to `ScoringConfig`, trend assembly, contradiction, macro, competitive, and projection stages
|
||||
- Log pipeline mode at cycle start
|
||||
- Ensure flag is read once per cycle (mid-cycle changes take effect next cycle)
|
||||
- _Requirements: 16.3, 16.6, 16.7_
|
||||
|
||||
- [x] 15.2 Wire source accuracy fetch into the scoring pipeline
|
||||
- At cycle start, batch-fetch source accuracy for all source IDs in the current signal set
|
||||
- Pass `source_accuracy_factor` to `compute_signal_weight` for each signal
|
||||
- Handle DB errors gracefully (default to 1.0)
|
||||
- _Requirements: 4.1, 4.2, 4.3_
|
||||
|
||||
- [x] 15.3 Wire regime detection into the aggregation cycle
|
||||
- Fetch closing prices and returns for each ticker from market data
|
||||
- Call `classify_regime` and pass result to trend assembly and scoring stages
|
||||
- Handle missing market data (default to uncertainty regime)
|
||||
- _Requirements: 7.1, 7.8, 7.9_
|
||||
|
||||
- [x] 15.4 Store probabilistic outputs in existing JSONB columns
|
||||
- Store Bayesian fields in `trend_windows.market_context` JSONB under `"probabilistic"` key
|
||||
- Store EV fields in `recommendation_evaluations.risk_checks` JSONB
|
||||
- Store regime classification in trend window JSONB
|
||||
- _Requirements: 16.2_
|
||||
|
||||
- [ ] 16. Numerical stability and edge case hardening
|
||||
- [x] 16.1 Add input validation and edge case guards across all new functions
|
||||
- Guard `log₂(0)` in entropy and information gain computations
|
||||
- Floor `max(σ_20, 0.01)` for momentum volatility scaling
|
||||
- Default to uncertainty regime when σ values are zero
|
||||
- Return `M_regime = 1.0` when z-score σ = 0
|
||||
- Skip signals with NaN weight or sentiment
|
||||
- Clamp all outputs to documented ranges
|
||||
- _Requirements: 17.9, 6.4_
|
||||
|
||||
- [x] 16.2 Write property test for numerical stability across all formulas
|
||||
- **Property 14: Numerical Stability Across All Formulas**
|
||||
- **Validates: Requirements 17.9, 6.4**
|
||||
|
||||
- [ ] 17. Unit tests for all new and modified modules
|
||||
- [x] 17.1 Write unit tests for Bayesian accumulator (`tests/test_bayesian.py`)
|
||||
- Test uninformative prior (empty signals → P_bull=0.5, α=1, β=1, C=0)
|
||||
- Test specific sigmoid gate values (x=0.5→0.5, x=0.2→<0.05, x=0.8→>0.95)
|
||||
- Test entropy direction mapping (H>0.9→mixed, P_bull>0.65→bullish, etc.)
|
||||
- _Requirements: 1.1, 1.2, 1.3, 1.4, 1.5_
|
||||
|
||||
- [x] 17.2 Write unit tests for regime detector (`tests/test_regime.py`)
|
||||
- Test specific (R, V_r) → expected regime classification
|
||||
- Test threshold adjustments per regime (panic→0.10, mean_reversion→0.20)
|
||||
- Test insufficient data fallback to uncertainty
|
||||
- _Requirements: 7.1, 7.2, 7.3, 7.4, 7.5, 7.6, 7.7, 7.9_
|
||||
|
||||
- [x] 17.3 Write unit tests for source accuracy tracker (`tests/test_source_accuracy.py`)
|
||||
- Test accuracy_factor property: sample_count < 10 → 1.0, else 0.5 + ratio
|
||||
- Test corrupted data clamping
|
||||
- _Requirements: 4.1, 4.2, 4.3_
|
||||
|
||||
- [x] 17.4 Write unit tests for signal scoring upgrades (`tests/test_signal_math_unit.py`)
|
||||
- Test info gain clamp (very rare event → factor ≤ 3.0)
|
||||
- Test default base rate (unknown event type → 0.1)
|
||||
- Test adaptive decay edge cases (all zeros → τ_base, all max → 6×τ_base)
|
||||
- Test zero overlap → zero macro impact
|
||||
- Test max overlap → ≈severity×0.724
|
||||
- Test macro fallback behaviors (only macro → additive, only company → no modifier)
|
||||
- Test graph distance cutoff (d>3 → no propagation)
|
||||
- Test momentum fallback (<2 cycles → heuristic)
|
||||
- Test EV threshold behavior (EV>0.005→proceed, EV≤0.005→informational)
|
||||
- Test feature flag behaviors (flag=false→heuristic, flag=true→probabilistic)
|
||||
- _Requirements: 3.1, 3.4, 5.5, 5.6, 10.3, 10.4, 11.3, 13.3, 14.3, 14.4, 16.4, 16.5_
|
||||
|
||||
- [x] 18. Final checkpoint — Ensure all tests pass
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
|
||||
## Notes
|
||||
|
||||
- Tasks marked with `*` are optional and can be skipped for faster MVP
|
||||
- Each task references specific requirements for traceability
|
||||
- Checkpoints ensure incremental validation after each major phase
|
||||
- Property tests validate the 14 universal correctness properties from the design document
|
||||
- Unit tests validate specific examples, edge cases, and integration points
|
||||
- The design uses Python throughout — no language selection needed
|
||||
- Migration number is 034 (existing migrations go up to 033)
|
||||
- All new dataclass fields use optional defaults for backward compatibility
|
||||
- Feature flag `probabilistic_scoring_enabled` gates every behavioral change
|
||||
Binary file not shown.
@@ -0,0 +1 @@
|
||||
{"specId": "d76705a8-fb91-4fce-b59e-c4b3b0dbbd83", "workflowType": "requirements-first", "specType": "feature"}
|
||||
@@ -0,0 +1,802 @@
|
||||
# Design Document — Trading Feedback Engine
|
||||
|
||||
## Overview
|
||||
|
||||
This design adds a periodic trading performance reporting system to Stonks Oracle. The system collects trading data (P&L, recommendations, positions, risk metrics, model quality), generates structured JSON reports with AI-powered summaries, validates report metrics against live data, and stores reports for retrieval via API.
|
||||
|
||||
The core challenge is fitting AI summarization within the 8k-token context window of the `qwen3.5:9b-fast` model on the local Ollama instance. The design addresses this with a chunking strategy that serializes report section data into ≤6,000-character chunks, summarizes each chunk independently, then merges chunk summaries into a final section summary. This hierarchical summarization approach keeps each LLM call well within the token budget while producing coherent narratives.
|
||||
|
||||
### Design Rationale
|
||||
|
||||
A trading system without periodic performance feedback forces the operator to manually query tables and compute metrics. The feedback engine closes this gap by:
|
||||
|
||||
1. **Automating data collection** — pulling from 7+ tables (trading_decisions, orders, positions, portfolio_snapshots, recommendations, prediction_outcomes, model_metric_snapshots) into a single structured report
|
||||
2. **AI-powered summarization** — using the existing agent infrastructure (ai_agents, AgentConfigResolver, llm_factory) to generate natural-language summaries that highlight trends and anomalies
|
||||
3. **Cross-validation** — comparing computed metrics against live validation data (prediction_outcomes, model_metric_snapshots) and flagging discrepancies >5%
|
||||
4. **Persistent storage** — storing reports as JSONB for historical comparison and trend analysis
|
||||
5. **Scheduled generation** — daily (after market close) and weekly (Saturday) reports via Redis queue jobs
|
||||
|
||||
The design reuses existing infrastructure: asyncpg for persistence, FastAPI for API endpoints, Redis queues for async job processing, the ai_agents/AgentConfigResolver/llm_factory stack for LLM access, and TanStack Query hooks on the frontend.
|
||||
|
||||
---
|
||||
|
||||
## Architecture
|
||||
|
||||
### High-Level Data Flow
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
subgraph "Scheduling (Trigger)"
|
||||
A[Scheduler Service] -->|after 16:30 ET daily| B[Redis Queue<br/>stonks:queue:report_generation]
|
||||
A -->|Saturday weekly| B
|
||||
C[Manual API Trigger] --> B
|
||||
end
|
||||
|
||||
subgraph "Report Generation (Async Worker)"
|
||||
B --> D[Report Generator<br/>services/reporting/generator.py]
|
||||
D -->|1. Collect| E[Data Collector<br/>services/reporting/collector.py]
|
||||
E -->|queries| F[(trading_decisions<br/>orders, positions<br/>portfolio_snapshots<br/>recommendations)]
|
||||
D -->|2. Build sections| G[Section Builder<br/>services/reporting/sections.py]
|
||||
G -->|P&L, accuracy,<br/>positions, risk,<br/>model quality| H[Report Sections]
|
||||
D -->|3. Validate| I[Report Validator<br/>services/reporting/validator.py]
|
||||
I -->|cross-check| J[(prediction_outcomes<br/>model_metric_snapshots)]
|
||||
D -->|4. Summarize| K[AI Summarizer<br/>services/reporting/summarizer.py]
|
||||
K -->|chunk & summarize| L[Report_Summarizer_Agent<br/>via AgentConfigResolver<br/>+ llm_factory]
|
||||
D -->|5. Store| M[(trading_reports table)]
|
||||
end
|
||||
|
||||
subgraph "API Layer"
|
||||
N[GET /api/reports] -->|paginated list| M
|
||||
O[GET /api/reports/:id] -->|full report| M
|
||||
end
|
||||
|
||||
subgraph "Frontend"
|
||||
P[useReports hook] --> N
|
||||
Q[useReport hook] --> O
|
||||
end
|
||||
```
|
||||
|
||||
### Scheduling Strategy
|
||||
|
||||
| Component | Trigger | Cadence |
|
||||
|-----------|---------|---------|
|
||||
| Daily Report | Scheduler after 16:30 ET | Every trading day |
|
||||
| Weekly Report | Scheduler on Saturday | Weekly (Mon–Fri coverage) |
|
||||
| Report Generator Worker | Redis queue consumer | On-demand from queue |
|
||||
| AI Summarizer | Called by generator | Per report section |
|
||||
|
||||
### Chunking Strategy
|
||||
|
||||
The `qwen3.5:9b-fast` model has an 8k-token context window. With the system prompt (~200 tokens) and response budget (~200 tokens), roughly 7,600 tokens remain for input. At ~4 chars/token for structured data, that's ~30,400 characters. The 6,000-character chunk limit provides a 5x safety margin to account for JSON overhead, prompt framing, and tokenization variance.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
A[Section Data<br/>e.g. 15,000 chars] --> B{> 6,000 chars?}
|
||||
B -->|No| C[Single LLM call<br/>→ summary]
|
||||
B -->|Yes| D[Split into chunks<br/>≤ 6,000 chars each]
|
||||
D --> E[Chunk 1 → LLM → summary 1]
|
||||
D --> F[Chunk 2 → LLM → summary 2]
|
||||
D --> G[Chunk N → LLM → summary N]
|
||||
E --> H[Merge summaries<br/>→ final LLM call<br/>→ section summary]
|
||||
F --> H
|
||||
G --> H
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Components and Interfaces
|
||||
|
||||
### New Modules
|
||||
|
||||
| Module | File | Responsibility |
|
||||
|--------|------|----------------|
|
||||
| Report Data Collector | `services/reporting/collector.py` | Queries trading data for a reporting period |
|
||||
| Report Section Builder | `services/reporting/sections.py` | Builds structured report sections from raw data |
|
||||
| Report Validator | `services/reporting/validator.py` | Cross-checks metrics against validation tables |
|
||||
| AI Summarizer | `services/reporting/summarizer.py` | Chunks data and generates AI summaries |
|
||||
| Report Generator | `services/reporting/generator.py` | Orchestrates the full report generation pipeline |
|
||||
| Report Models | `services/reporting/models.py` | Pydantic models for report structure and serialization |
|
||||
|
||||
### Modified Modules
|
||||
|
||||
| Module | File | Changes |
|
||||
|--------|------|---------|
|
||||
| Query API | `services/api/app.py` | 2 new `/api/reports` endpoints |
|
||||
| Redis Keys | `services/shared/redis_keys.py` | New `QUEUE_REPORT_GENERATION` constant |
|
||||
| Frontend Hooks | `frontend/src/api/hooks.ts` | 2 new report hooks |
|
||||
| DB Migration | `infra/migrations/038_trading_reports.sql` | New table + agent seed |
|
||||
|
||||
### Component Interface Details
|
||||
|
||||
#### 1. Report Models (`services/reporting/models.py`)
|
||||
|
||||
```python
|
||||
from __future__ import annotations
|
||||
from datetime import date, datetime
|
||||
from enum import Enum
|
||||
from typing import Optional
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
|
||||
class ReportType(str, Enum):
|
||||
DAILY = "daily"
|
||||
WEEKLY = "weekly"
|
||||
|
||||
|
||||
class ValidationStatus(str, Enum):
|
||||
PASSED = "passed"
|
||||
WARNINGS = "warnings"
|
||||
|
||||
|
||||
class ValidationWarning(BaseModel):
|
||||
field_name: str
|
||||
computed_value: float
|
||||
snapshot_value: float
|
||||
pct_difference: float
|
||||
|
||||
|
||||
class PLSection(BaseModel):
|
||||
realized_pnl: float
|
||||
unrealized_pnl: float
|
||||
daily_return: float
|
||||
cumulative_return: float
|
||||
win_count: int
|
||||
loss_count: int
|
||||
win_rate: float
|
||||
profit_factor: float
|
||||
sharpe_ratio: float
|
||||
summary: str = ""
|
||||
validation_warnings: list[ValidationWarning] = Field(default_factory=list)
|
||||
|
||||
|
||||
class RecommendationAccuracySection(BaseModel):
|
||||
total_evaluated: int
|
||||
act_count: int
|
||||
skip_count: int
|
||||
acted_win_rate: float
|
||||
avg_confidence_acted: float
|
||||
avg_confidence_skipped: float
|
||||
summary: str = ""
|
||||
validation_warnings: list[ValidationWarning] = Field(default_factory=list)
|
||||
|
||||
|
||||
class PositionDetail(BaseModel):
|
||||
ticker: str
|
||||
entry_price: float
|
||||
current_or_exit_price: float
|
||||
pnl: float
|
||||
pnl_pct: float
|
||||
hold_duration_hours: float
|
||||
status: str # "open" or "closed"
|
||||
|
||||
|
||||
class PositionPerformanceSection(BaseModel):
|
||||
positions: list[PositionDetail] = Field(default_factory=list)
|
||||
summary: str = ""
|
||||
|
||||
|
||||
class RiskMetricsSection(BaseModel):
|
||||
current_risk_tier: str
|
||||
portfolio_heat: float
|
||||
max_drawdown: float
|
||||
current_drawdown_pct: float
|
||||
reserve_pool_balance: float
|
||||
circuit_breaker_event_count: int
|
||||
summary: str = ""
|
||||
|
||||
|
||||
class ModelQualityWindow(BaseModel):
|
||||
lookback: str
|
||||
win_rate: float | None
|
||||
directional_accuracy: float | None
|
||||
information_coefficient: float | None
|
||||
calibration_error: float | None
|
||||
brier_score: float | None
|
||||
|
||||
|
||||
class ModelQualitySection(BaseModel):
|
||||
windows: list[ModelQualityWindow] = Field(default_factory=list)
|
||||
summary: str = ""
|
||||
validation_warnings: list[ValidationWarning] = Field(default_factory=list)
|
||||
|
||||
|
||||
class ReportData(BaseModel):
|
||||
"""Top-level report structure stored as JSONB."""
|
||||
pnl: PLSection
|
||||
recommendation_accuracy: RecommendationAccuracySection
|
||||
position_performance: PositionPerformanceSection
|
||||
risk_metrics: RiskMetricsSection
|
||||
model_quality: ModelQualitySection
|
||||
executive_summary: str = ""
|
||||
validation_status: ValidationStatus = ValidationStatus.PASSED
|
||||
generated_at: datetime
|
||||
period_start: date
|
||||
period_end: date
|
||||
report_type: ReportType
|
||||
```
|
||||
|
||||
#### 2. Report Data Collector (`services/reporting/collector.py`)
|
||||
|
||||
```python
|
||||
from __future__ import annotations
|
||||
from dataclasses import dataclass
|
||||
from datetime import date, datetime
|
||||
import asyncpg
|
||||
|
||||
|
||||
@dataclass
|
||||
class CollectedData:
|
||||
"""Raw data collected for a reporting period."""
|
||||
trading_decisions: list[dict]
|
||||
orders: list[dict]
|
||||
open_positions: list[dict]
|
||||
closed_positions: list[dict]
|
||||
portfolio_snapshot: dict | None
|
||||
previous_portfolio_snapshot: dict | None
|
||||
recommendations: list[dict]
|
||||
prediction_outcomes: list[dict]
|
||||
model_metric_snapshots: list[dict]
|
||||
circuit_breaker_events: list[dict]
|
||||
reserve_pool_balance: float
|
||||
|
||||
|
||||
async def collect_report_data(
|
||||
pool: asyncpg.Pool,
|
||||
period_start: date,
|
||||
period_end: date,
|
||||
) -> CollectedData:
|
||||
"""Query all trading data for the reporting period.
|
||||
|
||||
Queries: trading_decisions, orders, positions, portfolio_snapshots,
|
||||
recommendations, prediction_outcomes, model_metric_snapshots,
|
||||
circuit_breaker_events, reserve_pool_ledger.
|
||||
|
||||
Returns CollectedData with all raw query results.
|
||||
If no trading_decisions exist, returns empty lists (zero-activity).
|
||||
"""
|
||||
...
|
||||
```
|
||||
|
||||
#### 3. Report Section Builder (`services/reporting/sections.py`)
|
||||
|
||||
```python
|
||||
from __future__ import annotations
|
||||
from services.reporting.models import (
|
||||
PLSection, RecommendationAccuracySection,
|
||||
PositionPerformanceSection, PositionDetail,
|
||||
RiskMetricsSection, ModelQualitySection, ModelQualityWindow,
|
||||
)
|
||||
from services.reporting.collector import CollectedData
|
||||
|
||||
|
||||
def build_pnl_section(data: CollectedData) -> PLSection:
|
||||
"""Build P&L section from collected data.
|
||||
|
||||
Computes realized/unrealized P&L, daily return, cumulative return,
|
||||
win/loss counts, win rate, profit factor, and Sharpe ratio from
|
||||
portfolio_snapshot and closed positions.
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
def build_recommendation_accuracy_section(data: CollectedData) -> RecommendationAccuracySection:
|
||||
"""Build recommendation accuracy section.
|
||||
|
||||
Joins trading_decisions with prediction_outcomes to compute
|
||||
act/skip breakdown, win rate of acted recommendations, and
|
||||
average confidence of acted vs skipped.
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
def build_position_performance_section(data: CollectedData) -> PositionPerformanceSection:
|
||||
"""Build position performance section.
|
||||
|
||||
Lists each position (open and closed) with entry price,
|
||||
current/exit price, P&L, P&L%, and hold duration.
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
def build_risk_metrics_section(data: CollectedData) -> RiskMetricsSection:
|
||||
"""Build risk metrics section.
|
||||
|
||||
Extracts current risk tier, portfolio heat, max drawdown,
|
||||
current drawdown %, reserve pool balance, and circuit breaker
|
||||
event count from collected data.
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
def build_model_quality_section(data: CollectedData) -> ModelQualitySection:
|
||||
"""Build model quality section.
|
||||
|
||||
Extracts latest model_metric_snapshot values for 7d, 30d, 90d
|
||||
lookback windows.
|
||||
"""
|
||||
...
|
||||
```
|
||||
|
||||
#### 4. Report Validator (`services/reporting/validator.py`)
|
||||
|
||||
```python
|
||||
from __future__ import annotations
|
||||
import asyncpg
|
||||
from services.reporting.models import (
|
||||
ReportData, ValidationStatus, ValidationWarning,
|
||||
)
|
||||
|
||||
|
||||
DISCREPANCY_THRESHOLD_PCT = 5.0
|
||||
|
||||
|
||||
def validate_recommendation_accuracy(
|
||||
section: "RecommendationAccuracySection",
|
||||
prediction_outcomes: list[dict],
|
||||
) -> list[ValidationWarning]:
|
||||
"""Cross-reference reported win rates with prediction_outcomes.
|
||||
|
||||
Compares computed win rate against direction_correct/profitable
|
||||
fields from prediction_outcomes for the same tickers and period.
|
||||
Returns warnings for discrepancies > 5%.
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
def validate_model_quality(
|
||||
section: "ModelQualitySection",
|
||||
metric_snapshots: list[dict],
|
||||
) -> list[ValidationWarning]:
|
||||
"""Compare reported model quality metrics against model_metric_snapshots.
|
||||
|
||||
Flags discrepancies > 5% between computed and snapshot values
|
||||
for win_rate, directional_accuracy, IC, ECE, and Brier score.
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
def compute_validation_status(report: ReportData) -> ValidationStatus:
|
||||
"""Determine overall validation status.
|
||||
|
||||
Returns 'passed' if no warnings across all sections,
|
||||
'warnings' if any section has validation warnings.
|
||||
"""
|
||||
...
|
||||
```
|
||||
|
||||
#### 5. AI Summarizer (`services/reporting/summarizer.py`)
|
||||
|
||||
```python
|
||||
from __future__ import annotations
|
||||
import asyncpg
|
||||
from services.shared.agent_config import AgentConfigResolver
|
||||
|
||||
|
||||
CHUNK_SIZE_LIMIT = 6000 # characters per chunk
|
||||
MAX_SUMMARY_WORDS = 200 # per section summary
|
||||
MAX_EXECUTIVE_SUMMARY_WORDS = 300
|
||||
|
||||
|
||||
def chunk_data(serialized: str, max_chars: int = CHUNK_SIZE_LIMIT) -> list[str]:
|
||||
"""Split serialized data into chunks of at most max_chars.
|
||||
|
||||
Splits on newline boundaries to avoid breaking JSON structures.
|
||||
Each chunk is ≤ max_chars characters.
|
||||
Returns at least one chunk (even if empty input).
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
async def summarize_section(
|
||||
pool: asyncpg.Pool,
|
||||
resolver: AgentConfigResolver,
|
||||
section_name: str,
|
||||
section_data: str,
|
||||
) -> str:
|
||||
"""Generate AI summary for a report section.
|
||||
|
||||
1. Serialize section data to string
|
||||
2. Chunk if > CHUNK_SIZE_LIMIT
|
||||
3. Summarize each chunk via Report_Summarizer_Agent
|
||||
4. If multiple chunks, merge summaries with a final LLM call
|
||||
5. Log each invocation to agent_performance_log
|
||||
6. On failure after max_retries, fall back to deterministic summary
|
||||
|
||||
Uses AgentConfigResolver to resolve agent config by slug
|
||||
'report-summarizer', then llm_factory to build the LLM client.
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
def build_deterministic_summary(section_name: str, section_data: dict) -> str:
|
||||
"""Build a fallback deterministic summary from raw metrics.
|
||||
|
||||
Produces a template-based text summary when AI summarization fails.
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
async def generate_executive_summary(
|
||||
pool: asyncpg.Pool,
|
||||
resolver: AgentConfigResolver,
|
||||
section_summaries: dict[str, str],
|
||||
) -> str:
|
||||
"""Generate executive summary from all section summaries.
|
||||
|
||||
Concatenates section summaries, chunks if needed, and produces
|
||||
a ≤300-word synthesis via the Report_Summarizer_Agent.
|
||||
Falls back to concatenated section summaries on failure.
|
||||
"""
|
||||
...
|
||||
```
|
||||
|
||||
#### 6. Report Generator (`services/reporting/generator.py`)
|
||||
|
||||
```python
|
||||
from __future__ import annotations
|
||||
from datetime import date
|
||||
import asyncpg
|
||||
from services.reporting.models import ReportData, ReportType
|
||||
|
||||
|
||||
async def generate_report(
|
||||
pool: asyncpg.Pool,
|
||||
report_type: ReportType,
|
||||
period_start: date,
|
||||
period_end: date,
|
||||
) -> ReportData:
|
||||
"""Orchestrate full report generation.
|
||||
|
||||
1. Collect data via collector
|
||||
2. Build sections via section builder
|
||||
3. Validate sections via validator
|
||||
4. Generate AI summaries via summarizer
|
||||
5. Generate executive summary
|
||||
6. Assemble final ReportData
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
async def store_report(
|
||||
pool: asyncpg.Pool,
|
||||
report: ReportData,
|
||||
) -> str:
|
||||
"""Store report in trading_reports table.
|
||||
|
||||
Uses INSERT ... ON CONFLICT (report_type, period_start, period_end)
|
||||
DO UPDATE to handle regeneration of existing reports.
|
||||
|
||||
Returns the report UUID.
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
async def process_report_job(
|
||||
pool: asyncpg.Pool,
|
||||
job: dict,
|
||||
) -> None:
|
||||
"""Process a report generation job from the Redis queue.
|
||||
|
||||
Deserializes job payload, calls generate_report + store_report.
|
||||
Handles retries with exponential backoff (up to 3 attempts).
|
||||
Rejects duplicate jobs for the same report_type + period.
|
||||
"""
|
||||
...
|
||||
```
|
||||
|
||||
#### 7. API Endpoints (added to `services/api/app.py`)
|
||||
|
||||
| Endpoint | Method | Parameters | Returns |
|
||||
|----------|--------|------------|---------|
|
||||
| `GET /api/reports` | GET | `report_type`, `start_date`, `end_date`, `limit`, `offset` | Paginated list: id, report_type, period_start, period_end, validation_status, generated_at |
|
||||
| `GET /api/reports/{report_id}` | GET | — | Full report including report_data JSONB |
|
||||
|
||||
#### 8. Frontend Hooks (added to `frontend/src/api/hooks.ts`)
|
||||
|
||||
```typescript
|
||||
export interface ReportListItem {
|
||||
id: string;
|
||||
report_type: string;
|
||||
period_start: string;
|
||||
period_end: string;
|
||||
validation_status: string;
|
||||
generated_at: string;
|
||||
}
|
||||
|
||||
export interface ReportDetail extends ReportListItem {
|
||||
report_data: Record<string, unknown>;
|
||||
created_at: string;
|
||||
}
|
||||
|
||||
export function useReports(params?: {
|
||||
report_type?: string;
|
||||
start_date?: string;
|
||||
end_date?: string;
|
||||
limit?: number;
|
||||
offset?: number;
|
||||
}) {
|
||||
const qs = new URLSearchParams();
|
||||
if (params?.report_type) qs.set('report_type', params.report_type);
|
||||
if (params?.start_date) qs.set('start_date', params.start_date);
|
||||
if (params?.end_date) qs.set('end_date', params.end_date);
|
||||
if (params?.limit) qs.set('limit', String(params.limit));
|
||||
if (params?.offset) qs.set('offset', String(params.offset));
|
||||
const path = `/api/reports${qs.toString() ? '?' + qs : ''}`;
|
||||
return useGet<ReportListItem[]>(['reports', params], 'query', path);
|
||||
}
|
||||
|
||||
export function useReport(id: string | undefined) {
|
||||
return useGet<ReportDetail>(
|
||||
['report', id], 'query', `/api/reports/${id}`, !!id
|
||||
);
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Data Models
|
||||
|
||||
### Database Schema (Migration 038)
|
||||
|
||||
#### trading_reports
|
||||
|
||||
```sql
|
||||
CREATE TABLE IF NOT EXISTS trading_reports (
|
||||
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||
report_type VARCHAR(20) NOT NULL,
|
||||
period_start DATE NOT NULL,
|
||||
period_end DATE NOT NULL,
|
||||
report_data JSONB NOT NULL,
|
||||
validation_status VARCHAR(20) NOT NULL DEFAULT 'passed',
|
||||
generated_at TIMESTAMPTZ NOT NULL,
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW(),
|
||||
CONSTRAINT uq_trading_reports_period UNIQUE (report_type, period_start, period_end),
|
||||
CONSTRAINT chk_report_type CHECK (report_type IN ('daily', 'weekly'))
|
||||
);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS idx_trading_reports_type ON trading_reports(report_type);
|
||||
CREATE INDEX IF NOT EXISTS idx_trading_reports_period ON trading_reports(period_start, period_end);
|
||||
CREATE INDEX IF NOT EXISTS idx_trading_reports_generated ON trading_reports(generated_at DESC);
|
||||
```
|
||||
|
||||
#### Report Summarizer Agent Seed
|
||||
|
||||
```sql
|
||||
INSERT INTO ai_agents (name, slug, purpose, model_provider, model_name, system_prompt, prompt_version, schema_version, temperature, max_tokens, timeout_seconds, max_retries, source)
|
||||
SELECT * FROM (VALUES
|
||||
(
|
||||
'Report Summarizer',
|
||||
'report-summarizer',
|
||||
'Generates concise natural-language summaries of trading performance report sections. Processes chunked data within the 8k-token context window.',
|
||||
'ollama',
|
||||
'qwen3.5:9b-fast',
|
||||
E'You are a concise financial performance analyst. You summarize trading performance data into clear, professional prose.\n\nSTRICT RULES:\n1. Do NOT fabricate any data not present in the input.\n2. Do NOT add opinions, predictions, or recommendations.\n3. Keep each summary under 200 words.\n4. Highlight notable trends, outliers, and changes from prior periods.\n5. Use precise numbers from the input data.\n6. Use a neutral, professional tone.\n7. Return ONLY the summary text. No JSON, no markdown, no commentary.',
|
||||
'report-summarizer-v1',
|
||||
'1.0.0',
|
||||
0.0,
|
||||
1024,
|
||||
60,
|
||||
2,
|
||||
'system'
|
||||
)
|
||||
) AS v(name, slug, purpose, model_provider, model_name, system_prompt, prompt_version, schema_version, temperature, max_tokens, timeout_seconds, max_retries, source)
|
||||
WHERE NOT EXISTS (SELECT 1 FROM ai_agents WHERE slug = 'report-summarizer');
|
||||
```
|
||||
|
||||
### Report JSONB Structure
|
||||
|
||||
The `report_data` column stores a JSON object matching the `ReportData` Pydantic model:
|
||||
|
||||
```json
|
||||
{
|
||||
"pnl": {
|
||||
"realized_pnl": 125.50,
|
||||
"unrealized_pnl": -30.20,
|
||||
"daily_return": 0.012,
|
||||
"cumulative_return": 0.085,
|
||||
"win_count": 8,
|
||||
"loss_count": 3,
|
||||
"win_rate": 0.727,
|
||||
"profit_factor": 2.15,
|
||||
"sharpe_ratio": 1.42,
|
||||
"summary": "AI-generated summary...",
|
||||
"validation_warnings": []
|
||||
},
|
||||
"recommendation_accuracy": {
|
||||
"total_evaluated": 15,
|
||||
"act_count": 8,
|
||||
"skip_count": 7,
|
||||
"acted_win_rate": 0.75,
|
||||
"avg_confidence_acted": 0.72,
|
||||
"avg_confidence_skipped": 0.48,
|
||||
"summary": "AI-generated summary...",
|
||||
"validation_warnings": []
|
||||
},
|
||||
"position_performance": {
|
||||
"positions": [
|
||||
{
|
||||
"ticker": "AAPL",
|
||||
"entry_price": 185.50,
|
||||
"current_or_exit_price": 192.30,
|
||||
"pnl": 68.00,
|
||||
"pnl_pct": 3.66,
|
||||
"hold_duration_hours": 72.5,
|
||||
"status": "open"
|
||||
}
|
||||
],
|
||||
"summary": "AI-generated summary..."
|
||||
},
|
||||
"risk_metrics": {
|
||||
"current_risk_tier": "moderate",
|
||||
"portfolio_heat": 0.12,
|
||||
"max_drawdown": 0.08,
|
||||
"current_drawdown_pct": 0.03,
|
||||
"reserve_pool_balance": 450.00,
|
||||
"circuit_breaker_event_count": 1,
|
||||
"summary": "AI-generated summary..."
|
||||
},
|
||||
"model_quality": {
|
||||
"windows": [
|
||||
{
|
||||
"lookback": "7d",
|
||||
"win_rate": 0.65,
|
||||
"directional_accuracy": 0.62,
|
||||
"information_coefficient": 0.08,
|
||||
"calibration_error": 0.12,
|
||||
"brier_score": 0.22
|
||||
}
|
||||
],
|
||||
"summary": "AI-generated summary...",
|
||||
"validation_warnings": []
|
||||
},
|
||||
"executive_summary": "AI-generated executive summary...",
|
||||
"validation_status": "passed",
|
||||
"generated_at": "2025-01-15T21:30:00Z",
|
||||
"period_start": "2025-01-15",
|
||||
"period_end": "2025-01-15",
|
||||
"report_type": "daily"
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
---
|
||||
|
||||
## Correctness Properties
|
||||
|
||||
*A property is a characteristic or behavior that should hold true across all valid executions of a system — essentially, a formal statement about what the system should do. Properties serve as the bridge between human-readable specifications and machine-verifiable correctness guarantees.*
|
||||
|
||||
The following properties were derived from the acceptance criteria through systematic prework analysis. After reflection, 5 unique properties remain. Report section structure checks (3.1–3.5) are subsumed by the round-trip property — if a ReportData object survives serialization and deserialization, its structure is correct by construction (Pydantic enforces required fields). Validation status computation (4.4) is subsumed by the discrepancy detection property. ISO 8601 datetime formatting (8.4) is verified as part of the round-trip property since Pydantic's JSON serialization uses ISO 8601 by default and the round-trip would fail if datetimes were mangled.
|
||||
|
||||
### Property 1: Chunking Round-Trip and Size Constraint
|
||||
|
||||
*For any* input string, splitting it into chunks with a maximum size limit SHALL produce chunks where (a) every chunk is ≤ the size limit in characters, (b) no chunk is empty (except when the input itself is empty, which produces exactly one empty chunk), and (c) concatenating all chunks in order reconstructs the original input string.
|
||||
|
||||
**Validates: Requirements 2.2**
|
||||
|
||||
### Property 2: Report Serialization Round-Trip
|
||||
|
||||
*For any* valid ReportData object (with valid P&L, recommendation accuracy, position performance, risk metrics, and model quality sections), serializing to JSON and then deserializing back SHALL produce a ReportData object equivalent to the original. All datetime fields in the serialized JSON SHALL be in ISO 8601 format.
|
||||
|
||||
**Validates: Requirements 8.1, 8.2, 8.3, 8.4**
|
||||
|
||||
### Property 3: Validation Discrepancy Detection Correctness
|
||||
|
||||
*For any* pair of computed metric value and snapshot metric value (both finite, non-negative floats), the validation function SHALL produce a warning if and only if the percentage difference exceeds 5%. The percentage difference SHALL be computed as `|computed - snapshot| / snapshot * 100` when snapshot > 0, and SHALL flag any non-zero computed value when snapshot is 0.
|
||||
|
||||
**Validates: Requirements 4.1, 4.2, 4.3, 4.4**
|
||||
|
||||
### Property 4: Recommendation Accuracy Aggregation
|
||||
|
||||
*For any* non-empty list of trading decisions with associated prediction outcomes (each having a boolean `direction_correct`, boolean `profitable`, and float `excess_return_vs_spy`), the computed win rate SHALL equal the count of profitable outcomes divided by total outcomes, the directional accuracy SHALL equal the count of direction-correct outcomes divided by total outcomes, and the average excess return SHALL equal the arithmetic mean of all excess_return_vs_spy values. All three values SHALL be in [0.0, 1.0] for rates and finite for the average.
|
||||
|
||||
**Validates: Requirements 1.4**
|
||||
|
||||
### Property 5: Portfolio Period-Over-Period Delta Computation
|
||||
|
||||
*For any* two valid portfolio snapshots (current and previous) with non-negative portfolio_value, active_pool, reserve_pool, and finite cumulative_return, the period-over-period deltas SHALL equal (current - previous) for each field. When no previous snapshot exists, the deltas SHALL be zero.
|
||||
|
||||
**Validates: Requirements 1.3**
|
||||
|
||||
---
|
||||
|
||||
## Error Handling
|
||||
|
||||
### Data Collection Failures
|
||||
|
||||
| Scenario | Handling |
|
||||
|----------|----------|
|
||||
| No trading_decisions for period | Generate zero-activity report with note "No trading activity during this period" |
|
||||
| No portfolio_snapshot for period | Use most recent snapshot before period_start; if none exists, use zero values |
|
||||
| No prediction_outcomes for period | Skip recommendation accuracy validation; set validation_warnings noting missing data |
|
||||
| No model_metric_snapshots for period | Model quality section shows NULL values for all metrics |
|
||||
| Database connection failure during collection | Propagate error to job processor for retry |
|
||||
|
||||
### AI Summarization Failures
|
||||
|
||||
| Scenario | Handling |
|
||||
|----------|----------|
|
||||
| LLM timeout (>60s) | Retry up to max_retries (from agent config, default 2) |
|
||||
| LLM returns empty response | Treat as failure, retry |
|
||||
| LLM returns response > 200 words | Truncate to 200 words at sentence boundary |
|
||||
| All LLM retries exhausted | Fall back to deterministic template summary |
|
||||
| AgentConfigResolver returns None (agent not found) | Log error, use deterministic summary for all sections |
|
||||
| Chunk merge LLM call fails | Use concatenation of chunk summaries (joined with newlines) |
|
||||
|
||||
### Validation Edge Cases
|
||||
|
||||
| Scenario | Handling |
|
||||
|----------|----------|
|
||||
| Snapshot value is 0 and computed value is non-zero | Flag as warning with pct_difference = 100.0 |
|
||||
| Both snapshot and computed values are 0 | No warning (0% difference) |
|
||||
| Snapshot value is NULL | Skip validation for that metric, no warning |
|
||||
| Computed value is NaN or infinity | Replace with 0.0, log warning |
|
||||
| No prediction_outcomes to cross-reference | Skip recommendation accuracy validation entirely |
|
||||
|
||||
### Report Storage Failures
|
||||
|
||||
| Scenario | Handling |
|
||||
|----------|----------|
|
||||
| Unique constraint violation on insert | Use ON CONFLICT DO UPDATE to upsert |
|
||||
| JSONB serialization failure | Log error with report structure, propagate to job processor |
|
||||
| Report exceeds PostgreSQL JSONB size limit (~255 MB) | Extremely unlikely given report structure; log error if it occurs |
|
||||
|
||||
### Job Processing Failures
|
||||
|
||||
| Scenario | Handling |
|
||||
|----------|----------|
|
||||
| Job fails on first attempt | Retry with exponential backoff: 30s, 60s, 120s |
|
||||
| Job fails after 3 retries | Mark job as failed, log error with full context |
|
||||
| Duplicate job submitted for same period | Reject with log message, return without error |
|
||||
| Redis connection failure | Job stays in queue, picked up on reconnection |
|
||||
|
||||
---
|
||||
|
||||
## Testing Strategy
|
||||
|
||||
### Property-Based Tests (Hypothesis)
|
||||
|
||||
Property-based tests use the Hypothesis library with `@settings(max_examples=100)`. Test files are prefixed `test_pbt_*` per project convention.
|
||||
|
||||
| Property | Test File | What It Tests |
|
||||
|----------|-----------|---------------|
|
||||
| Property 1: Chunking Round-Trip | `tests/test_pbt_report_chunking.py` | `chunk_data()` preserves content and respects size limits |
|
||||
| Property 2: Report Serialization Round-Trip | `tests/test_pbt_report_serialization.py` | `ReportData.model_dump_json()` → `ReportData.model_validate_json()` round-trip |
|
||||
| Property 3: Validation Discrepancy Detection | `tests/test_pbt_report_validation.py` | Discrepancy detection correctly flags >5% differences |
|
||||
| Property 4: Recommendation Accuracy Aggregation | `tests/test_pbt_report_sections.py` | `build_recommendation_accuracy_section()` computes correct aggregates |
|
||||
| Property 5: Portfolio Delta Computation | `tests/test_pbt_report_sections.py` | `build_pnl_section()` computes correct period-over-period deltas |
|
||||
|
||||
Each property test is tagged with a comment referencing the design property:
|
||||
```python
|
||||
# Feature: trading-feedback-engine, Property 1: Chunking round-trip and size constraint
|
||||
```
|
||||
|
||||
### Unit Tests (pytest)
|
||||
|
||||
| Test File | Coverage |
|
||||
|-----------|----------|
|
||||
| `tests/test_report_sections.py` | Section builders with known inputs, edge cases (empty data, single position, zero-activity) |
|
||||
| `tests/test_report_validator.py` | Specific discrepancy scenarios, boundary cases (exactly 5%), NULL snapshot values |
|
||||
| `tests/test_report_summarizer.py` | Deterministic fallback summary, chunk splitting edge cases (empty input, single char) |
|
||||
| `tests/test_report_models.py` | Pydantic model validation, enum constraints, default values |
|
||||
| `tests/test_report_generator.py` | Orchestration with mocked dependencies, zero-activity report, upsert behavior |
|
||||
|
||||
### Integration Tests
|
||||
|
||||
| Test File | Coverage |
|
||||
|-----------|----------|
|
||||
| `tests/test_report_api.py` | API endpoints with seeded database, pagination, filtering by report_type and date range |
|
||||
| `tests/test_report_storage.py` | Store/retrieve round-trip against real asyncpg pool, upsert behavior, unique constraint |
|
||||
|
||||
### Frontend Tests (Vitest)
|
||||
|
||||
| Test File | Coverage |
|
||||
|-----------|----------|
|
||||
| `frontend/src/test/reports.test.ts` | useReports and useReport hooks with MSW mocks, loading/error states |
|
||||
|
||||
### Test Configuration
|
||||
|
||||
- Python PBT: Hypothesis with `@settings(max_examples=100)`, files prefixed `test_pbt_*`
|
||||
- Python unit/integration: pytest with pytest-asyncio for async code
|
||||
- Frontend: Vitest with MSW for deterministic API mocking
|
||||
- Lint: `ruff check services/` before all commits
|
||||
- CI: Woodpecker runs all tests automatically on push to Gitea
|
||||
@@ -0,0 +1,117 @@
|
||||
# Requirements Document
|
||||
|
||||
## Introduction
|
||||
|
||||
The Trading Feedback Engine generates periodic performance reports from the Stonks Oracle trading system. Reports cover trading P&L, recommendation accuracy, position performance, risk metrics, and model quality trends. An AI agent (registered in the `ai_agents` table) summarizes sections of the report by processing data in small chunks that fit within the 8k-token context window. Reports are validated against live data from the prediction outcomes and model metric snapshots tables, stored in the database for retrieval, and exposed via API endpoints.
|
||||
|
||||
## Glossary
|
||||
|
||||
- **Feedback_Engine**: The backend service that orchestrates report generation, data collection, AI summarization, and report storage.
|
||||
- **Report_Summarizer_Agent**: The AI agent registered in the `ai_agents` table that generates natural-language summaries for report sections. Uses the existing `AgentConfigResolver` and `llm_factory` infrastructure.
|
||||
- **Report**: A structured JSON document containing trading performance metrics, AI-generated summaries, and validation data for a specific period (daily or weekly).
|
||||
- **Report_Section**: A self-contained portion of a report (e.g., P&L summary, recommendation accuracy, position performance) that can be independently generated and summarized.
|
||||
- **Chunk**: A subset of data rows small enough to fit within the 8k-token context window when serialized, allowing the Report_Summarizer_Agent to process it in a single LLM call.
|
||||
- **Portfolio_Snapshot**: A daily record in the `portfolio_snapshots` table containing portfolio value, pool balances, returns, win/loss counts, Sharpe ratio, max drawdown, and risk tier.
|
||||
- **Prediction_Outcome**: A record in the `prediction_outcomes` table containing realized returns, direction correctness, and excess returns vs benchmarks for a prediction at a specific horizon.
|
||||
- **Model_Metric_Snapshot**: A record in the `model_metric_snapshots` table containing aggregate model quality metrics (win rate, IC, ECE, Brier score) for a lookback/horizon combination.
|
||||
- **Trading_Decision**: A record in the `trading_decisions` table capturing the act/skip decision, skip reason, position sizing, risk tier, circuit breaker status, and decision trace for a recommendation evaluation.
|
||||
- **Validation_Data**: Live data from `prediction_outcomes`, `model_metric_snapshots`, and `signal_evidence_links` used to cross-check report claims against actual measured performance.
|
||||
- **Query_API**: The existing FastAPI service (`services/api/app.py`) that serves HTTP endpoints for the dashboard and external consumers.
|
||||
|
||||
## Requirements
|
||||
|
||||
### Requirement 1: Report Data Collection
|
||||
|
||||
**User Story:** As a trader, I want the feedback engine to collect all relevant trading data for a reporting period, so that reports reflect the complete picture of trading activity.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN a report generation is triggered for a date range, THE Feedback_Engine SHALL query trading_decisions, orders, positions, portfolio_snapshots, recommendations, prediction_outcomes, and model_metric_snapshots for that period.
|
||||
2. WHEN collecting trading decision data, THE Feedback_Engine SHALL include the decision type, skip reason, ticker, computed position size, risk tier, circuit breaker status, and correlation check result for each Trading_Decision.
|
||||
3. WHEN collecting portfolio data, THE Feedback_Engine SHALL retrieve the most recent Portfolio_Snapshot within the reporting period and compute period-over-period changes in portfolio value, active pool, reserve pool, and cumulative return.
|
||||
4. WHEN collecting recommendation accuracy data, THE Feedback_Engine SHALL join recommendations with Prediction_Outcomes to compute win rate, directional accuracy, and average excess return vs SPY for the period.
|
||||
5. IF no trading_decisions exist for the requested period, THEN THE Feedback_Engine SHALL generate a report with zero-activity sections and a note indicating no trading occurred.
|
||||
|
||||
### Requirement 2: Chunked AI Summarization
|
||||
|
||||
**User Story:** As a trader, I want AI-generated summaries in my reports, so that I can quickly understand performance trends without reading raw numbers.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Report_Summarizer_Agent SHALL be registered in the `ai_agents` table with slug `report-summarizer`, model `qwen3.5:9b-fast`, and source `system`.
|
||||
2. WHEN generating a summary for a Report_Section, THE Feedback_Engine SHALL serialize the section data into Chunks of no more than 6,000 characters each to stay within the 8k-token context window.
|
||||
3. WHEN a Report_Section contains data that exceeds a single Chunk, THE Feedback_Engine SHALL split the data into multiple Chunks, summarize each Chunk independently, and then produce a final merged summary from the individual Chunk summaries.
|
||||
4. WHEN invoking the Report_Summarizer_Agent, THE Feedback_Engine SHALL use the existing `AgentConfigResolver` and `llm_factory` infrastructure to resolve model configuration and build the LLM client.
|
||||
5. WHEN invoking the Report_Summarizer_Agent, THE Feedback_Engine SHALL log each invocation to the `agent_performance_log` table with agent_id, success status, duration_ms, and token estimates.
|
||||
6. IF the Report_Summarizer_Agent fails after max_retries, THEN THE Feedback_Engine SHALL fall back to a deterministic text summary built from the raw metrics and continue report generation.
|
||||
|
||||
### Requirement 3: Report Structure and Content
|
||||
|
||||
**User Story:** As a trader, I want reports to cover P&L, recommendation accuracy, position performance, risk metrics, and model quality, so that I have a comprehensive view of system performance.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Report SHALL contain a P&L section with realized P&L, unrealized P&L, daily return, cumulative return, win count, loss count, win rate, profit factor, and Sharpe ratio for the reporting period.
|
||||
2. THE Report SHALL contain a recommendation accuracy section with total recommendations evaluated, act/skip breakdown, win rate of acted-upon recommendations, and average confidence of acted vs skipped recommendations.
|
||||
3. THE Report SHALL contain a position performance section listing each position held during the period with ticker, entry price, current or exit price, unrealized or realized P&L, P&L percentage, and hold duration.
|
||||
4. THE Report SHALL contain a risk metrics section with current risk tier, portfolio heat, max drawdown, current drawdown percentage, reserve pool balance, and a count of circuit breaker events during the period.
|
||||
5. THE Report SHALL contain a model quality section with the latest Model_Metric_Snapshot values for win rate, directional accuracy, information coefficient, calibration error (ECE), and Brier score across the 7d, 30d, and 90d lookback windows.
|
||||
6. THE Report SHALL contain an AI-generated executive summary that synthesizes the key findings from all sections into a concise narrative of no more than 300 words.
|
||||
|
||||
### Requirement 4: Report Validation Against Live Data
|
||||
|
||||
**User Story:** As a trader, I want report metrics to be cross-checked against live validation data, so that I can trust the accuracy of the reported numbers.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. WHEN generating the recommendation accuracy section, THE Feedback_Engine SHALL cross-reference reported win rates with the `direction_correct` and `profitable` fields from Prediction_Outcomes for the same tickers and period.
|
||||
2. WHEN generating the model quality section, THE Feedback_Engine SHALL compare the reported metrics against the most recent Model_Metric_Snapshot records and flag discrepancies greater than 5% between computed and snapshot values.
|
||||
3. WHEN a validation discrepancy is detected, THE Feedback_Engine SHALL include a `validation_warnings` array in the report section with the field name, computed value, snapshot value, and percentage difference.
|
||||
4. THE Report SHALL include a `validation_status` field set to `passed` when no discrepancies exceed 5%, or `warnings` when one or more discrepancies are detected.
|
||||
|
||||
### Requirement 5: Report Storage and Retrieval
|
||||
|
||||
**User Story:** As a trader, I want reports stored in the database and accessible via API, so that I can review historical performance at any time.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Feedback_Engine SHALL store each generated Report as a row in a `trading_reports` table with columns for id (UUID), report_type (daily/weekly), period_start (DATE), period_end (DATE), report_data (JSONB), validation_status (VARCHAR), generated_at (TIMESTAMPTZ), and created_at (TIMESTAMPTZ).
|
||||
2. THE Feedback_Engine SHALL enforce a unique constraint on (report_type, period_start, period_end) to prevent duplicate reports for the same period.
|
||||
3. WHEN a report for an existing period is regenerated, THE Feedback_Engine SHALL update the existing row with the new report_data, validation_status, and generated_at timestamp.
|
||||
4. THE Query_API SHALL expose a `GET /api/reports` endpoint that returns a paginated list of reports with id, report_type, period_start, period_end, validation_status, and generated_at.
|
||||
5. THE Query_API SHALL expose a `GET /api/reports/{report_id}` endpoint that returns the full report including report_data JSONB.
|
||||
6. THE Query_API SHALL support filtering reports by report_type and date range via query parameters on the `GET /api/reports` endpoint.
|
||||
|
||||
### Requirement 6: Periodic Report Generation
|
||||
|
||||
**User Story:** As a trader, I want reports generated automatically on a daily and weekly schedule, so that I always have up-to-date performance feedback.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Feedback_Engine SHALL generate a daily report after market close (after 16:30 ET) covering the current trading day.
|
||||
2. THE Feedback_Engine SHALL generate a weekly report on Saturday covering the Monday-through-Friday trading week.
|
||||
3. WHEN a scheduled report generation is triggered, THE Feedback_Engine SHALL enqueue a report generation job on a Redis queue for asynchronous processing.
|
||||
4. IF a report generation job fails, THEN THE Feedback_Engine SHALL retry the job up to 3 times with exponential backoff before marking the job as failed.
|
||||
5. WHILE a report generation job is in progress for a given period, THE Feedback_Engine SHALL reject duplicate job submissions for the same report_type and period.
|
||||
|
||||
### Requirement 7: Agent Registration and Editability
|
||||
|
||||
**User Story:** As a trader, I want the report summarizer agent registered in the ai_agents table, so that I can edit its prompts, model, and parameters through the existing agent management API.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Feedback_Engine SHALL register the Report_Summarizer_Agent in the `ai_agents` table via a database migration with slug `report-summarizer`, source `system`, model_provider `ollama`, and model_name `qwen3.5:9b-fast`.
|
||||
2. THE Report_Summarizer_Agent system prompt SHALL instruct the model to produce concise financial performance summaries, avoid fabricating data not present in the input, and keep each summary under 200 words.
|
||||
3. THE Report_Summarizer_Agent SHALL support variant creation and activation through the existing agent variants system, allowing A/B testing of different summarization prompts.
|
||||
4. WHEN the Report_Summarizer_Agent configuration is updated via the agent management API, THE Feedback_Engine SHALL pick up the new configuration within 60 seconds via the `AgentConfigResolver` TTL cache.
|
||||
|
||||
### Requirement 8: Report Serialization Round-Trip
|
||||
|
||||
**User Story:** As a developer, I want report data to survive serialization and deserialization without data loss, so that stored reports are always faithful to the generated content.
|
||||
|
||||
#### Acceptance Criteria
|
||||
|
||||
1. THE Feedback_Engine SHALL serialize Report objects to JSON for storage in the `report_data` JSONB column.
|
||||
2. THE Feedback_Engine SHALL deserialize stored JSON back into Report objects for API responses.
|
||||
3. FOR ALL valid Report objects, serializing to JSON then deserializing back SHALL produce an equivalent Report object (round-trip property).
|
||||
4. THE Feedback_Engine SHALL use ISO 8601 format for all datetime fields in serialized reports.
|
||||
@@ -0,0 +1,195 @@
|
||||
# Implementation Plan: Trading Feedback Engine
|
||||
|
||||
## Overview
|
||||
|
||||
Add a periodic trading performance reporting system to Stonks Oracle. The system collects trading data, generates structured JSON reports with AI-powered summaries, validates metrics against live data, and stores reports for retrieval via API. Implementation follows the four-phase approach from the design: foundation → validation & AI → generator & API → scheduling & tests.
|
||||
|
||||
## Tasks
|
||||
|
||||
- [x] 1. Database migration 038 — trading_reports table and report-summarizer agent
|
||||
- [x] 1.1 Create `infra/migrations/038_trading_reports.sql`
|
||||
- Create `trading_reports` table with columns: id (UUID PK, gen_random_uuid()), report_type (VARCHAR(20) NOT NULL), period_start (DATE NOT NULL), period_end (DATE NOT NULL), report_data (JSONB NOT NULL), validation_status (VARCHAR(20) NOT NULL DEFAULT 'passed'), generated_at (TIMESTAMPTZ NOT NULL), created_at (TIMESTAMPTZ NOT NULL DEFAULT NOW())
|
||||
- Add UNIQUE constraint on (report_type, period_start, period_end)
|
||||
- Add CHECK constraint: report_type IN ('daily', 'weekly')
|
||||
- Create indexes: idx_trading_reports_type, idx_trading_reports_period, idx_trading_reports_generated
|
||||
- Seed Report_Summarizer_Agent into ai_agents table with slug 'report-summarizer', model_provider 'ollama', model_name 'qwen3.5:9b-fast', source 'system', temperature 0.0, max_tokens 1024, timeout_seconds 60, max_retries 2
|
||||
- Use WHERE NOT EXISTS guard on agent insert to be idempotent
|
||||
- _Requirements: 5.1, 5.2, 7.1, 7.2_
|
||||
|
||||
- [x] 1.2 Add `QUEUE_REPORT_GENERATION` constant to `services/shared/redis_keys.py`
|
||||
- Add `QUEUE_REPORT_GENERATION = "report_generation"` following existing queue naming convention
|
||||
- _Requirements: 6.3_
|
||||
|
||||
- [x] 2. Phase 1 — Report models, data collector, and section builders
|
||||
- [x] 2.1 Create report models (`services/reporting/models.py`)
|
||||
- Create `services/reporting/__init__.py`
|
||||
- Define enums: ReportType (daily, weekly), ValidationStatus (passed, warnings)
|
||||
- Define Pydantic models: ValidationWarning, PLSection, RecommendationAccuracySection, PositionDetail, PositionPerformanceSection, RiskMetricsSection, ModelQualityWindow, ModelQualitySection, ReportData
|
||||
- ReportData includes all sections, executive_summary, validation_status, generated_at, period_start, period_end, report_type
|
||||
- _Requirements: 3.1, 3.2, 3.3, 3.4, 3.5, 3.6, 8.1, 8.2, 8.4_
|
||||
|
||||
- [x] 2.2 Implement data collector (`services/reporting/collector.py`)
|
||||
- Define CollectedData dataclass with fields: trading_decisions, orders, open_positions, closed_positions, portfolio_snapshot, previous_portfolio_snapshot, recommendations, prediction_outcomes, model_metric_snapshots, circuit_breaker_events, reserve_pool_balance
|
||||
- Implement `collect_report_data(pool, period_start, period_end)` → CollectedData
|
||||
- Query trading_decisions, orders, positions (open + closed), portfolio_snapshots (current + previous), recommendations, prediction_outcomes, model_metric_snapshots, circuit_breaker_events, reserve_pool_ledger for the period
|
||||
- Return empty lists for tables with no data (zero-activity case)
|
||||
- Use `_row_dict()` pattern for UUID conversion from asyncpg rows
|
||||
- _Requirements: 1.1, 1.2, 1.3, 1.4, 1.5_
|
||||
|
||||
- [x] 2.3 Implement section builders (`services/reporting/sections.py`)
|
||||
- Implement `build_pnl_section(data: CollectedData) -> PLSection` — compute realized/unrealized P&L, daily return, cumulative return, win/loss counts, win rate, profit factor, Sharpe ratio from portfolio_snapshot and closed positions
|
||||
- Implement `build_recommendation_accuracy_section(data: CollectedData) -> RecommendationAccuracySection` — join trading_decisions with prediction_outcomes, compute act/skip breakdown, win rate of acted, avg confidence acted vs skipped
|
||||
- Implement `build_position_performance_section(data: CollectedData) -> PositionPerformanceSection` — list each position with ticker, entry price, current/exit price, P&L, P&L%, hold duration
|
||||
- Implement `build_risk_metrics_section(data: CollectedData) -> RiskMetricsSection` — extract risk tier, portfolio heat, max drawdown, current drawdown %, reserve pool balance, circuit breaker event count
|
||||
- Implement `build_model_quality_section(data: CollectedData) -> ModelQualitySection` — extract model_metric_snapshot values for 7d, 30d, 90d lookback windows
|
||||
- Handle zero-activity gracefully (zero values, empty lists)
|
||||
- _Requirements: 1.3, 1.4, 3.1, 3.2, 3.3, 3.4, 3.5_
|
||||
|
||||
- [x] 3. Checkpoint — Verify foundation modules
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
- Run `.venv/bin/ruff check services/reporting/`
|
||||
- Run `.venv/bin/python -m pytest tests/ -x --tb=short -q -k "report"` to verify models and section builders
|
||||
|
||||
- [x] 4. Phase 2 — Report validator and AI summarizer
|
||||
- [x] 4.1 Implement report validator (`services/reporting/validator.py`)
|
||||
- Define `DISCREPANCY_THRESHOLD_PCT = 5.0`
|
||||
- Implement `validate_recommendation_accuracy(section, prediction_outcomes)` → list[ValidationWarning] — compare computed win rate against direction_correct/profitable from prediction_outcomes, flag >5% discrepancies
|
||||
- Implement `validate_model_quality(section, metric_snapshots)` → list[ValidationWarning] — compare reported metrics against model_metric_snapshots for win_rate, directional_accuracy, IC, ECE, Brier score, flag >5% discrepancies
|
||||
- Implement `compute_validation_status(report: ReportData)` → ValidationStatus — return 'passed' if no warnings, 'warnings' if any section has validation_warnings
|
||||
- Handle edge cases: snapshot=0 with computed≠0 → 100% difference; both=0 → no warning; snapshot=NULL → skip; computed=NaN → replace with 0.0
|
||||
- _Requirements: 4.1, 4.2, 4.3, 4.4_
|
||||
|
||||
- [x] 4.2 Implement AI summarizer (`services/reporting/summarizer.py`)
|
||||
- Define constants: CHUNK_SIZE_LIMIT=6000, MAX_SUMMARY_WORDS=200, MAX_EXECUTIVE_SUMMARY_WORDS=300
|
||||
- Implement `chunk_data(serialized: str, max_chars: int)` → list[str] — split on newline boundaries, each chunk ≤ max_chars, at least one chunk returned
|
||||
- Implement `summarize_section(pool, resolver, section_name, section_data)` → str — serialize, chunk if needed, summarize each chunk via Report_Summarizer_Agent (resolved by slug 'report-summarizer'), merge if multiple chunks, log to agent_performance_log, fall back to deterministic on failure
|
||||
- Implement `build_deterministic_summary(section_name, section_data)` → str — template-based fallback summary from raw metrics
|
||||
- Implement `generate_executive_summary(pool, resolver, section_summaries)` → str — concatenate section summaries, chunk if needed, produce ≤300-word synthesis, fall back to concatenation on failure
|
||||
- Use AgentConfigResolver + llm_factory for LLM access
|
||||
- Log each invocation to agent_performance_log with agent_id, success, duration_ms, token estimates
|
||||
- _Requirements: 2.1, 2.2, 2.3, 2.4, 2.5, 2.6, 3.6_
|
||||
|
||||
- [x] 5. Checkpoint — Verify validator and summarizer
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
- Run `.venv/bin/ruff check services/reporting/`
|
||||
- Run `.venv/bin/python -m pytest tests/ -x --tb=short -q -k "report"` to verify validator and summarizer
|
||||
|
||||
- [x] 6. Phase 3 — Report generator orchestrator and API endpoints
|
||||
- [x] 6.1 Implement report generator (`services/reporting/generator.py`)
|
||||
- Implement `generate_report(pool, report_type, period_start, period_end)` → ReportData — orchestrate: collect data → build sections → validate → summarize → assemble ReportData
|
||||
- Implement `store_report(pool, report)` → str (UUID) — INSERT ... ON CONFLICT (report_type, period_start, period_end) DO UPDATE for upsert, return report id
|
||||
- Implement `process_report_job(pool, job: dict)` → None — deserialize job payload, call generate_report + store_report, handle retries with exponential backoff (30s, 60s, 120s up to 3 attempts), reject duplicate jobs for same report_type + period
|
||||
- _Requirements: 5.1, 5.2, 5.3, 6.3, 6.4, 6.5_
|
||||
|
||||
- [x] 6.2 Add API endpoints to `services/api/app.py`
|
||||
- Add `GET /api/reports` — paginated list with query params: report_type, start_date, end_date, limit (default 20), offset (default 0); returns id, report_type, period_start, period_end, validation_status, generated_at
|
||||
- Add `GET /api/reports/{report_id}` — full report including report_data JSONB
|
||||
- Use asyncpg pool from existing app state
|
||||
- Return 404 for non-existent report_id
|
||||
- _Requirements: 5.4, 5.5, 5.6_
|
||||
|
||||
- [x] 6.3 Add frontend hooks to `frontend/src/api/hooks.ts`
|
||||
- Add `ReportListItem` and `ReportDetail` TypeScript interfaces
|
||||
- Implement `useReports(params?)` hook — builds query string from report_type, start_date, end_date, limit, offset; uses `useGet` with 'query' base
|
||||
- Implement `useReport(id)` hook — fetches single report by id, enabled only when id is defined
|
||||
- _Requirements: 5.4, 5.5_
|
||||
|
||||
- [x] 7. Checkpoint — Verify generator and API
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
- Run `.venv/bin/ruff check services/`
|
||||
- Run `.venv/bin/python -m pytest tests/ -x --tb=short -q -k "report"` to verify generator and API endpoints
|
||||
|
||||
- [x] 8. Phase 4 — Scheduling, property-based tests, unit tests, and frontend tests
|
||||
- [x] 8.1 Wire Redis queue integration and scheduler
|
||||
- Add report generation job consumer to the scheduler service that listens on `stonks:queue:report_generation`
|
||||
- Add daily report trigger (after 16:30 ET on trading days) and weekly report trigger (Saturday) to the scheduler
|
||||
- Job payload: `{"report_type": "daily"|"weekly", "period_start": "YYYY-MM-DD", "period_end": "YYYY-MM-DD"}`
|
||||
- _Requirements: 6.1, 6.2, 6.3, 6.4, 6.5_
|
||||
|
||||
- [x] 8.2 Write property test: Chunking Round-Trip and Size Constraint
|
||||
- **Property 1: Chunking Round-Trip and Size Constraint**
|
||||
- File: `tests/test_pbt_report_chunking.py`
|
||||
- Use Hypothesis `@settings(max_examples=100)` with `@given(st.text())` and `@given(st.integers(min_value=1, max_value=10000))`
|
||||
- Assert: every chunk ≤ max_chars, no empty chunks (except empty input → one empty chunk), concatenation of chunks == original input
|
||||
- **Validates: Requirements 2.2**
|
||||
|
||||
- [x] 8.3 Write property test: Report Serialization Round-Trip
|
||||
- **Property 2: Report Serialization Round-Trip**
|
||||
- File: `tests/test_pbt_report_serialization.py`
|
||||
- Use Hypothesis with custom strategies for ReportData (valid PLSection, RecommendationAccuracySection, etc.)
|
||||
- Assert: `ReportData.model_validate_json(report.model_dump_json())` == original report
|
||||
- Assert: all datetime fields in serialized JSON are ISO 8601 format
|
||||
- **Validates: Requirements 8.1, 8.2, 8.3, 8.4**
|
||||
|
||||
- [x] 8.4 Write property test: Validation Discrepancy Detection Correctness
|
||||
- **Property 3: Validation Discrepancy Detection Correctness**
|
||||
- File: `tests/test_pbt_report_validation.py`
|
||||
- Use Hypothesis with `@given(st.floats(min_value=0, max_value=1e6), st.floats(min_value=0, max_value=1e6))`
|
||||
- Assert: warning iff |computed - snapshot| / snapshot * 100 > 5% (when snapshot > 0); flag any non-zero computed when snapshot == 0; no warning when both == 0
|
||||
- **Validates: Requirements 4.1, 4.2, 4.3, 4.4**
|
||||
|
||||
- [x] 8.5 Write property test: Recommendation Accuracy Aggregation
|
||||
- **Property 4: Recommendation Accuracy Aggregation**
|
||||
- File: `tests/test_pbt_report_sections.py`
|
||||
- Use Hypothesis with lists of trading decisions + prediction outcomes (direction_correct bool, profitable bool, excess_return_vs_spy float)
|
||||
- Assert: win_rate == count(profitable) / total, directional_accuracy == count(direction_correct) / total, avg excess return == mean(excess_return_vs_spy), all rates in [0.0, 1.0]
|
||||
- **Validates: Requirements 1.4**
|
||||
|
||||
- [x] 8.6 Write property test: Portfolio Period-Over-Period Delta Computation
|
||||
- **Property 5: Portfolio Period-Over-Period Delta Computation**
|
||||
- File: `tests/test_pbt_report_sections.py`
|
||||
- Use Hypothesis with two portfolio snapshots (non-negative portfolio_value, active_pool, reserve_pool, finite cumulative_return)
|
||||
- Assert: deltas == (current - previous) for each field; when no previous snapshot, deltas == 0
|
||||
- **Validates: Requirements 1.3**
|
||||
|
||||
- [x] 8.7 Write unit tests for section builders
|
||||
- File: `tests/test_report_sections.py`
|
||||
- Test each section builder with known inputs and expected outputs
|
||||
- Test edge cases: empty data (zero-activity), single position, no portfolio snapshot
|
||||
- _Requirements: 3.1, 3.2, 3.3, 3.4, 3.5_
|
||||
|
||||
- [x] 8.8 Write unit tests for report validator
|
||||
- File: `tests/test_report_validator.py`
|
||||
- Test specific discrepancy scenarios: exactly 5% (no warning), 5.1% (warning), snapshot=0 computed≠0, both=0, NULL snapshot
|
||||
- _Requirements: 4.1, 4.2, 4.3, 4.4_
|
||||
|
||||
- [x] 8.9 Write unit tests for AI summarizer
|
||||
- File: `tests/test_report_summarizer.py`
|
||||
- Test deterministic fallback summary generation
|
||||
- Test chunk_data edge cases: empty input, single character, exactly at limit, one char over limit
|
||||
- _Requirements: 2.2, 2.6_
|
||||
|
||||
- [x] 8.10 Write unit tests for report generator
|
||||
- File: `tests/test_report_generator.py`
|
||||
- Test orchestration with mocked dependencies (collector, sections, validator, summarizer)
|
||||
- Test zero-activity report generation
|
||||
- Test upsert behavior (regeneration of existing report)
|
||||
- _Requirements: 5.1, 5.2, 5.3_
|
||||
|
||||
- [x] 8.11 Write API integration tests
|
||||
- File: `tests/test_report_api.py`
|
||||
- Test GET /api/reports with pagination, filtering by report_type and date range
|
||||
- Test GET /api/reports/{report_id} with valid and invalid IDs
|
||||
- _Requirements: 5.4, 5.5, 5.6_
|
||||
|
||||
- [x] 8.12 Write frontend hook tests
|
||||
- File: `frontend/src/test/reports.test.ts`
|
||||
- Test useReports and useReport hooks with MSW mocks
|
||||
- Test loading and error states
|
||||
- _Requirements: 5.4, 5.5_
|
||||
|
||||
- [x] 9. Final checkpoint — Full test suite and lint
|
||||
- Ensure all tests pass, ask the user if questions arise.
|
||||
- Run `.venv/bin/ruff check services/`
|
||||
- Run `.venv/bin/python -m pytest tests/ -x --tb=short -q -k "report"`
|
||||
- Run frontend tests: `cd frontend && npx vitest --run`
|
||||
|
||||
## Notes
|
||||
|
||||
- Tasks marked with `*` are optional and can be skipped for faster MVP
|
||||
- Each task references specific requirements for traceability
|
||||
- Checkpoints ensure incremental validation after each phase
|
||||
- Property tests validate the 5 universal correctness properties from the design document
|
||||
- Unit tests validate specific examples and edge cases
|
||||
- The design document contains full interface signatures — use those as the implementation guide
|
||||
- Always run `.venv/bin/ruff check services/` before committing Python changes
|
||||
@@ -30,18 +30,26 @@
|
||||
- Ruff config: `ruff.toml` with `known-first-party = ["services"]` for consistent import sorting
|
||||
- Pre-existing test failures (not regressions): `test_extractor_prompts.py`, `test_extractor_schemas.py`, `test_filings_adapter.py`, `test_ollama_client.py`
|
||||
|
||||
## CI/CD — GitHub Actions
|
||||
- Workflow: `.github/workflows/build.yml`
|
||||
- Triggers on push to `main` and PRs
|
||||
- Jobs:
|
||||
- `lint-and-test`: ruff lint + pytest + frontend vitest (Node 24)
|
||||
- `build-services`: matrix build of all Python services → GHCR
|
||||
- `build-dashboard`: frontend/Dockerfile → GHCR (TypeScript strict mode — catches unused imports)
|
||||
- `build-superset`: docker/Dockerfile.superset → GHCR
|
||||
## CI/CD — Woodpecker CI (Gitea) → GitHub promotion
|
||||
- Woodpecker pipelines in `.woodpecker/` — triggered by push to `main` on Gitea
|
||||
- Push to Gitea: `git push gitea main`
|
||||
- Gitea remote: `http://admin:<password>@10.1.1.12:30300/admin/stonks-oracle.git`
|
||||
- Pipeline stages: lint → pytest → frontend vitest → build all service images + dashboard + superset → push to Harbor
|
||||
- Build pipelines split across `build-1.yml`, `build-2.yml`, `build-3.yml` for parallelism
|
||||
- ArgoCD watches Gitea `main` and auto-syncs beta/paper/live stages
|
||||
- **Do NOT push directly to GitHub** — GitHub is the promotion target after CI passes
|
||||
- Once Woodpecker builds and tests pass, code is promoted to GitHub (`git push origin main`)
|
||||
- CI handles all image builds and pushes — do NOT manually docker push
|
||||
- Check CI: `gh run list -L 3`
|
||||
- Re-run failed: `gh run rerun <id> --failed`
|
||||
- View failure logs: `gh run view <id> --log-failed`
|
||||
- Check Woodpecker CI status from the Gitea web UI or Woodpecker dashboard
|
||||
|
||||
### Dashboard Build (npm ci in K8s)
|
||||
- `build-3.yml` has a `npm-install-dashboard` step that runs `npm ci` in a `node:24-alpine` pod
|
||||
- K8s CoreDNS causes `EAI_AGAIN` errors for Node.js under concurrent DNS lookups
|
||||
- Fix: the step resolves `registry.npmjs.org` to IPv4 via Google DoH and pins it in `/etc/hosts`
|
||||
- `NODE_OPTIONS=--dns-result-order=ipv4first` env var is set as additional safety
|
||||
- `frontend/.dockerignore` must NOT exclude `node_modules` — the Dockerfile expects it pre-staged
|
||||
- The subsequent `build-dashboard` step uses `frontend/` as Docker context (includes `node_modules`)
|
||||
- If `npm ci` hangs: check `/etc/hosts` pinning worked, check `npm config set loglevel http` for which request is stuck
|
||||
|
||||
## Deploy
|
||||
- Full deploy/redeploy: `bash ~/sources/kube/stonks-oracle/runmefirst.sh` (from gremlin-1)
|
||||
@@ -74,7 +82,9 @@ Ingestion jobs MUST include `source_id`, `source_type`, `ticker`, `company_id`,
|
||||
## Git Conventions
|
||||
- Commit after each completed phase task
|
||||
- Commit message format: `feat:`, `fix:`, `phase N:` prefix
|
||||
- Push to `main` triggers CI
|
||||
- Always push to Gitea: `git push gitea main`
|
||||
- Do NOT push to GitHub (`origin`) directly — GitHub is the promotion target after CI passes
|
||||
- ArgoCD syncs from Gitea automatically
|
||||
|
||||
## Code Style
|
||||
- Python 3.12, type hints everywhere
|
||||
@@ -93,8 +103,40 @@ Ingestion jobs MUST include `source_id`, `source_type`, `ticker`, `company_id`,
|
||||
- The `competitor_relationships` table uses UUID company IDs — queries must join through `companies` to match by ticker
|
||||
- The dashboard Docker build uses TypeScript strict mode — unused imports that pass local diagnostics will fail in CI
|
||||
- Ingestion jobs require `source_id` from the `sources` table — don't just pass `ticker`
|
||||
- `frontend/.dockerignore` must NOT contain `node_modules` — the CI pre-installs it and the Dockerfile relies on `COPY . .` including it
|
||||
- `npm config set prefer-ip-address-family 4` does NOT exist in npm 10.x (Node 24) — don't use it
|
||||
- Woodpecker `environment:` uses map syntax (`KEY: "value"`) not list syntax (`- KEY=value`)
|
||||
- Node.js in Alpine K8s pods gets `EAI_AGAIN` from CoreDNS under load — pin hostnames in `/etc/hosts` for reliability
|
||||
- Every Helm-deployed service MUST have a corresponding image build step in `.woodpecker/build-*.yml`
|
||||
- **Bash `!` in passwords/strings**: Bash interprets `!` inside double quotes as history expansion. NEVER use double quotes around strings containing `!`. Use single quotes instead: `'St0nks0racl3!'`. For kubectl exec with psql, use: `kubectl exec ... -- psql -U postgres -c "ALTER USER x WITH PASSWORD '"'"'password!'"'"';"` (single-quote escaping trick)
|
||||
|
||||
## No Premature Simplification
|
||||
Do NOT "simplify" code on impulse. When the urge arises to simplify a section, STOP and do this instead:
|
||||
|
||||
1. **Evaluate the section**: Read the full function/module, not just the part that looks complex.
|
||||
2. **Map the dependencies**: Identify every caller, every consumer, every downstream component that depends on this code's behavior, return shape, or side effects.
|
||||
3. **Assess blast radius**: Would changing this function break other implementations? Check imports, tests, API contracts, database queries, and frontend expectations.
|
||||
4. **Respect intentional complexity**: If the code is complex because the domain is complex (financial math, multi-layer signal aggregation, Bayesian shrinkage), the complexity is load-bearing. Simplifying it will introduce bugs.
|
||||
5. **Only simplify when**: The complexity is accidental (dead code, redundant branches, copy-paste artifacts) AND you have confirmed no downstream dependencies break.
|
||||
|
||||
This codebase has interconnected layers (ingestion → extraction → aggregation → recommendation → trading → validation). A "simple" change to a scoring function can cascade through trend summaries, recommendations, snapshot capture, and outcome evaluation. Always trace the full path before refactoring.
|
||||
|
||||
## Documentation
|
||||
- Do NOT create large summary/success markdown files after each step
|
||||
- Keep notes short, concise, and organized under `docs/notes/`
|
||||
- If a note isn't useful for future reference, don't write it
|
||||
|
||||
## Documentation Maintenance on Feature Changes
|
||||
When implementing a feature or fix that introduces an impactful change, update the relevant documentation as part of the same commit or task. "Impactful" means any change that affects how someone installs, deploys, configures, operates, or understands the system. Specifically:
|
||||
|
||||
- **New database migrations**: Update `docs/architecture-data-pipeline.md` or `docs/api-reference.md` if new tables, views, or endpoints are added. Update `project-context.md` steering file with the new migration number.
|
||||
- **New API endpoints**: Update `docs/api-reference.md` with the endpoint path, method, parameters, and response shape.
|
||||
- **New services or service changes**: Update `docs/architecture-docker-compose.md` and `docs/docker-deployment.md` if a new service is added or an existing service's configuration changes.
|
||||
- **Helm chart changes**: Update `docs/helm-reference.md` if new values, services, or config options are added.
|
||||
- **New environment variables or secrets**: Update `docs/LOCAL_DEV_SETUP.md` and the project-context steering file.
|
||||
- **Install/deploy script changes**: Update `deploy-docker.sh`, `docs/docker-deployment.md`, or the relevant runme scripts if the deploy process changes.
|
||||
- **Frontend route or page additions**: Update `docs/api-reference.md` (if it covers UI routes) and ensure the nav item is documented.
|
||||
- **README.md**: Update the top-level `README.md` when a major new capability is added (new signal layer, new dashboard section, new trading feature).
|
||||
- **Steering files**: Update `.kiro/steering/project-context.md` when migration numbers advance, new services are added, or key conventions change.
|
||||
|
||||
The goal is that someone reading the docs can always understand the current state of the system without reading the source code. When in doubt, update the doc.
|
||||
|
||||
@@ -40,14 +40,25 @@ Three-layer signal aggregation engine:
|
||||
- Container registry: `registry.celestium.life/stonks-oracle`
|
||||
|
||||
## CI/CD
|
||||
- GitHub Actions workflow at `.github/workflows/build.yml`
|
||||
- Push to `main` triggers: lint → pytest → frontend vitest → build all service images + dashboard + superset → push to Harbor
|
||||
- Woodpecker CI pipelines in `.woodpecker/` — triggered by push to `main` on Gitea
|
||||
- Push to Gitea: `git push gitea main` — this is the primary push target
|
||||
- ArgoCD watches Gitea `main` and auto-syncs beta/paper/live stages
|
||||
- Pipeline stages: lint → pytest → frontend vitest → build all service images + dashboard + superset → push to Harbor
|
||||
- Images tagged as `registry.celestium.life/stonks-oracle/<service>:<sha>` and `:latest`
|
||||
- Dashboard image: `frontend/Dockerfile` (multi-stage: node:24 → nginx-unprivileged on port 8080)
|
||||
- Dashboard build: `npm-install-dashboard` step in `build-3.yml` pre-installs `node_modules`, then `build-dashboard` runs the Docker build with `node_modules` in context
|
||||
- Superset image: `docker/Dockerfile.superset` (apache/superset + trino + psycopg2)
|
||||
- Python service images: `docker/Dockerfile` with `SERVICE_CMD` build arg
|
||||
- Specialist image: built in `build-3.yml` like other Python services (`SERVICE_CMD=uvicorn services.specialist.app:app --host 0.0.0.0 --port 8000`)
|
||||
- Let CI handle image builds and pushes — do NOT manually `docker build && docker push`
|
||||
- Check CI status: `gh run list -L 3`
|
||||
- **Do NOT push directly to GitHub** — GitHub (`origin`) is the promotion target after CI builds and tests pass
|
||||
- Promotion to GitHub: `git push origin main` (only after Woodpecker CI succeeds)
|
||||
|
||||
### CI DNS Workaround (npm)
|
||||
- K8s CoreDNS causes `EAI_AGAIN` (temporary DNS failure) for Node.js/libuv under concurrent requests
|
||||
- Fix: `npm-install-dashboard` step resolves `registry.npmjs.org` via Google DoH (`dns.google/resolve`) and pins the IPv4 address in `/etc/hosts` before running `npm ci`
|
||||
- `NODE_OPTIONS=--dns-result-order=ipv4first` is set as env var (belt-and-suspenders)
|
||||
- `frontend/.dockerignore` does NOT exclude `node_modules` (it must be in the Docker build context since the Dockerfile has no `RUN npm ci`)
|
||||
|
||||
## Deployment Scripts
|
||||
- `~/sources/kube/stonks-oracle/runmefirst.sh` — full deploy: DB setup, migrations, Helm install, rolling restart (runs from gremlin-1 at 192.168.42.254 where secrets are available)
|
||||
@@ -76,15 +87,16 @@ When a full reset is needed:
|
||||
- Ollama: `ollama.ollama-service.svc.cluster.local:11434` (cluster-internal), also at `http://10.1.1.12:2701` (external), GPU: 4070 Ti Super 16GB
|
||||
|
||||
## Database Migrations
|
||||
- Located in `infra/migrations/001_*.sql` through `027_*.sql`
|
||||
- Located in `infra/migrations/001_*.sql` through `030_*.sql`
|
||||
- Applied automatically by `runmefirst.sh` in sorted order
|
||||
- Next migration number: **029**
|
||||
- Next migration number: **038**
|
||||
- Key migrations:
|
||||
- 016: Global news interpolation (global_events, macro_impact_records, exposure_profiles, trend_projections)
|
||||
- 017: Competitive intelligence (competitor_relationships, competitive_signal_records)
|
||||
- 024: Trend history time-series table
|
||||
- 026: AI agents management (ai_agents, agent_performance_log)
|
||||
- 027: Agent variants (agent_variants table for A/B testing)
|
||||
- 035: Model validation (prediction_snapshots, prediction_outcomes, signal_evidence_links, model_metric_snapshots, v_prediction_performance, v_source_performance)
|
||||
|
||||
## Key Conventions
|
||||
- All services use `services/shared/config.py` for configuration via env vars
|
||||
|
||||
+49
-31
@@ -3,6 +3,11 @@ depends_on:
|
||||
when:
|
||||
event: push
|
||||
branch: main
|
||||
clone:
|
||||
git:
|
||||
image: woodpeckerci/plugin-git
|
||||
settings:
|
||||
remote: http://10.43.73.77:3000/admin/stonks-oracle.git
|
||||
steps:
|
||||
build-scheduler:
|
||||
image: woodpeckerci/plugin-docker-buildx
|
||||
@@ -10,11 +15,17 @@ steps:
|
||||
settings:
|
||||
repo: registry.celestium.life/stonks-oracle/scheduler
|
||||
registry: registry.celestium.life
|
||||
custom_dns: 192.168.42.1
|
||||
buildx_image: registry.celestium.life/dockerhub-cache/moby/buildkit:buildx-stable-1
|
||||
add_host: registry.celestium.life:10.1.1.12
|
||||
buildx_flags: --driver-opt network=host
|
||||
buildkitd_config: "[registry.\"docker.io\"]\n mirrors = [\"registry.celestium.life/v2/dockerhub-cache\"]\n[registry.\"ghcr.io\"]\n mirrors = [\"registry.celestium.life/v2/ghcr-cache\"]\n"
|
||||
insecure: true
|
||||
buildkit_config: |
|
||||
[registry."registry.celestium.life"]
|
||||
insecure = true
|
||||
[registry."docker.io"]
|
||||
mirrors = ["registry.celestium.life/v2/dockerhub-cache"]
|
||||
[registry."ghcr.io"]
|
||||
mirrors = ["registry.celestium.life/v2/ghcr-cache"]
|
||||
http_proxy: ""
|
||||
https_proxy: ""
|
||||
no_proxy: ""
|
||||
@@ -24,11 +35,6 @@ steps:
|
||||
from_secret: harbor_username
|
||||
password:
|
||||
from_secret: harbor_password
|
||||
- registry: https://index.docker.io/v1/
|
||||
username:
|
||||
from_secret: docker_username
|
||||
password:
|
||||
from_secret: docker_password
|
||||
tags:
|
||||
- ${CI_COMMIT_SHA}
|
||||
- latest
|
||||
@@ -50,11 +56,17 @@ steps:
|
||||
settings:
|
||||
repo: registry.celestium.life/stonks-oracle/symbol-registry
|
||||
registry: registry.celestium.life
|
||||
custom_dns: 192.168.42.1
|
||||
buildx_image: registry.celestium.life/dockerhub-cache/moby/buildkit:buildx-stable-1
|
||||
add_host: registry.celestium.life:10.1.1.12
|
||||
buildx_flags: --driver-opt network=host
|
||||
buildkitd_config: "[registry.\"docker.io\"]\n mirrors = [\"registry.celestium.life/v2/dockerhub-cache\"]\n[registry.\"ghcr.io\"]\n mirrors = [\"registry.celestium.life/v2/ghcr-cache\"]\n"
|
||||
insecure: true
|
||||
buildkit_config: |
|
||||
[registry."registry.celestium.life"]
|
||||
insecure = true
|
||||
[registry."docker.io"]
|
||||
mirrors = ["registry.celestium.life/v2/dockerhub-cache"]
|
||||
[registry."ghcr.io"]
|
||||
mirrors = ["registry.celestium.life/v2/ghcr-cache"]
|
||||
http_proxy: ""
|
||||
https_proxy: ""
|
||||
no_proxy: ""
|
||||
@@ -64,17 +76,15 @@ steps:
|
||||
from_secret: harbor_username
|
||||
password:
|
||||
from_secret: harbor_password
|
||||
- registry: https://index.docker.io/v1/
|
||||
username:
|
||||
from_secret: docker_username
|
||||
password:
|
||||
from_secret: docker_password
|
||||
tags:
|
||||
- ${CI_COMMIT_SHA}
|
||||
- latest
|
||||
dockerfile: docker/Dockerfile
|
||||
no_cache: true
|
||||
context: .
|
||||
build_args: SERVICE_CMD=uvicorn services.symbol_registry.app:app --host 0.0.0.0 --port 8000
|
||||
build_args:
|
||||
- CACHE_BUST=${CI_COMMIT_SHA}
|
||||
- SERVICE_CMD=uvicorn services.symbol_registry.app:app --host 0.0.0.0 --port 8000
|
||||
backend_options:
|
||||
kubernetes:
|
||||
resources:
|
||||
@@ -91,11 +101,17 @@ steps:
|
||||
settings:
|
||||
repo: registry.celestium.life/stonks-oracle/ingestion
|
||||
registry: registry.celestium.life
|
||||
custom_dns: 192.168.42.1
|
||||
buildx_image: registry.celestium.life/dockerhub-cache/moby/buildkit:buildx-stable-1
|
||||
add_host: registry.celestium.life:10.1.1.12
|
||||
buildx_flags: --driver-opt network=host
|
||||
buildkitd_config: "[registry.\"docker.io\"]\n mirrors = [\"registry.celestium.life/v2/dockerhub-cache\"]\n[registry.\"ghcr.io\"]\n mirrors = [\"registry.celestium.life/v2/ghcr-cache\"]\n"
|
||||
insecure: true
|
||||
buildkit_config: |
|
||||
[registry."registry.celestium.life"]
|
||||
insecure = true
|
||||
[registry."docker.io"]
|
||||
mirrors = ["registry.celestium.life/v2/dockerhub-cache"]
|
||||
[registry."ghcr.io"]
|
||||
mirrors = ["registry.celestium.life/v2/ghcr-cache"]
|
||||
http_proxy: ""
|
||||
https_proxy: ""
|
||||
no_proxy: ""
|
||||
@@ -105,17 +121,15 @@ steps:
|
||||
from_secret: harbor_username
|
||||
password:
|
||||
from_secret: harbor_password
|
||||
- registry: https://index.docker.io/v1/
|
||||
username:
|
||||
from_secret: docker_username
|
||||
password:
|
||||
from_secret: docker_password
|
||||
tags:
|
||||
- ${CI_COMMIT_SHA}
|
||||
- latest
|
||||
dockerfile: docker/Dockerfile
|
||||
no_cache: true
|
||||
context: .
|
||||
build_args: SERVICE_CMD=python -m services.ingestion.worker
|
||||
build_args:
|
||||
- CACHE_BUST=${CI_COMMIT_SHA}
|
||||
- SERVICE_CMD=python -m services.ingestion.worker
|
||||
backend_options:
|
||||
kubernetes:
|
||||
resources:
|
||||
@@ -132,11 +146,17 @@ steps:
|
||||
settings:
|
||||
repo: registry.celestium.life/stonks-oracle/parser
|
||||
registry: registry.celestium.life
|
||||
custom_dns: 192.168.42.1
|
||||
buildx_image: registry.celestium.life/dockerhub-cache/moby/buildkit:buildx-stable-1
|
||||
add_host: registry.celestium.life:10.1.1.12
|
||||
buildx_flags: --driver-opt network=host
|
||||
buildkitd_config: "[registry.\"docker.io\"]\n mirrors = [\"registry.celestium.life/v2/dockerhub-cache\"]\n[registry.\"ghcr.io\"]\n mirrors = [\"registry.celestium.life/v2/ghcr-cache\"]\n"
|
||||
insecure: true
|
||||
buildkit_config: |
|
||||
[registry."registry.celestium.life"]
|
||||
insecure = true
|
||||
[registry."docker.io"]
|
||||
mirrors = ["registry.celestium.life/v2/dockerhub-cache"]
|
||||
[registry."ghcr.io"]
|
||||
mirrors = ["registry.celestium.life/v2/ghcr-cache"]
|
||||
http_proxy: ""
|
||||
https_proxy: ""
|
||||
no_proxy: ""
|
||||
@@ -146,17 +166,15 @@ steps:
|
||||
from_secret: harbor_username
|
||||
password:
|
||||
from_secret: harbor_password
|
||||
- registry: https://index.docker.io/v1/
|
||||
username:
|
||||
from_secret: docker_username
|
||||
password:
|
||||
from_secret: docker_password
|
||||
tags:
|
||||
- ${CI_COMMIT_SHA}
|
||||
- latest
|
||||
dockerfile: docker/Dockerfile
|
||||
no_cache: true
|
||||
context: .
|
||||
build_args: SERVICE_CMD=python -m services.parser.worker
|
||||
build_args:
|
||||
- CACHE_BUST=${CI_COMMIT_SHA}
|
||||
- SERVICE_CMD=python -m services.parser.worker
|
||||
backend_options:
|
||||
kubernetes:
|
||||
resources:
|
||||
|
||||
+98
-32
@@ -3,6 +3,11 @@ depends_on:
|
||||
when:
|
||||
event: push
|
||||
branch: main
|
||||
clone:
|
||||
git:
|
||||
image: woodpeckerci/plugin-git
|
||||
settings:
|
||||
remote: http://10.43.73.77:3000/admin/stonks-oracle.git
|
||||
steps:
|
||||
build-extractor:
|
||||
image: woodpeckerci/plugin-docker-buildx
|
||||
@@ -10,11 +15,17 @@ steps:
|
||||
settings:
|
||||
repo: registry.celestium.life/stonks-oracle/extractor
|
||||
registry: registry.celestium.life
|
||||
custom_dns: 192.168.42.1
|
||||
buildx_image: registry.celestium.life/dockerhub-cache/moby/buildkit:buildx-stable-1
|
||||
add_host: registry.celestium.life:10.1.1.12
|
||||
buildx_flags: --driver-opt network=host
|
||||
buildkitd_config: "[registry.\"docker.io\"]\n mirrors = [\"registry.celestium.life/v2/dockerhub-cache\"]\n[registry.\"ghcr.io\"]\n mirrors = [\"registry.celestium.life/v2/ghcr-cache\"]\n"
|
||||
insecure: true
|
||||
buildkit_config: |
|
||||
[registry."registry.celestium.life"]
|
||||
insecure = true
|
||||
[registry."docker.io"]
|
||||
mirrors = ["registry.celestium.life/v2/dockerhub-cache"]
|
||||
[registry."ghcr.io"]
|
||||
mirrors = ["registry.celestium.life/v2/ghcr-cache"]
|
||||
http_proxy: ""
|
||||
https_proxy: ""
|
||||
no_proxy: ""
|
||||
@@ -24,17 +35,15 @@ steps:
|
||||
from_secret: harbor_username
|
||||
password:
|
||||
from_secret: harbor_password
|
||||
- registry: https://index.docker.io/v1/
|
||||
username:
|
||||
from_secret: docker_username
|
||||
password:
|
||||
from_secret: docker_password
|
||||
tags:
|
||||
- ${CI_COMMIT_SHA}
|
||||
- latest
|
||||
dockerfile: docker/Dockerfile
|
||||
no_cache: true
|
||||
context: .
|
||||
build_args: SERVICE_CMD=python -m services.extractor.worker
|
||||
build_args:
|
||||
- CACHE_BUST=${CI_COMMIT_SHA}
|
||||
- SERVICE_CMD=python -m services.extractor.worker
|
||||
backend_options:
|
||||
kubernetes:
|
||||
resources:
|
||||
@@ -51,11 +60,17 @@ steps:
|
||||
settings:
|
||||
repo: registry.celestium.life/stonks-oracle/aggregation
|
||||
registry: registry.celestium.life
|
||||
custom_dns: 192.168.42.1
|
||||
buildx_image: registry.celestium.life/dockerhub-cache/moby/buildkit:buildx-stable-1
|
||||
add_host: registry.celestium.life:10.1.1.12
|
||||
buildx_flags: --driver-opt network=host
|
||||
buildkitd_config: "[registry.\"docker.io\"]\n mirrors = [\"registry.celestium.life/v2/dockerhub-cache\"]\n[registry.\"ghcr.io\"]\n mirrors = [\"registry.celestium.life/v2/ghcr-cache\"]\n"
|
||||
insecure: true
|
||||
buildkit_config: |
|
||||
[registry."registry.celestium.life"]
|
||||
insecure = true
|
||||
[registry."docker.io"]
|
||||
mirrors = ["registry.celestium.life/v2/dockerhub-cache"]
|
||||
[registry."ghcr.io"]
|
||||
mirrors = ["registry.celestium.life/v2/ghcr-cache"]
|
||||
http_proxy: ""
|
||||
https_proxy: ""
|
||||
no_proxy: ""
|
||||
@@ -65,17 +80,15 @@ steps:
|
||||
from_secret: harbor_username
|
||||
password:
|
||||
from_secret: harbor_password
|
||||
- registry: https://index.docker.io/v1/
|
||||
username:
|
||||
from_secret: docker_username
|
||||
password:
|
||||
from_secret: docker_password
|
||||
tags:
|
||||
- ${CI_COMMIT_SHA}
|
||||
- latest
|
||||
dockerfile: docker/Dockerfile
|
||||
no_cache: true
|
||||
context: .
|
||||
build_args: SERVICE_CMD=python -m services.aggregation.worker
|
||||
build_args:
|
||||
- CACHE_BUST=${CI_COMMIT_SHA}
|
||||
- SERVICE_CMD=python -m services.aggregation.worker
|
||||
backend_options:
|
||||
kubernetes:
|
||||
resources:
|
||||
@@ -92,11 +105,17 @@ steps:
|
||||
settings:
|
||||
repo: registry.celestium.life/stonks-oracle/recommendation
|
||||
registry: registry.celestium.life
|
||||
custom_dns: 192.168.42.1
|
||||
buildx_image: registry.celestium.life/dockerhub-cache/moby/buildkit:buildx-stable-1
|
||||
add_host: registry.celestium.life:10.1.1.12
|
||||
buildx_flags: --driver-opt network=host
|
||||
buildkitd_config: "[registry.\"docker.io\"]\n mirrors = [\"registry.celestium.life/v2/dockerhub-cache\"]\n[registry.\"ghcr.io\"]\n mirrors = [\"registry.celestium.life/v2/ghcr-cache\"]\n"
|
||||
insecure: true
|
||||
buildkit_config: |
|
||||
[registry."registry.celestium.life"]
|
||||
insecure = true
|
||||
[registry."docker.io"]
|
||||
mirrors = ["registry.celestium.life/v2/dockerhub-cache"]
|
||||
[registry."ghcr.io"]
|
||||
mirrors = ["registry.celestium.life/v2/ghcr-cache"]
|
||||
http_proxy: ""
|
||||
https_proxy: ""
|
||||
no_proxy: ""
|
||||
@@ -106,17 +125,60 @@ steps:
|
||||
from_secret: harbor_username
|
||||
password:
|
||||
from_secret: harbor_password
|
||||
- registry: https://index.docker.io/v1/
|
||||
username:
|
||||
from_secret: docker_username
|
||||
password:
|
||||
from_secret: docker_password
|
||||
tags:
|
||||
- ${CI_COMMIT_SHA}
|
||||
- latest
|
||||
dockerfile: docker/Dockerfile
|
||||
no_cache: true
|
||||
context: .
|
||||
build_args: SERVICE_CMD=python -m services.recommendation.worker
|
||||
build_args:
|
||||
- CACHE_BUST=${CI_COMMIT_SHA}
|
||||
- SERVICE_CMD=python -m services.recommendation.worker
|
||||
backend_options:
|
||||
kubernetes:
|
||||
resources:
|
||||
requests:
|
||||
memory: 1Gi
|
||||
cpu: 1000m
|
||||
limits:
|
||||
memory: 2Gi
|
||||
cpu: 4000m
|
||||
depends_on: []
|
||||
build-signal-engine:
|
||||
image: woodpeckerci/plugin-docker-buildx
|
||||
privileged: true
|
||||
settings:
|
||||
repo: registry.celestium.life/stonks-oracle/signal-engine
|
||||
registry: registry.celestium.life
|
||||
buildx_image: registry.celestium.life/dockerhub-cache/moby/buildkit:buildx-stable-1
|
||||
add_host: registry.celestium.life:10.1.1.12
|
||||
buildx_flags: --driver-opt network=host
|
||||
insecure: true
|
||||
buildkit_config: |
|
||||
[registry."registry.celestium.life"]
|
||||
insecure = true
|
||||
[registry."docker.io"]
|
||||
mirrors = ["registry.celestium.life/v2/dockerhub-cache"]
|
||||
[registry."ghcr.io"]
|
||||
mirrors = ["registry.celestium.life/v2/ghcr-cache"]
|
||||
http_proxy: ""
|
||||
https_proxy: ""
|
||||
no_proxy: ""
|
||||
logins:
|
||||
- registry: https://registry.celestium.life
|
||||
username:
|
||||
from_secret: harbor_username
|
||||
password:
|
||||
from_secret: harbor_password
|
||||
tags:
|
||||
- ${CI_COMMIT_SHA}
|
||||
- latest
|
||||
dockerfile: docker/Dockerfile
|
||||
no_cache: true
|
||||
context: .
|
||||
build_args:
|
||||
- CACHE_BUST=${CI_COMMIT_SHA}
|
||||
- SERVICE_CMD=python -m services.signal_engine.main
|
||||
backend_options:
|
||||
kubernetes:
|
||||
resources:
|
||||
@@ -133,11 +195,17 @@ steps:
|
||||
settings:
|
||||
repo: registry.celestium.life/stonks-oracle/risk
|
||||
registry: registry.celestium.life
|
||||
custom_dns: 192.168.42.1
|
||||
buildx_image: registry.celestium.life/dockerhub-cache/moby/buildkit:buildx-stable-1
|
||||
add_host: registry.celestium.life:10.1.1.12
|
||||
buildx_flags: --driver-opt network=host
|
||||
buildkitd_config: "[registry.\"docker.io\"]\n mirrors = [\"registry.celestium.life/v2/dockerhub-cache\"]\n[registry.\"ghcr.io\"]\n mirrors = [\"registry.celestium.life/v2/ghcr-cache\"]\n"
|
||||
insecure: true
|
||||
buildkit_config: |
|
||||
[registry."registry.celestium.life"]
|
||||
insecure = true
|
||||
[registry."docker.io"]
|
||||
mirrors = ["registry.celestium.life/v2/dockerhub-cache"]
|
||||
[registry."ghcr.io"]
|
||||
mirrors = ["registry.celestium.life/v2/ghcr-cache"]
|
||||
http_proxy: ""
|
||||
https_proxy: ""
|
||||
no_proxy: ""
|
||||
@@ -147,17 +215,15 @@ steps:
|
||||
from_secret: harbor_username
|
||||
password:
|
||||
from_secret: harbor_password
|
||||
- registry: https://index.docker.io/v1/
|
||||
username:
|
||||
from_secret: docker_username
|
||||
password:
|
||||
from_secret: docker_password
|
||||
tags:
|
||||
- ${CI_COMMIT_SHA}
|
||||
- latest
|
||||
dockerfile: docker/Dockerfile
|
||||
no_cache: true
|
||||
context: .
|
||||
build_args: SERVICE_CMD=uvicorn services.risk.app:app --host 0.0.0.0 --port 8000
|
||||
build_args:
|
||||
- CACHE_BUST=${CI_COMMIT_SHA}
|
||||
- SERVICE_CMD=uvicorn services.risk.app:app --host 0.0.0.0 --port 8000
|
||||
backend_options:
|
||||
kubernetes:
|
||||
resources:
|
||||
|
||||
+141
-47
@@ -3,6 +3,11 @@ depends_on:
|
||||
when:
|
||||
event: push
|
||||
branch: main
|
||||
clone:
|
||||
git:
|
||||
image: woodpeckerci/plugin-git
|
||||
settings:
|
||||
remote: http://10.43.73.77:3000/admin/stonks-oracle.git
|
||||
steps:
|
||||
build-broker-adapter:
|
||||
image: woodpeckerci/plugin-docker-buildx
|
||||
@@ -10,11 +15,17 @@ steps:
|
||||
settings:
|
||||
repo: registry.celestium.life/stonks-oracle/broker-adapter
|
||||
registry: registry.celestium.life
|
||||
custom_dns: 192.168.42.1
|
||||
buildx_image: registry.celestium.life/dockerhub-cache/moby/buildkit:buildx-stable-1
|
||||
add_host: registry.celestium.life:10.1.1.12
|
||||
buildx_flags: --driver-opt network=host
|
||||
buildkitd_config: "[registry.\"docker.io\"]\n mirrors = [\"registry.celestium.life/v2/dockerhub-cache\"]\n[registry.\"ghcr.io\"]\n mirrors = [\"registry.celestium.life/v2/ghcr-cache\"]\n"
|
||||
insecure: true
|
||||
buildkit_config: |
|
||||
[registry."registry.celestium.life"]
|
||||
insecure = true
|
||||
[registry."docker.io"]
|
||||
mirrors = ["registry.celestium.life/v2/dockerhub-cache"]
|
||||
[registry."ghcr.io"]
|
||||
mirrors = ["registry.celestium.life/v2/ghcr-cache"]
|
||||
http_proxy: ""
|
||||
https_proxy: ""
|
||||
no_proxy: ""
|
||||
@@ -24,17 +35,15 @@ steps:
|
||||
from_secret: harbor_username
|
||||
password:
|
||||
from_secret: harbor_password
|
||||
- registry: https://index.docker.io/v1/
|
||||
username:
|
||||
from_secret: docker_username
|
||||
password:
|
||||
from_secret: docker_password
|
||||
tags:
|
||||
- ${CI_COMMIT_SHA}
|
||||
- latest
|
||||
dockerfile: docker/Dockerfile
|
||||
no_cache: true
|
||||
context: .
|
||||
build_args: SERVICE_CMD=python -m services.adapters.broker_adapter
|
||||
build_args:
|
||||
- CACHE_BUST=${CI_COMMIT_SHA}
|
||||
- SERVICE_CMD=python -m services.adapters.broker_adapter
|
||||
backend_options:
|
||||
kubernetes:
|
||||
resources:
|
||||
@@ -51,11 +60,17 @@ steps:
|
||||
settings:
|
||||
repo: registry.celestium.life/stonks-oracle/lake-publisher
|
||||
registry: registry.celestium.life
|
||||
custom_dns: 192.168.42.1
|
||||
buildx_image: registry.celestium.life/dockerhub-cache/moby/buildkit:buildx-stable-1
|
||||
add_host: registry.celestium.life:10.1.1.12
|
||||
buildx_flags: --driver-opt network=host
|
||||
buildkitd_config: "[registry.\"docker.io\"]\n mirrors = [\"registry.celestium.life/v2/dockerhub-cache\"]\n[registry.\"ghcr.io\"]\n mirrors = [\"registry.celestium.life/v2/ghcr-cache\"]\n"
|
||||
insecure: true
|
||||
buildkit_config: |
|
||||
[registry."registry.celestium.life"]
|
||||
insecure = true
|
||||
[registry."docker.io"]
|
||||
mirrors = ["registry.celestium.life/v2/dockerhub-cache"]
|
||||
[registry."ghcr.io"]
|
||||
mirrors = ["registry.celestium.life/v2/ghcr-cache"]
|
||||
http_proxy: ""
|
||||
https_proxy: ""
|
||||
no_proxy: ""
|
||||
@@ -65,17 +80,15 @@ steps:
|
||||
from_secret: harbor_username
|
||||
password:
|
||||
from_secret: harbor_password
|
||||
- registry: https://index.docker.io/v1/
|
||||
username:
|
||||
from_secret: docker_username
|
||||
password:
|
||||
from_secret: docker_password
|
||||
tags:
|
||||
- ${CI_COMMIT_SHA}
|
||||
- latest
|
||||
dockerfile: docker/Dockerfile
|
||||
no_cache: true
|
||||
context: .
|
||||
build_args: SERVICE_CMD=python -m services.lake_publisher.worker
|
||||
build_args:
|
||||
- CACHE_BUST=${CI_COMMIT_SHA}
|
||||
- SERVICE_CMD=python -m services.lake_publisher.worker
|
||||
backend_options:
|
||||
kubernetes:
|
||||
resources:
|
||||
@@ -92,11 +105,17 @@ steps:
|
||||
settings:
|
||||
repo: registry.celestium.life/stonks-oracle/query-api
|
||||
registry: registry.celestium.life
|
||||
custom_dns: 192.168.42.1
|
||||
buildx_image: registry.celestium.life/dockerhub-cache/moby/buildkit:buildx-stable-1
|
||||
add_host: registry.celestium.life:10.1.1.12
|
||||
buildx_flags: --driver-opt network=host
|
||||
buildkitd_config: "[registry.\"docker.io\"]\n mirrors = [\"registry.celestium.life/v2/dockerhub-cache\"]\n[registry.\"ghcr.io\"]\n mirrors = [\"registry.celestium.life/v2/ghcr-cache\"]\n"
|
||||
insecure: true
|
||||
buildkit_config: |
|
||||
[registry."registry.celestium.life"]
|
||||
insecure = true
|
||||
[registry."docker.io"]
|
||||
mirrors = ["registry.celestium.life/v2/dockerhub-cache"]
|
||||
[registry."ghcr.io"]
|
||||
mirrors = ["registry.celestium.life/v2/ghcr-cache"]
|
||||
http_proxy: ""
|
||||
https_proxy: ""
|
||||
no_proxy: ""
|
||||
@@ -106,17 +125,15 @@ steps:
|
||||
from_secret: harbor_username
|
||||
password:
|
||||
from_secret: harbor_password
|
||||
- registry: https://index.docker.io/v1/
|
||||
username:
|
||||
from_secret: docker_username
|
||||
password:
|
||||
from_secret: docker_password
|
||||
tags:
|
||||
- ${CI_COMMIT_SHA}
|
||||
- latest
|
||||
dockerfile: docker/Dockerfile
|
||||
no_cache: true
|
||||
context: .
|
||||
build_args: SERVICE_CMD=uvicorn services.api.app:app --host 0.0.0.0 --port 8000
|
||||
build_args:
|
||||
- CACHE_BUST=${CI_COMMIT_SHA}
|
||||
- SERVICE_CMD=uvicorn services.api.app:app --host 0.0.0.0 --port 8000
|
||||
backend_options:
|
||||
kubernetes:
|
||||
resources:
|
||||
@@ -133,11 +150,17 @@ steps:
|
||||
settings:
|
||||
repo: registry.celestium.life/stonks-oracle/trading-engine
|
||||
registry: registry.celestium.life
|
||||
custom_dns: 192.168.42.1
|
||||
buildx_image: registry.celestium.life/dockerhub-cache/moby/buildkit:buildx-stable-1
|
||||
add_host: registry.celestium.life:10.1.1.12
|
||||
buildx_flags: --driver-opt network=host
|
||||
buildkitd_config: "[registry.\"docker.io\"]\n mirrors = [\"registry.celestium.life/v2/dockerhub-cache\"]\n[registry.\"ghcr.io\"]\n mirrors = [\"registry.celestium.life/v2/ghcr-cache\"]\n"
|
||||
insecure: true
|
||||
buildkit_config: |
|
||||
[registry."registry.celestium.life"]
|
||||
insecure = true
|
||||
[registry."docker.io"]
|
||||
mirrors = ["registry.celestium.life/v2/dockerhub-cache"]
|
||||
[registry."ghcr.io"]
|
||||
mirrors = ["registry.celestium.life/v2/ghcr-cache"]
|
||||
http_proxy: ""
|
||||
https_proxy: ""
|
||||
no_proxy: ""
|
||||
@@ -147,17 +170,85 @@ steps:
|
||||
from_secret: harbor_username
|
||||
password:
|
||||
from_secret: harbor_password
|
||||
- registry: https://index.docker.io/v1/
|
||||
username:
|
||||
from_secret: docker_username
|
||||
password:
|
||||
from_secret: docker_password
|
||||
tags:
|
||||
- ${CI_COMMIT_SHA}
|
||||
- latest
|
||||
dockerfile: docker/Dockerfile
|
||||
no_cache: true
|
||||
context: .
|
||||
build_args: SERVICE_CMD=uvicorn services.trading.app:app --host 0.0.0.0 --port 8000
|
||||
build_args:
|
||||
- CACHE_BUST=${CI_COMMIT_SHA}
|
||||
- SERVICE_CMD=uvicorn services.trading.app:app --host 0.0.0.0 --port 8000
|
||||
backend_options:
|
||||
kubernetes:
|
||||
resources:
|
||||
requests:
|
||||
memory: 1Gi
|
||||
cpu: 1000m
|
||||
limits:
|
||||
memory: 2Gi
|
||||
cpu: 4000m
|
||||
depends_on: []
|
||||
build-specialist:
|
||||
image: woodpeckerci/plugin-docker-buildx
|
||||
privileged: true
|
||||
settings:
|
||||
repo: registry.celestium.life/stonks-oracle/specialist
|
||||
registry: registry.celestium.life
|
||||
buildx_image: registry.celestium.life/dockerhub-cache/moby/buildkit:buildx-stable-1
|
||||
add_host: registry.celestium.life:10.1.1.12
|
||||
buildx_flags: --driver-opt network=host
|
||||
insecure: true
|
||||
buildkit_config: |
|
||||
[registry."registry.celestium.life"]
|
||||
insecure = true
|
||||
[registry."docker.io"]
|
||||
mirrors = ["registry.celestium.life/v2/dockerhub-cache"]
|
||||
[registry."ghcr.io"]
|
||||
mirrors = ["registry.celestium.life/v2/ghcr-cache"]
|
||||
http_proxy: ""
|
||||
https_proxy: ""
|
||||
no_proxy: ""
|
||||
logins:
|
||||
- registry: https://registry.celestium.life
|
||||
username:
|
||||
from_secret: harbor_username
|
||||
password:
|
||||
from_secret: harbor_password
|
||||
tags:
|
||||
- ${CI_COMMIT_SHA}
|
||||
- latest
|
||||
dockerfile: docker/Dockerfile
|
||||
no_cache: true
|
||||
context: .
|
||||
build_args:
|
||||
- CACHE_BUST=${CI_COMMIT_SHA}
|
||||
- SERVICE_CMD=uvicorn services.specialist.app:app --host 0.0.0.0 --port 8000
|
||||
backend_options:
|
||||
kubernetes:
|
||||
resources:
|
||||
requests:
|
||||
memory: 1Gi
|
||||
cpu: 1000m
|
||||
limits:
|
||||
memory: 2Gi
|
||||
cpu: 4000m
|
||||
depends_on: []
|
||||
npm-install-dashboard:
|
||||
image: registry.celestium.life/dockerhub-cache/library/node:24-alpine
|
||||
environment:
|
||||
NODE_OPTIONS: "--dns-result-order=ipv4first"
|
||||
commands:
|
||||
- echo "=== Pinning registry.npmjs.org to IPv4 in /etc/hosts ==="
|
||||
- REGISTRY_IP=$(wget -4 -q -O- https://dns.google/resolve?name=registry.npmjs.org\&type=A 2>/dev/null | sed -n 's/.*"data":"\([0-9.]*\)".*/\1/p' | head -1)
|
||||
- echo "Resolved registry.npmjs.org to $REGISTRY_IP"
|
||||
- if [ -n "$REGISTRY_IP" ]; then echo "$REGISTRY_IP registry.npmjs.org" >> /etc/hosts; else echo "104.16.1.35 registry.npmjs.org" >> /etc/hosts; fi
|
||||
- cat /etc/hosts
|
||||
- echo "=== npm/node versions ==="
|
||||
- node --version
|
||||
- npm --version
|
||||
- echo "=== Starting npm ci ==="
|
||||
- cd frontend && npm ci
|
||||
backend_options:
|
||||
kubernetes:
|
||||
resources:
|
||||
@@ -174,11 +265,17 @@ steps:
|
||||
settings:
|
||||
repo: registry.celestium.life/stonks-oracle/dashboard
|
||||
registry: registry.celestium.life
|
||||
custom_dns: 192.168.42.1
|
||||
buildx_image: registry.celestium.life/dockerhub-cache/moby/buildkit:buildx-stable-1
|
||||
add_host: registry.celestium.life:10.1.1.12
|
||||
buildx_flags: --driver-opt network=host
|
||||
buildkitd_config: "[registry.\"docker.io\"]\n mirrors = [\"registry.celestium.life/v2/dockerhub-cache\"]\n[registry.\"ghcr.io\"]\n mirrors = [\"registry.celestium.life/v2/ghcr-cache\"]\n"
|
||||
insecure: true
|
||||
buildkit_config: |
|
||||
[registry."registry.celestium.life"]
|
||||
insecure = true
|
||||
[registry."docker.io"]
|
||||
mirrors = ["registry.celestium.life/v2/dockerhub-cache"]
|
||||
[registry."ghcr.io"]
|
||||
mirrors = ["registry.celestium.life/v2/ghcr-cache"]
|
||||
http_proxy: ""
|
||||
https_proxy: ""
|
||||
no_proxy: ""
|
||||
@@ -188,11 +285,6 @@ steps:
|
||||
from_secret: harbor_username
|
||||
password:
|
||||
from_secret: harbor_password
|
||||
- registry: https://index.docker.io/v1/
|
||||
username:
|
||||
from_secret: docker_username
|
||||
password:
|
||||
from_secret: docker_password
|
||||
tags:
|
||||
- ${CI_COMMIT_SHA}
|
||||
- latest
|
||||
@@ -208,18 +300,25 @@ steps:
|
||||
limits:
|
||||
memory: 2Gi
|
||||
cpu: 4000m
|
||||
depends_on: []
|
||||
depends_on:
|
||||
- npm-install-dashboard
|
||||
build-superset:
|
||||
image: woodpeckerci/plugin-docker-buildx
|
||||
privileged: true
|
||||
settings:
|
||||
repo: registry.celestium.life/stonks-oracle/superset
|
||||
registry: registry.celestium.life
|
||||
custom_dns: 192.168.42.1
|
||||
buildx_image: registry.celestium.life/dockerhub-cache/moby/buildkit:buildx-stable-1
|
||||
add_host: registry.celestium.life:10.1.1.12
|
||||
buildx_flags: --driver-opt network=host
|
||||
buildkitd_config: "[registry.\"docker.io\"]\n mirrors = [\"registry.celestium.life/v2/dockerhub-cache\"]\n[registry.\"ghcr.io\"]\n mirrors = [\"registry.celestium.life/v2/ghcr-cache\"]\n"
|
||||
insecure: true
|
||||
buildkit_config: |
|
||||
[registry."registry.celestium.life"]
|
||||
insecure = true
|
||||
[registry."docker.io"]
|
||||
mirrors = ["registry.celestium.life/v2/dockerhub-cache"]
|
||||
[registry."ghcr.io"]
|
||||
mirrors = ["registry.celestium.life/v2/ghcr-cache"]
|
||||
http_proxy: ""
|
||||
https_proxy: ""
|
||||
no_proxy: ""
|
||||
@@ -229,11 +328,6 @@ steps:
|
||||
from_secret: harbor_username
|
||||
password:
|
||||
from_secret: harbor_password
|
||||
- registry: https://index.docker.io/v1/
|
||||
username:
|
||||
from_secret: docker_username
|
||||
password:
|
||||
from_secret: docker_password
|
||||
tags:
|
||||
- ${CI_COMMIT_SHA}
|
||||
- latest
|
||||
|
||||
@@ -8,6 +8,12 @@ when:
|
||||
event: push
|
||||
branch: main
|
||||
|
||||
clone:
|
||||
git:
|
||||
image: woodpeckerci/plugin-git
|
||||
settings:
|
||||
remote: http://10.43.73.77:3000/admin/stonks-oracle.git
|
||||
|
||||
steps:
|
||||
integration-test:
|
||||
image: registry.celestium.life/dockerhub-cache/alpine/k8s:1.30.2
|
||||
|
||||
@@ -2,6 +2,11 @@ when:
|
||||
event:
|
||||
- push
|
||||
- pull_request
|
||||
clone:
|
||||
git:
|
||||
image: woodpeckerci/plugin-git
|
||||
settings:
|
||||
remote: http://10.43.73.77:3000/admin/stonks-oracle.git
|
||||
steps:
|
||||
lint-python:
|
||||
image: registry.celestium.life/dockerhub-cache/library/python:3.12-slim
|
||||
|
||||
@@ -8,6 +8,21 @@ Licensed under the [Business Source License 1.1](LICENSE). Production use requir
|
||||
|
||||
AI-powered market intelligence and autonomous paper-trading platform. Ingests market data, company news, and regulatory filings; extracts structured intelligence with local LLMs; aggregates signals across three layers (company, macro, competitive); and autonomously executes paper trades — all self-hosted on Kubernetes.
|
||||
|
||||
## Documentation
|
||||
|
||||
| Document | Description |
|
||||
|----------|-------------|
|
||||
| [Service Reference](docs/services.md) | All 13 services — purpose, configuration, queue topology, database tables |
|
||||
| [API Reference](docs/api-reference.md) | Complete endpoint reference for Query API, Symbol Registry, Trading, and Risk services |
|
||||
| [Helm Chart Reference](docs/helm-reference.md) | All Helm values: services, config, secrets, ingress, network policies, analytics stack |
|
||||
| [Docker Deployment Guide](docs/docker-deployment.md) | Docker Compose setup, environment variables, volumes, operational commands |
|
||||
| [Kubernetes Architecture](docs/architecture-kubernetes.md) | Mermaid diagram of the K8s deployment topology, namespaces, ingress, and secrets |
|
||||
| [Docker Compose Architecture](docs/architecture-docker-compose.md) | Mermaid diagram of all containers, port mappings, volumes, and dependencies |
|
||||
| [Data Pipeline Architecture](docs/architecture-data-pipeline.md) | Mermaid diagram of the end-to-end data pipeline, queue topology, and signal layers |
|
||||
| [AI Agents Guide](docs/ai-agents.md) | Built-in agents, variant management, prompt tuning, and performance monitoring |
|
||||
| [Backup & Restore Guide](docs/backup-restore.md) | Backup scripts, restore procedures, retention policies, and disaster recovery |
|
||||
| [Observability Reference](docs/observability.md) | Prometheus metrics, alerting rules, structured logging, and dead-letter queues |
|
||||
|
||||
## What It Does
|
||||
|
||||
Stonks Oracle tracks 50 companies across 10 sectors. It monitors multiple data sources, runs every article and filing through a local Ollama model to extract structured intelligence, aggregates those signals into rolling trend summaries with contradiction detection, and generates explainable trade recommendations. An autonomous trading engine then evaluates those recommendations and executes paper trades through Alpaca without manual intervention.
|
||||
@@ -16,41 +31,47 @@ Everything is auditable — raw artifacts, prompts, model outputs, decision trac
|
||||
|
||||
## Architecture
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
subgraph sources ["Data Sources"]
|
||||
polygon["Polygon.io"]
|
||||
sec["SEC EDGAR"]
|
||||
macro_src["Macro News"]
|
||||
end
|
||||
|
||||
subgraph pipeline ["Signal Processing"]
|
||||
scheduler["Scheduler"]
|
||||
ingestion["Ingestion"]
|
||||
parser["Parser"]
|
||||
extractor["Extractor"]
|
||||
aggregation["Aggregation"]
|
||||
recommendation["Recommendation"]
|
||||
end
|
||||
|
||||
subgraph trading ["Trading"]
|
||||
risk["Risk Engine"]
|
||||
engine["Trading Engine"]
|
||||
broker["Broker Adapter"]
|
||||
alpaca["Alpaca (paper)"]
|
||||
end
|
||||
|
||||
subgraph analytics ["Analytics"]
|
||||
lake["Lake Publisher"]
|
||||
trino["Trino"]
|
||||
superset["Superset"]
|
||||
dashboard["Dashboard"]
|
||||
end
|
||||
|
||||
sources --> scheduler --> ingestion --> parser --> extractor --> aggregation --> recommendation
|
||||
recommendation --> risk --> engine --> broker --> alpaca
|
||||
aggregation --> lake --> trino --> superset
|
||||
trino --> dashboard
|
||||
```
|
||||
┌──────────────────────────────────────────┐
|
||||
│ Signal Aggregation │
|
||||
│ │
|
||||
┌───────────┐ ┌──────────┐ │ ┌──────────┐ ┌────────────────┐ │
|
||||
│ Scheduler │─▶│Ingestion │─▶│ │ Parser │─▶│ Extractor │ │
|
||||
└───────────┘ └──────────┘ │ └──────────┘ └──────┬─────────┘ │
|
||||
│ │ │
|
||||
│ ┌─────────────┘ │
|
||||
│ ▼ │
|
||||
│ ┌─────────────┐ ┌────────────────┐ │
|
||||
│ │ Aggregation │───▶│ Recommendation │ │
|
||||
│ └──────┬──────┘ └───────┬────────┘ │
|
||||
│ │ │ │
|
||||
│ Macro signals Competitive │
|
||||
│ + Competitive signals │
|
||||
│ signals merged │
|
||||
└──────────────────────────────────────────┘
|
||||
│
|
||||
┌───────────────────────┘
|
||||
▼
|
||||
┌─────────────┐ ┌────────────────┐ ┌──────────────┐
|
||||
│ Risk Engine │───▶│ Trading Engine │───▶│Broker Adapter│
|
||||
└─────────────┘ └────────────────┘ └──────────────┘
|
||||
│
|
||||
┌────────────────┐ ┌──────────┘
|
||||
│ Lake Publisher │ ▼
|
||||
└───────┬────────┘ Alpaca (paper)
|
||||
│
|
||||
┌──────────────┼──────────────┐
|
||||
▼ ▼ ▼
|
||||
┌──────────┐ ┌──────────┐ ┌───────────┐
|
||||
│ Trino │ │ Superset │ │ Dashboard │
|
||||
└──────────┘ └──────────┘ └───────────┘
|
||||
```
|
||||
|
||||
For detailed architecture diagrams see:
|
||||
- [Kubernetes Deployment](docs/architecture-kubernetes.md)
|
||||
- [Docker Compose Deployment](docs/architecture-docker-compose.md)
|
||||
- [Data Pipeline](docs/architecture-data-pipeline.md)
|
||||
|
||||
Two planes:
|
||||
- **Operational** — ingestion, parsing, extraction, aggregation, recommendations, risk evaluation, autonomous trading, trade execution (PostgreSQL, Redis, MinIO)
|
||||
|
||||
Executable
+518
@@ -0,0 +1,518 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
# deploy-docker.sh — Deploy Stonks Oracle to a Docker host via SSH
|
||||
#
|
||||
# Usage: bash deploy-docker.sh [OPTIONS]
|
||||
#
|
||||
# Options:
|
||||
# --host USER@HOST SSH target (default: celes@192.168.42.254)
|
||||
# --ollama-url URL Ollama API URL (default: auto-detect or install)
|
||||
# --ollama-model MODEL Ollama model name (default: qwen3.5:9b-fast)
|
||||
# --dir PATH Remote install directory (default: ~/stonks-oracle)
|
||||
#
|
||||
# Examples:
|
||||
# bash deploy-docker.sh
|
||||
# bash deploy-docker.sh --ollama-url http://10.1.1.12:2701 --ollama-model qwen3.6
|
||||
# bash deploy-docker.sh --host user@myserver --dir /opt/stonks
|
||||
|
||||
# -------------------------------------------------------
|
||||
# Configuration (override via flags or environment)
|
||||
# -------------------------------------------------------
|
||||
REMOTE_HOST="${DEPLOY_HOST:-celes@192.168.42.254}"
|
||||
REMOTE_DIR="${DEPLOY_DIR:-/home/celes/stonks-oracle}"
|
||||
OLLAMA_URL="${DEPLOY_OLLAMA_URL:-}"
|
||||
OLLAMA_MODEL="${DEPLOY_OLLAMA_MODEL:-qwen3.5:9b-fast}"
|
||||
REPO_URL="http://admin:St0nks0racl3!@10.1.1.12:30300/admin/stonks-oracle.git"
|
||||
|
||||
# Parse command-line flags
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case $1 in
|
||||
--host) REMOTE_HOST="$2"; shift 2 ;;
|
||||
--ollama-url) OLLAMA_URL="$2"; shift 2 ;;
|
||||
--ollama-model) OLLAMA_MODEL="$2"; shift 2 ;;
|
||||
--dir) REMOTE_DIR="$2"; shift 2 ;;
|
||||
*) echo "Unknown option: $1"; exit 1 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
echo "=== Stonks Oracle Docker Deployment ==="
|
||||
echo " Target: ${REMOTE_HOST}:${REMOTE_DIR}"
|
||||
echo " Model: ${OLLAMA_MODEL}"
|
||||
echo " Ollama: Docker container (GPU-accelerated)"
|
||||
echo ""
|
||||
|
||||
# -------------------------------------------------------
|
||||
# Step 0: Ensure prerequisites (multi-distro support)
|
||||
# -------------------------------------------------------
|
||||
echo "--- Step 0: Checking prerequisites ---"
|
||||
ssh "$REMOTE_HOST" bash -s <<'REMOTE_SCRIPT'
|
||||
set -euo pipefail
|
||||
|
||||
# --- Detect OS and package manager ---
|
||||
detect_os() {
|
||||
if [ -f /etc/os-release ]; then
|
||||
. /etc/os-release
|
||||
OS_ID="${ID:-unknown}"
|
||||
OS_LIKE="${ID_LIKE:-$OS_ID}"
|
||||
elif [ -f /etc/redhat-release ]; then
|
||||
OS_ID="rhel"
|
||||
OS_LIKE="rhel"
|
||||
else
|
||||
OS_ID="unknown"
|
||||
OS_LIKE="unknown"
|
||||
fi
|
||||
|
||||
# Detect WSL
|
||||
IS_WSL=false
|
||||
if grep -qi microsoft /proc/version 2>/dev/null; then
|
||||
IS_WSL=true
|
||||
fi
|
||||
|
||||
# Determine package manager
|
||||
if command -v apt-get &>/dev/null; then
|
||||
PKG_MGR="apt"
|
||||
elif command -v dnf &>/dev/null; then
|
||||
PKG_MGR="dnf"
|
||||
elif command -v yum &>/dev/null; then
|
||||
PKG_MGR="yum"
|
||||
elif command -v pacman &>/dev/null; then
|
||||
PKG_MGR="pacman"
|
||||
elif command -v zypper &>/dev/null; then
|
||||
PKG_MGR="zypper"
|
||||
else
|
||||
PKG_MGR="unknown"
|
||||
fi
|
||||
|
||||
echo " Detected: OS=$OS_ID (like=$OS_LIKE), pkg=$PKG_MGR, WSL=$IS_WSL"
|
||||
}
|
||||
|
||||
install_pkg() {
|
||||
local pkg="$1"
|
||||
case "$PKG_MGR" in
|
||||
apt) sudo apt-get install -y "$pkg" ;;
|
||||
dnf) sudo dnf -y install "$pkg" ;;
|
||||
yum) sudo yum -y install "$pkg" ;;
|
||||
pacman) sudo pacman -S --noconfirm "$pkg" ;;
|
||||
zypper) sudo zypper install -y "$pkg" ;;
|
||||
*) echo " ERROR: Unknown package manager"; exit 1 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
update_pkg_cache() {
|
||||
case "$PKG_MGR" in
|
||||
apt) sudo apt-get update -qq ;;
|
||||
dnf|yum) ;; # dnf/yum auto-refresh
|
||||
pacman) sudo pacman -Sy ;;
|
||||
zypper) sudo zypper refresh -q ;;
|
||||
esac
|
||||
}
|
||||
|
||||
detect_os
|
||||
|
||||
# --- Git ---
|
||||
if ! command -v git &>/dev/null; then
|
||||
echo " Installing git..."
|
||||
update_pkg_cache
|
||||
install_pkg git
|
||||
echo " ✓ Git installed"
|
||||
else
|
||||
echo " ✓ Git present"
|
||||
fi
|
||||
|
||||
# --- Docker Engine ---
|
||||
if command -v docker &>/dev/null && docker info &>/dev/null; then
|
||||
echo " ✓ Docker already installed ($(docker --version | cut -d' ' -f3 | tr -d ','))"
|
||||
else
|
||||
echo " Installing Docker CE..."
|
||||
case "$PKG_MGR" in
|
||||
apt)
|
||||
# Debian/Ubuntu/WSL
|
||||
sudo apt-get update -qq
|
||||
sudo apt-get install -y ca-certificates curl gnupg
|
||||
sudo install -m 0755 -d /etc/apt/keyrings
|
||||
curl -fsSL https://download.docker.com/linux/${OS_ID}/gpg | sudo gpg --dearmor -o /etc/apt/keyrings/docker.gpg 2>/dev/null
|
||||
sudo chmod a+r /etc/apt/keyrings/docker.gpg
|
||||
echo "deb [arch=$(dpkg --print-architecture) signed-by=/etc/apt/keyrings/docker.gpg] https://download.docker.com/linux/${OS_ID} $(. /etc/os-release && echo "$VERSION_CODENAME") stable" | \
|
||||
sudo tee /etc/apt/sources.list.d/docker.list > /dev/null
|
||||
sudo apt-get update -qq
|
||||
sudo apt-get install -y docker-ce docker-ce-cli containerd.io docker-buildx-plugin docker-compose-plugin
|
||||
;;
|
||||
dnf|yum)
|
||||
# RHEL/Rocky/Fedora/CentOS
|
||||
sudo "$PKG_MGR" -y install dnf-plugins-core 2>/dev/null || true
|
||||
local repo_distro="rhel"
|
||||
if [[ "$OS_ID" == "fedora" ]]; then repo_distro="fedora"; fi
|
||||
sudo dnf config-manager --add-repo "https://download.docker.com/linux/${repo_distro}/docker-ce.repo" 2>/dev/null || \
|
||||
sudo yum-config-manager --add-repo "https://download.docker.com/linux/${repo_distro}/docker-ce.repo" 2>/dev/null
|
||||
sudo "$PKG_MGR" -y install docker-ce docker-ce-cli containerd.io docker-buildx-plugin docker-compose-plugin
|
||||
;;
|
||||
pacman)
|
||||
# Arch Linux
|
||||
sudo pacman -S --noconfirm docker docker-compose docker-buildx
|
||||
;;
|
||||
zypper)
|
||||
# openSUSE
|
||||
sudo zypper install -y docker docker-compose docker-buildx
|
||||
;;
|
||||
esac
|
||||
sudo systemctl enable --now docker 2>/dev/null || true
|
||||
sudo usermod -aG docker "$(whoami)" 2>/dev/null || true
|
||||
echo " ✓ Docker installed and started"
|
||||
fi
|
||||
|
||||
# --- Docker Compose plugin ---
|
||||
if docker compose version &>/dev/null; then
|
||||
echo " ✓ Docker Compose plugin available ($(docker compose version --short))"
|
||||
else
|
||||
echo " ERROR: docker compose plugin not found after Docker install"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# --- NVIDIA Driver (skip on WSL — uses host driver) ---
|
||||
if [ "$IS_WSL" = "true" ]; then
|
||||
echo " ✓ WSL detected — using host Windows NVIDIA driver"
|
||||
elif ! command -v nvidia-smi &>/dev/null; then
|
||||
echo " Installing NVIDIA drivers..."
|
||||
case "$PKG_MGR" in
|
||||
apt)
|
||||
sudo apt-get install -y nvidia-driver-560 2>/dev/null || \
|
||||
sudo apt-get install -y nvidia-driver 2>/dev/null || \
|
||||
echo " ⚠ NVIDIA driver install failed — install manually"
|
||||
;;
|
||||
dnf|yum)
|
||||
sudo dnf -y install epel-release 2>/dev/null || true
|
||||
sudo dnf config-manager --add-repo https://developer.download.nvidia.com/compute/cuda/repos/rhel9/x86_64/cuda-rhel9.repo 2>/dev/null || true
|
||||
sudo dnf -y module install nvidia-driver:latest-dkms 2>/dev/null || \
|
||||
echo " ⚠ NVIDIA driver install failed — install manually"
|
||||
;;
|
||||
pacman)
|
||||
sudo pacman -S --noconfirm nvidia nvidia-utils 2>/dev/null || \
|
||||
echo " ⚠ NVIDIA driver install failed — install manually"
|
||||
;;
|
||||
zypper)
|
||||
echo " ⚠ NVIDIA driver: install manually for openSUSE"
|
||||
;;
|
||||
esac
|
||||
else
|
||||
echo " ✓ NVIDIA driver present ($(nvidia-smi --query-gpu=driver_version --format=csv,noheader | head -1))"
|
||||
fi
|
||||
|
||||
# --- NVIDIA Container Toolkit ---
|
||||
if command -v nvidia-ctk &>/dev/null; then
|
||||
echo " ✓ NVIDIA Container Toolkit already installed"
|
||||
elif [ "$IS_WSL" = "true" ] && docker run --rm --gpus all nvidia/cuda:12.8.0-base-ubuntu24.04 nvidia-smi &>/dev/null 2>&1; then
|
||||
echo " ✓ WSL GPU passthrough working (no nvidia-ctk needed)"
|
||||
else
|
||||
echo " Installing NVIDIA Container Toolkit..."
|
||||
case "$PKG_MGR" in
|
||||
apt)
|
||||
curl -fsSL https://nvidia.github.io/libnvidia-container/gpgkey | sudo gpg --dearmor -o /usr/share/keyrings/nvidia-container-toolkit-keyring.gpg 2>/dev/null
|
||||
curl -s -L https://nvidia.github.io/libnvidia-container/stable/deb/nvidia-container-toolkit.list | \
|
||||
sed 's#deb https://#deb [signed-by=/usr/share/keyrings/nvidia-container-toolkit-keyring.gpg] https://#g' | \
|
||||
sudo tee /etc/apt/sources.list.d/nvidia-container-toolkit.list > /dev/null
|
||||
sudo apt-get update -qq
|
||||
sudo apt-get install -y nvidia-container-toolkit
|
||||
;;
|
||||
dnf|yum)
|
||||
curl -fsSL https://nvidia.github.io/libnvidia-container/stable/rpm/nvidia-container-toolkit.repo | \
|
||||
sudo tee /etc/yum.repos.d/nvidia-container-toolkit.repo > /dev/null
|
||||
sudo "$PKG_MGR" -y install nvidia-container-toolkit
|
||||
;;
|
||||
pacman)
|
||||
sudo pacman -S --noconfirm nvidia-container-toolkit 2>/dev/null || \
|
||||
echo " ⚠ Install nvidia-container-toolkit from AUR"
|
||||
;;
|
||||
zypper)
|
||||
echo " ⚠ NVIDIA Container Toolkit: install manually for openSUSE"
|
||||
;;
|
||||
esac
|
||||
sudo nvidia-ctk runtime configure --runtime=docker 2>/dev/null || true
|
||||
sudo systemctl restart docker 2>/dev/null || true
|
||||
echo " ✓ NVIDIA Container Toolkit installed and Docker configured"
|
||||
fi
|
||||
|
||||
# --- Verify GPU is accessible from Docker ---
|
||||
if docker run --rm --gpus all nvidia/cuda:12.8.0-base-ubuntu24.04 nvidia-smi &>/dev/null 2>&1; then
|
||||
echo " ✓ GPU passthrough verified"
|
||||
else
|
||||
echo " ⚠ GPU passthrough test failed — may need a reboot or manual NVIDIA setup"
|
||||
fi
|
||||
|
||||
# --- Firewall (open required ports if firewall is active) ---
|
||||
if command -v firewall-cmd &>/dev/null && systemctl is-active firewalld &>/dev/null; then
|
||||
echo " Configuring firewalld..."
|
||||
for port in 3000 8001 8002 8003 8004 9000 9001 11434; do
|
||||
sudo firewall-cmd --permanent --add-port="${port}/tcp" 2>/dev/null || true
|
||||
done
|
||||
sudo firewall-cmd --reload 2>/dev/null || true
|
||||
echo " ✓ Firewall ports opened"
|
||||
elif command -v ufw &>/dev/null && sudo ufw status 2>/dev/null | grep -q "active"; then
|
||||
echo " Configuring ufw..."
|
||||
for port in 3000 8001 8002 8003 8004 9000 9001 11434; do
|
||||
sudo ufw allow "${port}/tcp" 2>/dev/null || true
|
||||
done
|
||||
echo " ✓ UFW ports opened"
|
||||
fi
|
||||
REMOTE_SCRIPT
|
||||
echo ""
|
||||
|
||||
# -------------------------------------------------------
|
||||
# Step 1: Clone or update the repo on the remote host
|
||||
# -------------------------------------------------------
|
||||
echo "--- Step 1: Syncing repository ---"
|
||||
ssh "$REMOTE_HOST" bash -s -- "$REMOTE_DIR" "$REPO_URL" <<'REMOTE_SCRIPT'
|
||||
set -euo pipefail
|
||||
REMOTE_DIR="$1"
|
||||
REPO_URL="$2"
|
||||
|
||||
if [ -d "$REMOTE_DIR/.git" ]; then
|
||||
echo " Updating existing repo..."
|
||||
cd "$REMOTE_DIR"
|
||||
git fetch origin
|
||||
git reset --hard origin/main
|
||||
else
|
||||
echo " Cloning fresh..."
|
||||
git clone "$REPO_URL" "$REMOTE_DIR"
|
||||
cd "$REMOTE_DIR"
|
||||
fi
|
||||
|
||||
echo " ✓ Repo synced at $(git log --oneline -1)"
|
||||
REMOTE_SCRIPT
|
||||
echo ""
|
||||
|
||||
# -------------------------------------------------------
|
||||
# Step 2: Detect or configure Ollama
|
||||
# -------------------------------------------------------
|
||||
echo "--- Step 2: Configuring Ollama ---"
|
||||
# Always use the Docker Ollama container with GPU passthrough
|
||||
# The ollama/ollama image ships with CUDA runtime built-in
|
||||
USE_DOCKER_OLLAMA=true
|
||||
OLLAMA_URL="http://ollama:11434"
|
||||
echo " Using Docker Ollama container (GPU-accelerated via NVIDIA passthrough)"
|
||||
echo " Host-accessible at localhost:11434"
|
||||
echo ""
|
||||
|
||||
# -------------------------------------------------------
|
||||
# Step 3: Create .env and compose override
|
||||
# -------------------------------------------------------
|
||||
echo "--- Step 3: Configuring environment ---"
|
||||
ssh "$REMOTE_HOST" bash -s -- "$REMOTE_DIR" "$OLLAMA_URL" "$OLLAMA_MODEL" "$USE_DOCKER_OLLAMA" <<'REMOTE_SCRIPT'
|
||||
set -euo pipefail
|
||||
REMOTE_DIR="$1"
|
||||
OLLAMA_URL="$2"
|
||||
OLLAMA_MODEL="$3"
|
||||
USE_DOCKER_OLLAMA="$4"
|
||||
cd "$REMOTE_DIR"
|
||||
|
||||
# Read API keys from local files if they exist
|
||||
POLYGON_KEY=""
|
||||
ALPACA_KEY=""
|
||||
ALPACA_SECRET=""
|
||||
ALPACA_URL="https://paper-api.alpaca.markets"
|
||||
|
||||
[ -f polygon.io.key ] && POLYGON_KEY=$(cat polygon.io.key)
|
||||
[ -f alpaca.key ] && ALPACA_KEY=$(cat alpaca.key)
|
||||
[ -f alpaca.secret ] && ALPACA_SECRET=$(cat alpaca.secret)
|
||||
[ -f alpaca.url ] && ALPACA_URL=$(cat alpaca.url)
|
||||
|
||||
cat > .env <<EOF
|
||||
# Stonks Oracle — Docker Deployment Environment
|
||||
MARKET_DATA_API_KEY=${POLYGON_KEY}
|
||||
BROKER_API_KEY=${ALPACA_KEY}
|
||||
BROKER_API_SECRET=${ALPACA_SECRET}
|
||||
BROKER_BASE_URL=${ALPACA_URL}
|
||||
TRADING_ENABLED=true
|
||||
TRADING_RISK_TIER=moderate
|
||||
TRADING_MAX_OPEN_POSITIONS=15
|
||||
OLLAMA_MODEL=${OLLAMA_MODEL}
|
||||
MACRO_ENABLED=true
|
||||
COMPETITIVE_ENABLED=true
|
||||
EOF
|
||||
|
||||
# Create compose override based on Ollama configuration
|
||||
if [ "$USE_DOCKER_OLLAMA" = "true" ]; then
|
||||
# Using Docker Ollama — no override needed, default compose handles it
|
||||
rm -f docker-compose.override.yml
|
||||
echo " ✓ Using Docker Ollama container"
|
||||
else
|
||||
# Using external Ollama — disable the container and point services to it
|
||||
# Determine if URL is localhost (needs host-gateway) or remote
|
||||
if echo "$OLLAMA_URL" | grep -qE "localhost|127\.0\.0\.1"; then
|
||||
DOCKER_OLLAMA_URL="http://host.docker.internal:$(echo "$OLLAMA_URL" | grep -oP ':\K[0-9]+')"
|
||||
cat > docker-compose.override.yml <<EOF
|
||||
services:
|
||||
ollama:
|
||||
entrypoint: ["true"]
|
||||
restart: "no"
|
||||
ports: []
|
||||
extractor:
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
redis:
|
||||
condition: service_healthy
|
||||
environment:
|
||||
OLLAMA_BASE_URL: "${DOCKER_OLLAMA_URL}"
|
||||
extra_hosts:
|
||||
- "host.docker.internal:host-gateway"
|
||||
recommendation:
|
||||
environment:
|
||||
OLLAMA_BASE_URL: "${DOCKER_OLLAMA_URL}"
|
||||
extra_hosts:
|
||||
- "host.docker.internal:host-gateway"
|
||||
EOF
|
||||
else
|
||||
# Remote Ollama — containers can reach it directly
|
||||
cat > docker-compose.override.yml <<EOF
|
||||
services:
|
||||
ollama:
|
||||
entrypoint: ["true"]
|
||||
restart: "no"
|
||||
extractor:
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
redis:
|
||||
condition: service_healthy
|
||||
environment:
|
||||
OLLAMA_BASE_URL: "${OLLAMA_URL}"
|
||||
recommendation:
|
||||
environment:
|
||||
OLLAMA_BASE_URL: "${OLLAMA_URL}"
|
||||
EOF
|
||||
fi
|
||||
echo " ✓ Override created — services will use external Ollama at ${OLLAMA_URL}"
|
||||
fi
|
||||
|
||||
echo " ✓ .env configured (polygon=$([ -n "$POLYGON_KEY" ] && echo 'set' || echo 'empty'), alpaca=$([ -n "$ALPACA_KEY" ] && echo 'set' || echo 'empty'))"
|
||||
REMOTE_SCRIPT
|
||||
echo ""
|
||||
|
||||
# -------------------------------------------------------
|
||||
# Step 4: Build and start all services
|
||||
# -------------------------------------------------------
|
||||
echo "--- Step 4: Building and starting services ---"
|
||||
ssh "$REMOTE_HOST" bash -s -- "$REMOTE_DIR" "$USE_DOCKER_OLLAMA" <<'REMOTE_SCRIPT'
|
||||
set -euo pipefail
|
||||
REMOTE_DIR="$1"
|
||||
USE_DOCKER_OLLAMA="$2"
|
||||
cd "$REMOTE_DIR"
|
||||
|
||||
# Stop any existing deployment
|
||||
docker compose down 2>/dev/null || true
|
||||
|
||||
# Build all images
|
||||
echo " Building images (this may take a few minutes)..."
|
||||
docker compose build --quiet 2>&1 | tail -5
|
||||
|
||||
# Start infrastructure
|
||||
echo " Starting infrastructure..."
|
||||
if [ "$USE_DOCKER_OLLAMA" = "true" ]; then
|
||||
docker compose up -d postgres redis minio minio-init ollama
|
||||
else
|
||||
docker compose up -d postgres redis minio minio-init
|
||||
fi
|
||||
|
||||
# Wait for infrastructure to be healthy
|
||||
echo " Waiting for infrastructure health checks..."
|
||||
for svc in postgres redis minio; do
|
||||
for i in $(seq 1 30); do
|
||||
if docker compose ps "$svc" 2>/dev/null | grep -q healthy; then
|
||||
break
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
done
|
||||
echo " ✓ Infrastructure healthy"
|
||||
|
||||
# Start all application services
|
||||
echo " Starting application services..."
|
||||
docker compose up -d
|
||||
|
||||
echo " Waiting for services to stabilize..."
|
||||
sleep 20
|
||||
|
||||
# Show status
|
||||
echo ""
|
||||
echo " Service Status:"
|
||||
docker compose ps --format "table {{.Name}}\t{{.Status}}" 2>/dev/null | head -25 || docker compose ps
|
||||
REMOTE_SCRIPT
|
||||
echo ""
|
||||
|
||||
# -------------------------------------------------------
|
||||
# Step 5: Seed the database
|
||||
# -------------------------------------------------------
|
||||
echo "--- Step 5: Seeding database ---"
|
||||
ssh "$REMOTE_HOST" bash -s -- "$REMOTE_DIR" <<'REMOTE_SCRIPT'
|
||||
set -euo pipefail
|
||||
cd "$1"
|
||||
|
||||
# Wait for query-api to be healthy
|
||||
for i in $(seq 1 30); do
|
||||
if docker compose ps query-api 2>/dev/null | grep -q healthy; then
|
||||
break
|
||||
fi
|
||||
sleep 3
|
||||
done
|
||||
|
||||
# Run the symbol registry seed
|
||||
echo " Seeding symbol registry..."
|
||||
docker compose exec -T scheduler python -m services.symbol_registry.seed 2>/dev/null && echo " ✓ Database seeded" || echo " ⚠ Seed skipped (may already be seeded or service not ready)"
|
||||
REMOTE_SCRIPT
|
||||
echo ""
|
||||
|
||||
# -------------------------------------------------------
|
||||
# Step 6: Ensure Ollama model is available
|
||||
# -------------------------------------------------------
|
||||
echo "--- Step 6: Checking Ollama model ---"
|
||||
ssh "$REMOTE_HOST" bash -s -- "$OLLAMA_URL" "$OLLAMA_MODEL" "$USE_DOCKER_OLLAMA" "$REMOTE_DIR" <<'REMOTE_SCRIPT'
|
||||
set -euo pipefail
|
||||
OLLAMA_URL="$1"
|
||||
OLLAMA_MODEL="$2"
|
||||
USE_DOCKER_OLLAMA="$3"
|
||||
REMOTE_DIR="$4"
|
||||
|
||||
if [ "$USE_DOCKER_OLLAMA" = "true" ]; then
|
||||
# Pull via Docker container
|
||||
cd "$REMOTE_DIR"
|
||||
if docker compose exec -T ollama ollama list 2>/dev/null | grep -q "$OLLAMA_MODEL"; then
|
||||
echo " ✓ Model $OLLAMA_MODEL already available"
|
||||
else
|
||||
echo " Pulling $OLLAMA_MODEL via Docker Ollama..."
|
||||
docker compose exec -T ollama ollama pull "$OLLAMA_MODEL"
|
||||
echo " ✓ Model pulled"
|
||||
fi
|
||||
else
|
||||
# Check via API
|
||||
if curl -sf "$OLLAMA_URL/api/tags" 2>/dev/null | grep -q "$OLLAMA_MODEL"; then
|
||||
echo " ✓ Model $OLLAMA_MODEL already available at $OLLAMA_URL"
|
||||
else
|
||||
echo " Pulling $OLLAMA_MODEL via $OLLAMA_URL..."
|
||||
curl -sf "$OLLAMA_URL/api/pull" -d "{\"name\":\"$OLLAMA_MODEL\"}" | tail -1
|
||||
echo " ✓ Model pulled"
|
||||
fi
|
||||
fi
|
||||
REMOTE_SCRIPT
|
||||
echo ""
|
||||
|
||||
# -------------------------------------------------------
|
||||
# Done
|
||||
# -------------------------------------------------------
|
||||
REMOTE_IP=$(echo "$REMOTE_HOST" | cut -d@ -f2)
|
||||
echo "=== Deployment Complete ==="
|
||||
echo ""
|
||||
echo "Endpoints:"
|
||||
echo " Dashboard: http://${REMOTE_IP}:3000"
|
||||
echo " Query API: http://${REMOTE_IP}:8004"
|
||||
echo " Symbol Registry: http://${REMOTE_IP}:8001"
|
||||
echo " Trading Engine: http://${REMOTE_IP}:8002"
|
||||
echo " Risk Engine: http://${REMOTE_IP}:8003"
|
||||
echo " MinIO Console: http://${REMOTE_IP}:9001"
|
||||
echo " Superset: http://${REMOTE_IP}:8088"
|
||||
echo " Ollama: http://${REMOTE_IP}:11434"
|
||||
echo ""
|
||||
echo "Commands:"
|
||||
echo " ssh $REMOTE_HOST 'cd $REMOTE_DIR && docker compose logs -f'"
|
||||
echo " ssh $REMOTE_HOST 'cd $REMOTE_DIR && docker compose ps'"
|
||||
echo " ssh $REMOTE_HOST 'cd $REMOTE_DIR && docker compose down'"
|
||||
@@ -1,6 +1,21 @@
|
||||
version: "3.9"
|
||||
|
||||
x-app-env: &app-env
|
||||
POSTGRES_HOST: postgres
|
||||
POSTGRES_PORT: "5432"
|
||||
POSTGRES_DB: stonks
|
||||
POSTGRES_USER: stonks
|
||||
POSTGRES_PASSWORD: stonks_dev
|
||||
REDIS_HOST: redis
|
||||
REDIS_PORT: "6379"
|
||||
MINIO_ENDPOINT: minio:9000
|
||||
MINIO_ACCESS_KEY: minioadmin
|
||||
MINIO_SECRET_KEY: minioadmin
|
||||
OLLAMA_BASE_URL: http://ollama:11434
|
||||
|
||||
services:
|
||||
# ── Infrastructure ──────────────────────────────────────────────
|
||||
|
||||
postgres:
|
||||
image: postgres:16-alpine
|
||||
environment:
|
||||
@@ -67,6 +82,13 @@ services:
|
||||
- "11434:11434"
|
||||
volumes:
|
||||
- ollama_models:/root/.ollama
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
|
||||
trino:
|
||||
image: trinodb/trino:latest
|
||||
@@ -109,6 +131,295 @@ services:
|
||||
depends_on:
|
||||
- trino
|
||||
|
||||
# ── Application Services ────────────────────────────────────────
|
||||
|
||||
scheduler:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: docker/Dockerfile.scheduler
|
||||
environment:
|
||||
<<: *app-env
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
redis:
|
||||
condition: service_healthy
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pgrep -f 'python -m services.scheduler.app' || exit 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 15s
|
||||
restart: unless-stopped
|
||||
|
||||
symbol-registry:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: docker/Dockerfile
|
||||
args:
|
||||
SERVICE_CMD: "uvicorn services.symbol_registry.app:app --host 0.0.0.0 --port 8000"
|
||||
environment:
|
||||
<<: *app-env
|
||||
ports:
|
||||
- "8001:8000"
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "curl -f http://localhost:8000/health || exit 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 15s
|
||||
restart: unless-stopped
|
||||
|
||||
ingestion:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: docker/Dockerfile
|
||||
args:
|
||||
SERVICE_CMD: "python -m services.ingestion.worker"
|
||||
environment:
|
||||
<<: *app-env
|
||||
env_file:
|
||||
- .env
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
redis:
|
||||
condition: service_healthy
|
||||
minio:
|
||||
condition: service_healthy
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pgrep -f 'python -m services.ingestion.worker' || exit 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 15s
|
||||
restart: unless-stopped
|
||||
|
||||
parser:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: docker/Dockerfile
|
||||
args:
|
||||
SERVICE_CMD: "python -m services.parser.worker"
|
||||
environment:
|
||||
<<: *app-env
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
redis:
|
||||
condition: service_healthy
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pgrep -f 'python -m services.parser.worker' || exit 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 15s
|
||||
restart: unless-stopped
|
||||
|
||||
extractor:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: docker/Dockerfile
|
||||
args:
|
||||
SERVICE_CMD: "python -m services.extractor.main"
|
||||
environment:
|
||||
<<: *app-env
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
redis:
|
||||
condition: service_healthy
|
||||
ollama:
|
||||
condition: service_started
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pgrep -f 'python -m services.extractor.main' || exit 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 15s
|
||||
restart: unless-stopped
|
||||
|
||||
aggregation:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: docker/Dockerfile
|
||||
args:
|
||||
SERVICE_CMD: "python -m services.aggregation.main"
|
||||
environment:
|
||||
<<: *app-env
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
redis:
|
||||
condition: service_healthy
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pgrep -f 'python -m services.aggregation.main' || exit 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 15s
|
||||
restart: unless-stopped
|
||||
|
||||
recommendation:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: docker/Dockerfile
|
||||
args:
|
||||
SERVICE_CMD: "python -m services.recommendation.main"
|
||||
environment:
|
||||
<<: *app-env
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
redis:
|
||||
condition: service_healthy
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pgrep -f 'python -m services.recommendation.main' || exit 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 15s
|
||||
restart: unless-stopped
|
||||
|
||||
trading-engine:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: docker/Dockerfile
|
||||
args:
|
||||
SERVICE_CMD: "uvicorn services.trading.app:app --host 0.0.0.0 --port 8000"
|
||||
environment:
|
||||
<<: *app-env
|
||||
env_file:
|
||||
- .env
|
||||
ports:
|
||||
- "8002:8000"
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
redis:
|
||||
condition: service_healthy
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "curl -f http://localhost:8000/health || exit 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 15s
|
||||
restart: unless-stopped
|
||||
|
||||
risk-engine:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: docker/Dockerfile
|
||||
args:
|
||||
SERVICE_CMD: "uvicorn services.risk.app:app --host 0.0.0.0 --port 8000"
|
||||
environment:
|
||||
<<: *app-env
|
||||
ports:
|
||||
- "8003:8000"
|
||||
networks:
|
||||
default:
|
||||
aliases:
|
||||
- risk
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "curl -f http://localhost:8000/health || exit 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 15s
|
||||
restart: unless-stopped
|
||||
|
||||
broker-adapter:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: docker/Dockerfile
|
||||
args:
|
||||
SERVICE_CMD: "python -m services.adapters.broker_service"
|
||||
environment:
|
||||
<<: *app-env
|
||||
env_file:
|
||||
- .env
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
redis:
|
||||
condition: service_healthy
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pgrep -f 'python -m services.adapters.broker_service' || exit 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 15s
|
||||
restart: unless-stopped
|
||||
|
||||
lake-publisher:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: docker/Dockerfile
|
||||
args:
|
||||
SERVICE_CMD: "python -m services.lake_publisher.jobs"
|
||||
environment:
|
||||
<<: *app-env
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
minio:
|
||||
condition: service_healthy
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pgrep -f 'python -m services.lake_publisher.jobs' || exit 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 15s
|
||||
restart: unless-stopped
|
||||
|
||||
query-api:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: docker/Dockerfile
|
||||
args:
|
||||
SERVICE_CMD: "uvicorn services.api.app:app --host 0.0.0.0 --port 8000"
|
||||
environment:
|
||||
<<: *app-env
|
||||
ports:
|
||||
- "8004:8000"
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
redis:
|
||||
condition: service_healthy
|
||||
minio:
|
||||
condition: service_healthy
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "curl -f http://localhost:8000/health || exit 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 15s
|
||||
restart: unless-stopped
|
||||
|
||||
dashboard:
|
||||
build:
|
||||
context: ./frontend
|
||||
dockerfile: Dockerfile
|
||||
ports:
|
||||
- "3000:8080"
|
||||
depends_on:
|
||||
query-api:
|
||||
condition: service_healthy
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "curl -f http://localhost:8080/ || exit 1"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 10s
|
||||
restart: unless-stopped
|
||||
|
||||
volumes:
|
||||
pgdata:
|
||||
miniodata:
|
||||
|
||||
@@ -16,7 +16,9 @@ WORKDIR /app
|
||||
COPY requirements.txt .
|
||||
RUN pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
ARG CACHE_BUST
|
||||
COPY services/ /app/services/
|
||||
COPY scripts/ /app/scripts/
|
||||
COPY tests/ /app/tests/
|
||||
COPY conftest.py /app/conftest.py
|
||||
|
||||
|
||||
@@ -0,0 +1,709 @@
|
||||
# AI Agent Building Guide
|
||||
|
||||
Stonks Oracle uses three AI agents powered by local LLM inference (Ollama or vLLM). Each agent has a dedicated purpose in the pipeline, a database-backed configuration, and support for A/B testing through variants. This guide covers how each agent works, how to configure them, how to create and test variants, and how to monitor performance.
|
||||
|
||||
## Table of Contents
|
||||
|
||||
- [Built-in Agents](#built-in-agents)
|
||||
- [Document Intelligence Extractor](#1-document-intelligence-extractor)
|
||||
- [Global Event Classifier](#2-global-event-classifier)
|
||||
- [Thesis Rewriter](#3-thesis-rewriter)
|
||||
- [LLM Provider Abstraction](#llm-provider-abstraction)
|
||||
- [Database Schema](#database-schema)
|
||||
- [ai_agents Table](#ai_agents-table)
|
||||
- [agent_variants Table](#agent_variants-table)
|
||||
- [agent_performance_log Table](#agent_performance_log-table)
|
||||
- [AgentConfigResolver](#agentconfigresolver)
|
||||
- [Performance Logging and Variant Comparison](#performance-logging-and-variant-comparison)
|
||||
- [API Endpoints](#api-endpoints)
|
||||
- [Step-by-Step: Creating and Activating a Variant](#step-by-step-creating-and-activating-a-variant)
|
||||
|
||||
---
|
||||
|
||||
## Built-in Agents
|
||||
|
||||
Three agents are seeded into the `ai_agents` table on first migration (migration `026_ai_agents.sql`). They have `source = 'system'` and cannot be deleted through the API — only deactivated or edited.
|
||||
|
||||
### 1. Document Intelligence Extractor
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| **Slug** | `document-extractor` |
|
||||
| **Purpose** | Extracts structured intelligence (sentiment, catalysts, impact scores, key facts, risks) from company news, SEC filings, earnings transcripts, and press releases |
|
||||
| **Default Model** | `qwen3.5:9b-fast` (Ollama) |
|
||||
| **Supported Providers** | `ollama`, `vllm` |
|
||||
| **Prompt Version** | `document-intel-v2` |
|
||||
| **Schema Version** | `2.0.0` |
|
||||
| **Entry Point** | `services/extractor/main.py` → `services/extractor/llm_factory.py` → `services/extractor/client.py` (Ollama) or `services/extractor/vllm_client.py` (vLLM) |
|
||||
|
||||
**Input Data:**
|
||||
- Normalized document text (fetched from MinIO or passed in the Redis job payload)
|
||||
- Document type: `article`, `filing`, `transcript`, or `press_release`
|
||||
- List of tracked tickers for company identification
|
||||
- Document ID for traceability
|
||||
|
||||
**Output Schema** (`ExtractionResult` — defined in `services/extractor/schemas.py`):
|
||||
|
||||
```json
|
||||
{
|
||||
"summary": "1-3 sentence summary",
|
||||
"companies": [
|
||||
{
|
||||
"ticker": "AAPL",
|
||||
"company_name": "Apple Inc.",
|
||||
"relevance": 0.9,
|
||||
"sentiment": "positive|negative|neutral|mixed",
|
||||
"impact_score": 0.7,
|
||||
"impact_horizon": "intraday|1d|1d_7d|1d_30d|30d_90d|90d_plus",
|
||||
"catalyst_type": "earnings|product|legal|macro|supply_chain|m_and_a|rating_change|other",
|
||||
"key_facts": ["fact1", "fact2"],
|
||||
"risks": ["risk1"],
|
||||
"evidence_spans": ["verbatim quote from document"]
|
||||
}
|
||||
],
|
||||
"macro_themes": ["inflation", "ai_capex"],
|
||||
"novelty_score": 0.6,
|
||||
"confidence": 0.8,
|
||||
"extraction_warnings": []
|
||||
}
|
||||
```
|
||||
|
||||
**System Prompt:**
|
||||
|
||||
```
|
||||
You are a financial document analyst. Extract structured data as JSON.
|
||||
Return ONLY a single JSON object. No markdown fences, no explanation,
|
||||
no text before or after the JSON. Every field in the schema is required.
|
||||
Use "other" for catalyst_type if unsure. Keep evidence_spans short
|
||||
(under 20 words each). Keep key_facts to 3-5 items max.
|
||||
```
|
||||
|
||||
**User Prompt Template** (built by `build_extraction_prompt()` in `services/extractor/prompts.py`):
|
||||
- Includes document type and type-specific guidance (article, filing, transcript, press release)
|
||||
- Includes tracked ticker list with rules for company identification
|
||||
- Includes the full JSON schema field descriptions
|
||||
- Truncates documents to 8,000 characters to limit inference time
|
||||
- When an active variant has `input_token_limit > 0`, truncation uses `input_token_limit * 4` characters instead
|
||||
|
||||
---
|
||||
|
||||
### 2. Global Event Classifier
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| **Slug** | `event-classifier` |
|
||||
| **Purpose** | Classifies global/geopolitical news into structured macro events with impact type, severity, affected regions/sectors/commodities, and estimated duration |
|
||||
| **Default Model** | `qwen3.5:9b-fast` (Ollama) |
|
||||
| **Supported Providers** | `ollama`, `vllm` |
|
||||
| **Prompt Version** | `event-classification-v1` |
|
||||
| **Schema Version** | `1.0.0` |
|
||||
| **Entry Point** | `services/extractor/main.py` → `services/extractor/event_classifier.py` |
|
||||
|
||||
**Input Data:**
|
||||
- Normalized text of a macro news article (from the `stonks:queue:macro_classification` Redis queue)
|
||||
- Document ID for traceability
|
||||
|
||||
**Output Schema** (`GlobalEvent` — defined in `services/extractor/event_classifier.py`):
|
||||
|
||||
```json
|
||||
{
|
||||
"event_types": ["trade_barrier", "commodity_shock"],
|
||||
"severity": "low|moderate|high|critical",
|
||||
"affected_regions": ["US", "CN"],
|
||||
"affected_sectors": ["Energy", "Industrials"],
|
||||
"affected_commodities": ["crude_oil"],
|
||||
"summary": "1-3 sentence summary of event and market implications",
|
||||
"key_facts": ["fact1", "fact2"],
|
||||
"estimated_duration": "short_term|medium_term|long_term",
|
||||
"confidence": 0.75
|
||||
}
|
||||
```
|
||||
|
||||
Valid `event_types`: `supply_disruption`, `demand_shift`, `cost_increase`, `regulatory_pressure`, `currency_impact`, `commodity_shock`, `trade_barrier`, `geopolitical_risk`
|
||||
|
||||
Valid `severity`: `low`, `moderate`, `high`, `critical`
|
||||
|
||||
**System Prompt:**
|
||||
|
||||
```
|
||||
You classify MACRO-LEVEL global news into structured event JSON.
|
||||
Return ONLY a single JSON object. No markdown, no explanation.
|
||||
Every field is required. Keep key_facts to 3-5 items. Keep summary
|
||||
under 3 sentences.
|
||||
|
||||
CRITICAL: Only classify articles about MACRO events that affect entire
|
||||
markets, sectors, or economies. Examples: trade wars, interest rate
|
||||
changes, commodity supply disruptions, regulatory changes, geopolitical
|
||||
conflicts, natural disasters.
|
||||
|
||||
DO NOT classify as macro events: individual company earnings, lawsuits
|
||||
against a single company, single-company management changes, individual
|
||||
stock analysis, company-specific debt or bankruptcy, product launches
|
||||
by one company. For these, set severity to "low", confidence below 0.3,
|
||||
and leave affected_regions, affected_sectors, and affected_commodities
|
||||
as empty arrays.
|
||||
```
|
||||
|
||||
**User Prompt Template** (built by `build_event_classification_prompt()` in `services/extractor/event_classifier.py`):
|
||||
- Includes anti-hallucination rules (no fabrication, severity "critical" reserved for multi-country events)
|
||||
- Lists all valid enum values for each field
|
||||
- Truncates articles to 6,000 characters
|
||||
- When an active variant has `input_token_limit > 0`, truncation uses `input_token_limit * 4` characters instead
|
||||
- If a variant overrides the system prompt, the classifier ensures JSON output instructions are always appended if not already present
|
||||
|
||||
---
|
||||
|
||||
### 3. Thesis Rewriter
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| **Slug** | `thesis-rewriter` |
|
||||
| **Purpose** | Rewrites deterministic trade thesis summaries into clear, professional analyst prose. Optional layer — the system falls back to the deterministic thesis if this fails |
|
||||
| **Default Model** | `qwen3.5:9b-fast` (Ollama) |
|
||||
| **Supported Providers** | `ollama`, `vllm` |
|
||||
| **Prompt Version** | `thesis-rewrite-v1` |
|
||||
| **Schema Version** | `1.0.0` |
|
||||
| **Entry Point** | `services/recommendation/main.py` → `services/recommendation/thesis_llm.py` |
|
||||
|
||||
**Input Data:**
|
||||
- Deterministic thesis string (rule-based, built from trend data and eligibility rules)
|
||||
- `TrendSummary` context: ticker, window, direction, strength, confidence, contradiction score, dominant catalysts, material risks
|
||||
|
||||
**Output Schema:**
|
||||
- Plain text (not JSON). The model returns only the rewritten thesis as a string, under 150 words.
|
||||
- On failure or empty response, the original deterministic thesis is returned unchanged.
|
||||
- A `_strip_thinking_block()` post-processor removes `<think>` XML tags and "Thinking Process:" blocks that some models (e.g. Qwen3) emit before the actual response.
|
||||
|
||||
**System Prompt:**
|
||||
|
||||
```
|
||||
You are a concise financial analyst. You rewrite structured trade thesis
|
||||
summaries into clear, professional prose suitable for an internal
|
||||
research note.
|
||||
|
||||
STRICT RULES:
|
||||
1. Do NOT add any information that is not present in the input.
|
||||
2. Do NOT fabricate numbers, dates, company names, or analyst opinions.
|
||||
3. Keep the rewrite under 150 words.
|
||||
4. Preserve all factual claims, risk notes, and evidence counts from
|
||||
the input.
|
||||
5. Use a neutral, professional tone. Avoid hype or marketing language.
|
||||
6. Return ONLY the rewritten thesis text. No JSON, no markdown, no
|
||||
commentary.
|
||||
7. Do NOT show your thinking process. Do NOT include any reasoning
|
||||
steps. Output ONLY the final rewritten text.
|
||||
```
|
||||
|
||||
**User Prompt Template** (built by `build_thesis_rewrite_prompt()` in `services/recommendation/thesis_llm.py`):
|
||||
- Includes the deterministic thesis between delimiters
|
||||
- Includes trend context: ticker, window, direction, strength, confidence, contradiction score, top catalysts, top risks
|
||||
- Appends `/no_think` suffix to suppress reasoning mode on models that support it (e.g. Qwen3)
|
||||
- Ollama calls also set `"think": false` in the request payload
|
||||
|
||||
---
|
||||
|
||||
## LLM Provider Abstraction
|
||||
|
||||
All three agents support both **Ollama** and **vLLM** as inference providers. The provider is determined by the `model_provider` field in the agent config (or active variant).
|
||||
|
||||
**Module:** `services/extractor/llm_factory.py`
|
||||
|
||||
The `build_llm_client()` factory function routes to the correct client:
|
||||
|
||||
| `model_provider` value | Client class | API endpoint |
|
||||
|------------------------|-------------|--------------|
|
||||
| `ollama` (default), `""`, `None` | `OllamaClient` (`services/extractor/client.py`) | `{OLLAMA_BASE_URL}/api/chat` |
|
||||
| `vllm` | `VLLMClient` (`services/extractor/vllm_client.py`) | `{VLLM_BASE_URL}/v1/chat/completions` (OpenAI-compatible) |
|
||||
| Unknown value | `OllamaClient` (with warning log) | Falls back to Ollama |
|
||||
|
||||
Both clients implement the `LLMClient` protocol (`services/shared/llm_protocol.py`), providing `call_llm()` and `close()` methods.
|
||||
|
||||
**Provider switching at runtime:** When a variant changes the `model_provider`, the extractor worker detects this during its periodic config refresh (every 100 jobs) and creates a new client instance. The old client is closed gracefully. A safety guard prevents switching to Ollama if `OLLAMA_BASE_URL` is empty.
|
||||
|
||||
**vLLM health check:** At startup, if the resolved provider is `vllm`, the extractor runs a health check against the vLLM endpoint. If it fails, the worker falls back to Ollama automatically.
|
||||
|
||||
---
|
||||
|
||||
## Database Schema
|
||||
|
||||
### `ai_agents` Table
|
||||
|
||||
Defined in migration `026_ai_agents.sql`. Stores the base configuration for each agent.
|
||||
|
||||
| Column | Type | Default | Description |
|
||||
|--------|------|---------|-------------|
|
||||
| `id` | `UUID` | `gen_random_uuid()` | Primary key |
|
||||
| `name` | `VARCHAR(100)` | — | Human-readable name (unique) |
|
||||
| `slug` | `VARCHAR(100)` | — | URL-safe identifier (unique), used by `AgentConfigResolver` |
|
||||
| `purpose` | `TEXT` | `''` | Description of what the agent does |
|
||||
| `model_provider` | `VARCHAR(50)` | `'ollama'` | LLM provider (`ollama` or `vllm`) |
|
||||
| `model_name` | `VARCHAR(200)` | `'qwen3.5:9b-fast'` | Model identifier |
|
||||
| `system_prompt` | `TEXT` | `''` | System prompt sent to the model |
|
||||
| `user_prompt_template` | `TEXT` | `''` | User prompt template (optional — code-defined templates take precedence) |
|
||||
| `prompt_version` | `VARCHAR(100)` | `''` | Version tag for prompt tracking |
|
||||
| `schema_version` | `VARCHAR(50)` | `'1.0.0'` | Version of the output schema |
|
||||
| `temperature` | `FLOAT` | `0.0` | Model temperature |
|
||||
| `max_tokens` | `INTEGER` | `32768` | Maximum output tokens |
|
||||
| `timeout_seconds` | `INTEGER` | `120` | Request timeout |
|
||||
| `max_retries` | `INTEGER` | `2` | Retry count on failure |
|
||||
| `active` | `BOOLEAN` | `TRUE` | Whether the agent is enabled |
|
||||
| `source` | `VARCHAR(20)` | `'system'` | `'system'` for built-in agents, `'user'` for API-created |
|
||||
| `created_at` | `TIMESTAMPTZ` | `NOW()` | Creation timestamp |
|
||||
| `updated_at` | `TIMESTAMPTZ` | `NOW()` | Last update timestamp |
|
||||
|
||||
**Indexes:**
|
||||
- `idx_ai_agents_slug` on `slug`
|
||||
- `idx_ai_agents_active` on `active`
|
||||
|
||||
**Registration:**
|
||||
- **System-seeded**: The three built-in agents are inserted by migration 026 using `INSERT ... WHERE NOT EXISTS` — they are only created if no row with that slug exists. This means user edits to system agents are preserved across re-migrations.
|
||||
- **API-created**: Users can create custom agents via `POST /api/agents`. These get `source = 'user'` and can be deleted.
|
||||
|
||||
### `agent_variants` Table
|
||||
|
||||
Defined in migration `027_agent_variants.sql`. Stores alternative configurations for A/B testing.
|
||||
|
||||
| Column | Type | Default | Description |
|
||||
|--------|------|---------|-------------|
|
||||
| `id` | `UUID` | `gen_random_uuid()` | Primary key |
|
||||
| `agent_id` | `UUID` | — | Foreign key → `ai_agents(id)` (CASCADE delete) |
|
||||
| `variant_name` | `VARCHAR(200)` | — | Human-readable variant name |
|
||||
| `variant_slug` | `VARCHAR(200)` | — | URL-safe slug (unique per agent) |
|
||||
| `description` | `TEXT` | `''` | What this variant changes |
|
||||
| `model_provider` | `VARCHAR(50)` | `'ollama'` | LLM provider override |
|
||||
| `model_name` | `VARCHAR(200)` | — | Model override |
|
||||
| `system_prompt` | `TEXT` | `''` | System prompt override |
|
||||
| `user_prompt_template` | `TEXT` | `''` | User prompt template override |
|
||||
| `prompt_version` | `VARCHAR(100)` | `''` | Prompt version tag |
|
||||
| `temperature` | `FLOAT` | `0.0` | Temperature override |
|
||||
| `max_tokens` | `INTEGER` | `32768` | Max tokens override |
|
||||
| `context_window` | `INTEGER` | `0` | Ollama `num_ctx` override (0 = model default) |
|
||||
| `input_token_limit` | `INTEGER` | `0` | Max input tokens before truncation (0 = no limit) |
|
||||
| `token_budget` | `INTEGER` | `0` | Total tokens per hour budget (0 = unlimited) |
|
||||
| `timeout_seconds` | `INTEGER` | `120` | Timeout override |
|
||||
| `max_retries` | `INTEGER` | `2` | Retry count override |
|
||||
| `is_active` | `BOOLEAN` | `FALSE` | Whether this variant is the active override |
|
||||
| `created_at` | `TIMESTAMPTZ` | `NOW()` | Creation timestamp |
|
||||
| `updated_at` | `TIMESTAMPTZ` | `NOW()` | Last update timestamp |
|
||||
|
||||
**Indexes and Constraints:**
|
||||
- `idx_agent_variants_slug` — unique index on `(agent_id, variant_slug)` — each agent's variant slugs must be unique
|
||||
- `idx_agent_variants_active` — unique partial index on `(agent_id) WHERE is_active = TRUE` — **at most one active variant per agent** (database-enforced)
|
||||
- `idx_agent_variants_agent` — lookup by agent
|
||||
|
||||
### `agent_performance_log` Table
|
||||
|
||||
Defined in migration `026_ai_agents.sql`, extended in `027_agent_variants.sql` with `variant_id`.
|
||||
|
||||
| Column | Type | Default | Description |
|
||||
|--------|------|---------|-------------|
|
||||
| `id` | `UUID` | `gen_random_uuid()` | Primary key |
|
||||
| `agent_id` | `UUID` | — | Foreign key → `ai_agents(id)` (CASCADE delete) |
|
||||
| `variant_id` | `UUID` | `NULL` | Foreign key → `agent_variants(id)` (SET NULL on delete) |
|
||||
| `document_id` | `UUID` | `NULL` | Foreign key → `documents(id)` (SET NULL on delete) |
|
||||
| `ticker` | `VARCHAR(20)` | — | Stock ticker processed |
|
||||
| `success` | `BOOLEAN` | — | Whether the invocation succeeded |
|
||||
| `duration_ms` | `INTEGER` | `0` | Total invocation time in milliseconds |
|
||||
| `confidence` | `FLOAT` | `0.0` | Model confidence score (0.0 for thesis rewrites) |
|
||||
| `retry_count` | `INTEGER` | `0` | Number of retries before success/failure |
|
||||
| `input_tokens` | `INTEGER` | `0` | Estimated input tokens (chars / 4) |
|
||||
| `output_tokens` | `INTEGER` | `0` | Estimated output tokens (chars / 4) |
|
||||
| `error_message` | `TEXT` | `NULL` | Error description on failure |
|
||||
| `recorded_at` | `TIMESTAMPTZ` | `NOW()` | When the invocation occurred |
|
||||
|
||||
**Indexes:**
|
||||
- `idx_agent_perf_agent` on `(agent_id, recorded_at DESC)`
|
||||
- `idx_agent_perf_time` on `(recorded_at DESC)`
|
||||
- `idx_agent_perf_variant` on `(variant_id, recorded_at DESC)`
|
||||
|
||||
---
|
||||
|
||||
## AgentConfigResolver
|
||||
|
||||
**Module:** `services/shared/agent_config.py`
|
||||
|
||||
The `AgentConfigResolver` is the central mechanism for resolving runtime agent configuration. All three agent services use it instead of duplicating resolution logic.
|
||||
|
||||
### How It Works
|
||||
|
||||
1. **Lookup by slug**: The resolver queries the `ai_agents` table by slug (e.g., `"document-extractor"`), joining with `agent_variants` to find any active variant.
|
||||
|
||||
2. **COALESCE-based override**: The SQL query uses `COALESCE(variant_column, agent_column)` for every configuration field. If an active variant exists and has a non-NULL value for a field, that value is used. Otherwise, the base agent's value is used.
|
||||
|
||||
```sql
|
||||
SELECT a.id AS agent_id,
|
||||
v.id AS variant_id,
|
||||
COALESCE(v.model_provider, a.model_provider) AS model_provider,
|
||||
COALESCE(v.model_name, a.model_name) AS model_name,
|
||||
COALESCE(v.system_prompt, a.system_prompt) AS system_prompt,
|
||||
COALESCE(v.user_prompt_template, a.user_prompt_template) AS user_prompt_template,
|
||||
COALESCE(v.prompt_version, a.prompt_version) AS prompt_version,
|
||||
COALESCE(v.temperature, a.temperature) AS temperature,
|
||||
COALESCE(v.max_tokens, a.max_tokens) AS max_tokens,
|
||||
COALESCE(v.context_window, 0) AS context_window,
|
||||
COALESCE(v.input_token_limit, 0) AS input_token_limit,
|
||||
COALESCE(v.token_budget, 0) AS token_budget,
|
||||
COALESCE(v.timeout_seconds, a.timeout_seconds) AS timeout_seconds,
|
||||
COALESCE(v.max_retries, a.max_retries) AS max_retries
|
||||
FROM ai_agents a
|
||||
LEFT JOIN agent_variants v
|
||||
ON v.agent_id = a.id AND v.is_active = TRUE
|
||||
WHERE a.slug = $1
|
||||
AND a.active = TRUE
|
||||
```
|
||||
|
||||
3. **TTL cache (60 seconds)**: Resolved configurations are cached in memory using `time.monotonic()`. Cache entries expire after 60 seconds (configurable via `ttl_seconds`). This means variant swaps take effect within 60 seconds without restarting any service.
|
||||
|
||||
4. **Fallback behavior**: If the database query fails or returns no rows (agent not found or inactive), the resolver returns `None`. Callers fall back to environment-variable-based `OllamaConfig` defaults.
|
||||
|
||||
### Resolved Config Dataclass
|
||||
|
||||
```python
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ResolvedAgentConfig:
|
||||
agent_id: str
|
||||
variant_id: str | None # None if no active variant
|
||||
model_provider: str
|
||||
model_name: str
|
||||
system_prompt: str
|
||||
user_prompt_template: str
|
||||
prompt_version: str
|
||||
temperature: float
|
||||
max_tokens: int
|
||||
context_window: int # Ollama num_ctx; 0 = model default
|
||||
input_token_limit: int # Max input chars before truncation; 0 = no limit
|
||||
token_budget: int # Hourly token budget; 0 = unlimited
|
||||
timeout_seconds: int
|
||||
max_retries: int
|
||||
```
|
||||
|
||||
### Usage Pattern
|
||||
|
||||
```python
|
||||
from services.shared.agent_config import AgentConfigResolver
|
||||
|
||||
resolver = AgentConfigResolver(pool, ttl_seconds=60)
|
||||
config = await resolver.resolve("document-extractor")
|
||||
|
||||
if config is None:
|
||||
# Fall back to env-var defaults
|
||||
...
|
||||
else:
|
||||
# Use config.model_name, config.system_prompt, etc.
|
||||
...
|
||||
```
|
||||
|
||||
### Cache Invalidation
|
||||
|
||||
```python
|
||||
resolver.invalidate("document-extractor") # Clear one entry
|
||||
resolver.invalidate() # Clear all entries
|
||||
```
|
||||
|
||||
### Config Refresh in Workers
|
||||
|
||||
The extractor and recommendation workers periodically re-resolve their agent config to pick up variant swaps and model changes:
|
||||
|
||||
- **Extractor worker** (`services/extractor/main.py`): Re-resolves both `document-extractor` and `event-classifier` configs every **100 jobs**. If the resolved model or provider changes, the worker creates a new LLM client instance via `build_llm_client()` and closes the old one. A safety guard prevents switching to Ollama if `OLLAMA_BASE_URL` is empty.
|
||||
- **Recommendation worker** (`services/recommendation/main.py`): Re-resolves the `thesis-rewriter` config every **50 jobs**. If the model changes, a new `OllamaConfig` is built.
|
||||
|
||||
---
|
||||
|
||||
## Performance Logging and Variant Comparison
|
||||
|
||||
Every agent invocation is logged to `agent_performance_log` with the `agent_id` and `variant_id` (if a variant was active). This enables comparing variant effectiveness.
|
||||
|
||||
### What Gets Logged
|
||||
|
||||
- **Document extractor**: Logged in `services/extractor/main.py` after each extraction. Records success/failure, duration, confidence, retry count, token estimates.
|
||||
- **Event classifier**: Logged in `services/extractor/event_classifier.py` after each classification. Same fields.
|
||||
- **Thesis rewriter**: Logged in `services/recommendation/thesis_llm.py` after each rewrite attempt. Confidence is always 0.0 (not applicable for rewrites). `document_id` is always NULL.
|
||||
|
||||
### Querying for Variant Comparison
|
||||
|
||||
Compare two variants of the document extractor over the last 24 hours:
|
||||
|
||||
```sql
|
||||
SELECT
|
||||
v.variant_name,
|
||||
COUNT(*) AS total_invocations,
|
||||
COUNT(*) FILTER (WHERE p.success) AS successes,
|
||||
ROUND(100.0 * COUNT(*) FILTER (WHERE p.success) / COUNT(*), 1) AS success_rate_pct,
|
||||
ROUND(AVG(p.duration_ms)::numeric) AS avg_duration_ms,
|
||||
ROUND(PERCENTILE_CONT(0.95) WITHIN GROUP (ORDER BY p.duration_ms)::numeric) AS p95_duration_ms,
|
||||
ROUND(AVG(p.confidence)::numeric, 4) AS avg_confidence,
|
||||
ROUND(AVG(p.retry_count)::numeric, 2) AS avg_retries,
|
||||
SUM(p.input_tokens + p.output_tokens) AS total_tokens
|
||||
FROM agent_performance_log p
|
||||
JOIN agent_variants v ON v.id = p.variant_id
|
||||
WHERE p.agent_id = '<agent-uuid>'
|
||||
AND p.recorded_at >= NOW() - INTERVAL '24 hours'
|
||||
GROUP BY v.variant_name
|
||||
ORDER BY success_rate_pct DESC;
|
||||
```
|
||||
|
||||
Compare base agent (no variant) vs active variant:
|
||||
|
||||
```sql
|
||||
SELECT
|
||||
CASE WHEN p.variant_id IS NULL THEN 'base' ELSE v.variant_name END AS config,
|
||||
COUNT(*) AS invocations,
|
||||
ROUND(100.0 * COUNT(*) FILTER (WHERE p.success) / COUNT(*), 1) AS success_rate_pct,
|
||||
ROUND(AVG(p.duration_ms)::numeric) AS avg_duration_ms,
|
||||
ROUND(AVG(p.confidence)::numeric, 4) AS avg_confidence
|
||||
FROM agent_performance_log p
|
||||
LEFT JOIN agent_variants v ON v.id = p.variant_id
|
||||
WHERE p.agent_id = '<agent-uuid>'
|
||||
AND p.recorded_at >= NOW() - INTERVAL '48 hours'
|
||||
GROUP BY config
|
||||
ORDER BY config;
|
||||
```
|
||||
|
||||
### Token Budget Enforcement
|
||||
|
||||
Variants can set a `token_budget` (total tokens per hour). Before each invocation, the worker checks:
|
||||
|
||||
```sql
|
||||
SELECT COALESCE(SUM(input_tokens + output_tokens), 0) AS total_tokens
|
||||
FROM agent_performance_log
|
||||
WHERE variant_id = $1
|
||||
AND recorded_at >= NOW() - INTERVAL '1 hour'
|
||||
```
|
||||
|
||||
If the budget is exceeded, the invocation is skipped (extractor) or falls back to the deterministic thesis (thesis rewriter).
|
||||
|
||||
---
|
||||
|
||||
## API Endpoints
|
||||
|
||||
All agent endpoints are served by the Query API (`services/api/app.py`) under the `/api/agents` prefix.
|
||||
|
||||
### Agent CRUD
|
||||
|
||||
| Method | Path | Description |
|
||||
|--------|------|-------------|
|
||||
| `GET` | `/api/agents` | List all agents. Query param: `active_only` (bool, default `false`) |
|
||||
| `GET` | `/api/agents/{agent_id}` | Get a single agent by UUID |
|
||||
| `POST` | `/api/agents` | Create a new user-defined agent (returns 201) |
|
||||
| `PUT` | `/api/agents/{agent_id}` | Partial update an agent (system or user) |
|
||||
| `DELETE` | `/api/agents/{agent_id}` | Delete a user-created agent. Returns 403 for system agents |
|
||||
|
||||
**Create Agent Request Body:**
|
||||
|
||||
```json
|
||||
{
|
||||
"name": "My Custom Agent",
|
||||
"slug": "my-custom-agent",
|
||||
"purpose": "Custom extraction for earnings calls",
|
||||
"model_provider": "ollama",
|
||||
"model_name": "llama3.1:8b",
|
||||
"system_prompt": "You are a financial analyst...",
|
||||
"user_prompt_template": "",
|
||||
"prompt_version": "v1",
|
||||
"schema_version": "1.0.0",
|
||||
"temperature": 0.0,
|
||||
"max_tokens": 32768,
|
||||
"timeout_seconds": 120,
|
||||
"max_retries": 2
|
||||
}
|
||||
```
|
||||
|
||||
All fields except `name` have defaults. The `slug` is auto-generated from `name` if not provided. The `model_name` defaults to `llama3.1:8b` for user-created agents.
|
||||
|
||||
**Update Agent Request Body** (all fields optional):
|
||||
|
||||
```json
|
||||
{
|
||||
"model_name": "qwen3.5:14b",
|
||||
"system_prompt": "Updated prompt...",
|
||||
"temperature": 0.1,
|
||||
"active": false
|
||||
}
|
||||
```
|
||||
|
||||
### Agent Performance
|
||||
|
||||
| Method | Path | Description |
|
||||
|--------|------|-------------|
|
||||
| `GET` | `/api/agents/{agent_id}/performance` | Aggregated metrics. Query param: `hours` (int, default 24, max 720) |
|
||||
| `GET` | `/api/agents/{agent_id}/performance/history` | Hourly time-series. Query param: `hours` (int, default 24, max 720) |
|
||||
|
||||
**Performance Response:**
|
||||
|
||||
```json
|
||||
{
|
||||
"total_invocations": 1250,
|
||||
"successes": 1180,
|
||||
"failures": 70,
|
||||
"avg_duration_ms": 3400,
|
||||
"p95_duration_ms": 8200,
|
||||
"avg_confidence": 0.7234,
|
||||
"avg_retries": 0.15,
|
||||
"total_input_tokens": 5000000,
|
||||
"total_output_tokens": 1200000,
|
||||
"success_rate": 0.944
|
||||
}
|
||||
```
|
||||
|
||||
### Variant CRUD
|
||||
|
||||
| Method | Path | Description |
|
||||
|--------|------|-------------|
|
||||
| `GET` | `/api/agents/{agent_id}/variants` | List all variants for an agent |
|
||||
| `GET` | `/api/agents/{agent_id}/variants/{variant_id}` | Get a single variant |
|
||||
| `POST` | `/api/agents/{agent_id}/variants` | Create a new variant (returns 201, 409 on duplicate slug) |
|
||||
| `PUT` | `/api/agents/{agent_id}/variants/{variant_id}` | Partial update a variant |
|
||||
| `DELETE` | `/api/agents/{agent_id}/variants/{variant_id}` | Delete a variant (returns 400 if active) |
|
||||
|
||||
**Create Variant Request Body:**
|
||||
|
||||
```json
|
||||
{
|
||||
"variant_name": "Llama 3.1 8B Test",
|
||||
"variant_slug": "llama-3-1-8b-test",
|
||||
"description": "Testing llama3.1:8b as an alternative",
|
||||
"model_provider": "ollama",
|
||||
"model_name": "llama3.1:8b",
|
||||
"system_prompt": "",
|
||||
"user_prompt_template": "",
|
||||
"prompt_version": "",
|
||||
"temperature": 0.0,
|
||||
"max_tokens": 32768,
|
||||
"context_window": 0,
|
||||
"input_token_limit": 0,
|
||||
"token_budget": 0,
|
||||
"timeout_seconds": 120,
|
||||
"max_retries": 2
|
||||
}
|
||||
```
|
||||
|
||||
Required fields: `variant_name`, `model_name`. The `variant_slug` is auto-generated from `variant_name` if not provided.
|
||||
|
||||
### Clone Endpoints
|
||||
|
||||
| Method | Path | Description |
|
||||
|--------|------|-------------|
|
||||
| `POST` | `/api/agents/{agent_id}/clone` | Clone an agent's base config as a new variant |
|
||||
| `POST` | `/api/agents/{agent_id}/variants/{variant_id}/clone` | Clone an existing variant as a new variant |
|
||||
|
||||
Clone requests copy all configuration fields from the source, with optional overrides in the request body. The `variant_name` field is required. All other fields default to the source's values if not provided.
|
||||
|
||||
### Activate / Deactivate
|
||||
|
||||
| Method | Path | Description |
|
||||
|--------|------|-------------|
|
||||
| `POST` | `/api/agents/{agent_id}/variants/{variant_id}/activate` | Set a variant as active (deactivates any other active variant in a single transaction) |
|
||||
| `POST` | `/api/agents/{agent_id}/variants/deactivate` | Deactivate the currently active variant (agent falls back to base config) |
|
||||
|
||||
The activate endpoint uses a database transaction to atomically deactivate the current variant and activate the new one, ensuring exactly one active variant at all times.
|
||||
|
||||
### Per-Variant Performance
|
||||
|
||||
| Method | Path | Description |
|
||||
|--------|------|-------------|
|
||||
| `GET` | `/api/agents/{agent_id}/variants/{variant_id}/performance` | Aggregated metrics for a specific variant |
|
||||
| `GET` | `/api/agents/{agent_id}/variants/{variant_id}/performance/history` | Hourly time-series for a specific variant |
|
||||
|
||||
Both endpoints accept the same `hours` query parameter (default 24, max 720) and return the same response shape as the agent-level performance endpoints.
|
||||
|
||||
---
|
||||
|
||||
## Step-by-Step: Creating and Activating a Variant
|
||||
|
||||
This walkthrough creates a new variant of the document extractor that uses a different model and activates it for live traffic.
|
||||
|
||||
### 1. Find the Agent ID
|
||||
|
||||
```bash
|
||||
curl -s https://stonks-api.celestium.life/api/agents?active_only=true | jq '.[] | select(.slug == "document-extractor") | .id'
|
||||
```
|
||||
|
||||
Note the UUID — we'll call it `AGENT_ID`.
|
||||
|
||||
### 2. Clone the Agent as a Variant
|
||||
|
||||
```bash
|
||||
curl -s -X POST https://stonks-api.celestium.life/api/agents/$AGENT_ID/clone \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"variant_name": "Llama 3.1 8B Test",
|
||||
"description": "Testing llama3.1:8b as an alternative to qwen3.5:9b-fast",
|
||||
"model_name": "llama3.1:8b",
|
||||
"temperature": 0.1
|
||||
}' | jq .
|
||||
```
|
||||
|
||||
This creates a new variant with all fields copied from the base agent, except `model_name` and `temperature` which are overridden. The variant starts as `is_active: false`.
|
||||
|
||||
Note the variant's `id` — we'll call it `VARIANT_ID`.
|
||||
|
||||
### 3. Activate the Variant
|
||||
|
||||
```bash
|
||||
curl -s -X POST \
|
||||
https://stonks-api.celestium.life/api/agents/$AGENT_ID/variants/$VARIANT_ID/activate | jq .
|
||||
```
|
||||
|
||||
This atomically deactivates any previously active variant and activates the new one. Within 60 seconds (the TTL cache window), the extractor worker will pick up the new configuration and start using `llama3.1:8b`.
|
||||
|
||||
### 4. Monitor Performance
|
||||
|
||||
Wait for some documents to be processed, then compare:
|
||||
|
||||
```bash
|
||||
# Base agent performance (all invocations)
|
||||
curl -s "https://stonks-api.celestium.life/api/agents/$AGENT_ID/performance?hours=4" | jq .
|
||||
|
||||
# Variant-specific performance
|
||||
curl -s "https://stonks-api.celestium.life/api/agents/$AGENT_ID/variants/$VARIANT_ID/performance?hours=4" | jq .
|
||||
```
|
||||
|
||||
Check the hourly trend:
|
||||
|
||||
```bash
|
||||
curl -s "https://stonks-api.celestium.life/api/agents/$AGENT_ID/variants/$VARIANT_ID/performance/history?hours=12" | jq .
|
||||
```
|
||||
|
||||
### 5. Roll Back (Deactivate)
|
||||
|
||||
If the variant underperforms, deactivate it to revert to the base agent config:
|
||||
|
||||
```bash
|
||||
curl -s -X POST \
|
||||
https://stonks-api.celestium.life/api/agents/$AGENT_ID/variants/deactivate | jq .
|
||||
```
|
||||
|
||||
The extractor will revert to the base `qwen3.5:9b-fast` configuration within 60 seconds.
|
||||
|
||||
### 6. Iterate
|
||||
|
||||
You can update the variant's prompt or parameters without creating a new one:
|
||||
|
||||
```bash
|
||||
curl -s -X PUT \
|
||||
https://stonks-api.celestium.life/api/agents/$AGENT_ID/variants/$VARIANT_ID \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"system_prompt": "You are a financial document analyst. Extract structured data as JSON. Be extra conservative with impact scores — only assign > 0.7 for material events with concrete numbers.",
|
||||
"prompt_version": "document-intel-v2-conservative"
|
||||
}' | jq .
|
||||
```
|
||||
|
||||
Then re-activate and compare again.
|
||||
|
||||
### 7. Switch to vLLM Provider
|
||||
|
||||
To test a variant using vLLM instead of Ollama:
|
||||
|
||||
```bash
|
||||
curl -s -X POST https://stonks-api.celestium.life/api/agents/$AGENT_ID/clone \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"variant_name": "vLLM Qwen3 Test",
|
||||
"description": "Testing extraction with vLLM backend",
|
||||
"model_provider": "vllm",
|
||||
"model_name": "Qwen/Qwen3-8B"
|
||||
}' | jq .
|
||||
```
|
||||
|
||||
The extractor worker will detect the provider change during its next config refresh and build a `VLLMClient` instead of an `OllamaClient`. Ensure the `VLLM_BASE_URL` environment variable is set in the extractor deployment.
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,330 @@
|
||||
# Data Pipeline Architecture — Stonks Oracle
|
||||
|
||||
This document describes the end-to-end data pipeline from external data sources through signal processing to trade execution. The pipeline is queue-driven, with Redis lists connecting each stage and PostgreSQL/MinIO providing durable storage at every step.
|
||||
|
||||
All queue names follow the convention `stonks:queue:<name>` (see `services/shared/redis_keys.py`). Dead-letter queues mirror the pattern as `stonks:dlq:<name>`.
|
||||
|
||||
## Pipeline Overview
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
%% ── External Data Sources ─────────────────────────────────────
|
||||
subgraph sources ["External Data Sources"]
|
||||
direction LR
|
||||
polygon["Polygon.io<br/><i>News, Market Bars,<br/>Grouped Daily</i>"]
|
||||
sec["SEC EDGAR<br/><i>10-K, 10-Q Filings</i>"]
|
||||
macro_src["Macro News APIs<br/><i>Geopolitical &<br/>Economic Events</i>"]
|
||||
market_src["Market Data API<br/><i>Intraday Bars,<br/>Grouped Daily</i>"]
|
||||
end
|
||||
|
||||
%% ── Scheduler ─────────────────────────────────────────────────
|
||||
scheduler["<b>Scheduler</b><br/><i>services.scheduler.app</i><br/>Cadence polling, rate limiting,<br/>backoff, stale recovery,<br/>periodic aggregation,<br/>report scheduling"]
|
||||
|
||||
sources -.->|"API polling<br/>on cadence"| scheduler
|
||||
|
||||
%% ── Ingestion Queue ───────────────────────────────────────────
|
||||
q_ingestion[["stonks:queue:ingestion"]]
|
||||
scheduler -->|"rpush job<br/>(company, macro,<br/>global market)"| q_ingestion
|
||||
|
||||
%% ── Ingestion Worker ──────────────────────────────────────────
|
||||
ingestion["<b>Ingestion</b><br/><i>services.ingestion.worker</i><br/>Adapter dispatch, dedupe,<br/>raw artifact upload"]
|
||||
|
||||
q_ingestion -->|"lpop"| ingestion
|
||||
|
||||
%% ── Raw Storage ───────────────────────────────────────────────
|
||||
minio_raw[("MinIO<br/><i>Raw Artifacts</i><br/>JSON / HTML")]
|
||||
pg_docs[("PostgreSQL<br/><i>documents,<br/>ingestion_runs</i>")]
|
||||
redis_dedupe[("Redis<br/><i>Dedupe Markers</i><br/>stonks:dedupe:*")]
|
||||
|
||||
ingestion -->|"upload raw payload"| minio_raw
|
||||
ingestion -->|"persist metadata"| pg_docs
|
||||
ingestion -->|"set content hash"| redis_dedupe
|
||||
|
||||
%% ── Parsing Queue ─────────────────────────────────────────────
|
||||
q_parsing[["stonks:queue:parsing"]]
|
||||
ingestion -->|"rpush<br/>(news, filings,<br/>web_scrape, macro)"| q_parsing
|
||||
|
||||
%% ── Parser Worker ─────────────────────────────────────────────
|
||||
parser["<b>Parser</b><br/><i>services.parser.worker</i><br/>HTML parsing, quality scoring,<br/>company mention detection"]
|
||||
|
||||
q_parsing -->|"lpop"| parser
|
||||
|
||||
minio_norm[("MinIO<br/><i>Normalized Text</i><br/><i>Parser Output JSON</i>")]
|
||||
parser -->|"upload normalized text<br/>+ structured output"| minio_norm
|
||||
parser -->|"update document status,<br/>insert mentions"| pg_docs
|
||||
```
|
||||
|
||||
## Three Signal Layers
|
||||
|
||||
The parser routes documents into two extraction paths based on `document_type`. All three signal layers converge at the aggregation stage through the shared `WeightedSignal` abstraction.
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
%% ── Parser Output ─────────────────────────────────────────────
|
||||
parser(("Parser"))
|
||||
|
||||
%% ── Extraction Queues ─────────────────────────────────────────
|
||||
q_extraction[["stonks:queue:extraction"]]
|
||||
q_macro[["stonks:queue:macro_classification"]]
|
||||
|
||||
parser -->|"rpush<br/>(standard docs)"| q_extraction
|
||||
parser -->|"rpush<br/>(macro_event docs)"| q_macro
|
||||
|
||||
%% ── Scheduler Recovery ────────────────────────────────────────
|
||||
scheduler_recovery(("Scheduler<br/><i>stale recovery &<br/>failed retry</i>"))
|
||||
scheduler_recovery -.->|"re-enqueue orphaned<br/>parsed docs"| q_extraction
|
||||
scheduler_recovery -.->|"re-enqueue orphaned<br/>macro docs"| q_macro
|
||||
|
||||
%% ── Extractor Worker ──────────────────────────────────────────
|
||||
subgraph extractor_svc ["Extractor Service"]
|
||||
direction TB
|
||||
ext_main["<b>Extractor</b><br/><i>services.extractor.main</i><br/>Alternates between queues<br/>(2 extraction : 1 macro)<br/>Token budget enforcement"]
|
||||
end
|
||||
|
||||
q_extraction -->|"lpop"| ext_main
|
||||
q_macro -->|"lpop"| ext_main
|
||||
|
||||
%% ── Ollama LLM ───────────────────────────────────────────────
|
||||
ollama["<b>Ollama / vLLM</b><br/><i>LLM Inference</i><br/>document-extractor agent<br/>event-classifier agent"]
|
||||
ext_main <-->|"HTTP /api/generate<br/>(AgentConfigResolver<br/>selects model + variant)"| ollama
|
||||
|
||||
%% ── Signal Layer 1: Company ───────────────────────────────────
|
||||
subgraph layer1 ["Layer 1 — Company Signals"]
|
||||
direction LR
|
||||
di["document_intelligence<br/>document_impact_records"]
|
||||
end
|
||||
|
||||
ext_main -->|"persist extraction<br/>(standard docs)"| di
|
||||
|
||||
%% ── Signal Layer 2: Macro ─────────────────────────────────────
|
||||
subgraph layer2 ["Layer 2 — Macro Signals"]
|
||||
direction LR
|
||||
ge["global_events"]
|
||||
mir["macro_impact_records<br/><i>per-company interpolation<br/>via exposure profiles</i>"]
|
||||
ge --> mir
|
||||
end
|
||||
|
||||
ext_main -->|"classify & persist<br/>(macro_event docs)"| ge
|
||||
ext_main -->|"compute_macro_impact<br/>for all tracked companies"| mir
|
||||
|
||||
%% ── Aggregation Queue ─────────────────────────────────────────
|
||||
q_agg[["stonks:queue:aggregation"]]
|
||||
ext_main -->|"rpush<br/>(per ticker)"| q_agg
|
||||
|
||||
%% ── Scheduler Periodic Aggregation ────────────────────────────
|
||||
scheduler_agg(("Scheduler<br/><i>periodic aggregation<br/>every ~15 min</i>"))
|
||||
scheduler_agg -.->|"rpush all<br/>active tickers"| q_agg
|
||||
|
||||
%% ── Aggregation Worker ────────────────────────────────────────
|
||||
aggregation["<b>Aggregation</b><br/><i>services.aggregation.main</i><br/>Trend windows, scoring,<br/>contradiction detection"]
|
||||
|
||||
q_agg -->|"lpop"| aggregation
|
||||
|
||||
%% ── Signal Layer 3: Competitive ──────────────────────────────
|
||||
subgraph layer3 ["Layer 3 — Competitive Signals"]
|
||||
direction LR
|
||||
pm["pattern_matcher<br/><i>historical patterns</i>"]
|
||||
sp["signal_propagation<br/><i>cross-company signals</i>"]
|
||||
csr["competitive_signal_records"]
|
||||
pm --> sp --> csr
|
||||
end
|
||||
|
||||
aggregation -->|"trigger_signal_propagation<br/>(when competitive_enabled)"| layer3
|
||||
|
||||
%% ── All layers merge ──────────────────────────────────────────
|
||||
pg_trends[("PostgreSQL<br/><i>trend_windows,<br/>trend_history,<br/>trend_projections</i>")]
|
||||
|
||||
di -->|"WeightedSignal"| aggregation
|
||||
mir -->|"WeightedSignal"| aggregation
|
||||
csr -->|"WeightedSignal"| aggregation
|
||||
aggregation -->|"persist trend summaries"| pg_trends
|
||||
```
|
||||
|
||||
## Recommendation → Trading → Broker
|
||||
|
||||
The recommendation worker consumes from the recommendation queue. The trading engine does **not** consume from a queue — it polls the `recommendations` table in PostgreSQL on a configurable interval, evaluates each recommendation through its decision pipeline, and pushes "act" decisions to the broker queue.
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
%% ── Recommendation Queue ──────────────────────────────────────
|
||||
q_rec[["stonks:queue:recommendation"]]
|
||||
aggregation(("Aggregation")) -->|"rpush<br/>(ticker + window,<br/>dedup 5 min TTL)"| q_rec
|
||||
|
||||
%% ── Recommendation Worker ─────────────────────────────────────
|
||||
recommendation["<b>Recommendation</b><br/><i>services.recommendation.main</i><br/>Eligibility, suppression,<br/>thesis generation"]
|
||||
|
||||
q_rec -->|"lpop"| recommendation
|
||||
|
||||
ollama_thesis["<b>Ollama / vLLM</b><br/><i>thesis-rewriter agent</i><br/>(AgentConfigResolver<br/>selects model + variant)"]
|
||||
recommendation <-->|"rewrite thesis<br/>(trading-eligible only)"| ollama_thesis
|
||||
|
||||
pg_recs[("PostgreSQL<br/><i>recommendations,<br/>recommendation_evidence,<br/>risk_evaluations</i>")]
|
||||
recommendation -->|"persist recommendation<br/>+ evidence + risk eval"| pg_recs
|
||||
|
||||
%% ── Lake Publication (inline) ─────────────────────────────────
|
||||
minio_rec_lake[("MinIO<br/><i>Lakehouse</i><br/>recommendation facts")]
|
||||
recommendation -->|"publish_recommendation_facts<br/>(Parquet)"| minio_rec_lake
|
||||
|
||||
%% ── Trading Engine ────────────────────────────────────────────
|
||||
subgraph trading_loop ["Trading Engine Decision Loop"]
|
||||
direction TB
|
||||
poll["Poll recommendations<br/><i>action IN (buy, sell)<br/>mode IN (paper, live)<br/>generated_at > last_poll</i>"]
|
||||
dedup_check["Redis dedup check<br/><i>stonks:dedupe:trading:*</i>"]
|
||||
evaluate["evaluate_recommendation<br/><i>Circuit breaker check<br/>Trading window check<br/>Confidence gate<br/>Sector exposure check<br/>Correlation check<br/>Earnings blackout<br/>Max positions check</i>"]
|
||||
size["Position sizing<br/><i>Kelly criterion,<br/>risk tier limits,<br/>micro-trade support</i>"]
|
||||
decide{{"Decision"}}
|
||||
poll --> dedup_check --> evaluate --> size --> decide
|
||||
end
|
||||
|
||||
pg_recs -->|"SELECT recent<br/>recommendations"| poll
|
||||
|
||||
%% ── Broker Queue ──────────────────────────────────────────────
|
||||
q_broker[["stonks:queue:broker_orders"]]
|
||||
decide -->|"act → rpush<br/>order job"| q_broker
|
||||
decide -->|"skip → persist<br/>decision only"| pg_decisions
|
||||
|
||||
pg_decisions[("PostgreSQL<br/><i>trading_decisions</i>")]
|
||||
|
||||
%% ── Manual Override ───────────────────────────────────────────
|
||||
trading_api(("Trading API<br/><i>POST /override/order</i>"))
|
||||
trading_api -->|"rpush<br/>manual order"| q_broker
|
||||
|
||||
%% ── Broker Adapter ────────────────────────────────────────────
|
||||
broker["<b>Broker Adapter</b><br/><i>services.adapters.broker_service</i><br/>Idempotency, risk evaluation,<br/>approval gate, order submission,<br/>fill tracking, position sync"]
|
||||
|
||||
q_broker -->|"lpop"| broker
|
||||
|
||||
%% ── Risk Engine ───────────────────────────────────────────────
|
||||
risk["<b>Risk Engine</b><br/><i>services.risk.app</i><br/>evaluate_order()<br/>Position limits, sector exposure,<br/>daily loss caps, approval workflow"]
|
||||
broker -->|"evaluate order<br/>(inline call)"| risk
|
||||
|
||||
%% ── Alpaca ────────────────────────────────────────────────────
|
||||
alpaca["<b>Alpaca</b><br/><i>Paper Trading API</i><br/>Order submission,<br/>position sync,<br/>account state"]
|
||||
broker <-->|"submit order /<br/>sync positions /<br/>sync order status"| alpaca
|
||||
|
||||
pg_orders[("PostgreSQL<br/><i>orders, order_events,<br/>positions,<br/>portfolio_snapshots,<br/>broker_accounts</i>")]
|
||||
broker -->|"persist order,<br/>events, positions"| pg_orders
|
||||
|
||||
%% ── Lake Publication (broker inline) ──────────────────────────
|
||||
minio_broker_lake[("MinIO<br/><i>Lakehouse</i><br/>order + fill + position facts")]
|
||||
broker -->|"publish_trade_order<br/>publish_trade_fill<br/>publish_positions_daily_batch<br/>(Parquet)"| minio_broker_lake
|
||||
|
||||
%% ── Notifications ─────────────────────────────────────────────
|
||||
subgraph notifications ["Notifications"]
|
||||
direction LR
|
||||
sns["AWS SNS<br/><i>SMS alerts</i>"]
|
||||
gmail["Gmail SMTP<br/><i>Email alerts</i>"]
|
||||
end
|
||||
|
||||
trading_loop -->|"circuit breaker trips,<br/>order fills,<br/>stop-loss triggers"| notifications
|
||||
```
|
||||
|
||||
## Analytical Branch — Lake Publisher
|
||||
|
||||
The lake publisher runs as a separate worker, consuming from its own queue and writing partitioned Parquet fact tables to MinIO for analytical queries. Some services (broker adapter, recommendation worker) also publish facts directly to MinIO inline, bypassing the queue.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
%% ── Lake Publish Queue ────────────────────────────────────────
|
||||
q_lake[["stonks:queue:lake_publish"]]
|
||||
|
||||
various(("Upstream Services<br/><i>via enqueue_lake_job()</i>"))
|
||||
various -->|"rpush job<br/>(job_type + entity_id)"| q_lake
|
||||
|
||||
%% ── Lake Publisher Worker ─────────────────────────────────────
|
||||
lake["<b>Lake Publisher</b><br/><i>services.lake_publisher.jobs</i><br/>Transforms operational data<br/>into analytical facts<br/><i>15 job types supported</i>"]
|
||||
|
||||
q_lake -->|"lpop"| lake
|
||||
|
||||
pg_source[("PostgreSQL<br/><i>Operational Tables</i><br/>documents, extractions,<br/>orders, positions, events,<br/>global_events, macro_impacts,<br/>competitive_signals")]
|
||||
lake -->|"query source data"| pg_source
|
||||
|
||||
%% ── MinIO Parquet ─────────────────────────────────────────────
|
||||
minio_lake[("MinIO<br/><i>Lakehouse Bucket</i><br/>Partitioned Parquet<br/>/year=/month=/day=")]
|
||||
lake -->|"write Parquet files"| minio_lake
|
||||
|
||||
%% ── Inline Publishers ─────────────────────────────────────────
|
||||
inline(("Inline Publishers<br/><i>broker adapter,<br/>recommendation worker</i>"))
|
||||
inline -->|"publish_* functions<br/>(direct Parquet write)"| minio_lake
|
||||
|
||||
%% ── Trino ─────────────────────────────────────────────────────
|
||||
trino["<b>Trino</b><br/><i>SQL Query Engine</i><br/>Hive connector → MinIO"]
|
||||
minio_lake -->|"read via<br/>Hive Metastore"| trino
|
||||
|
||||
hive["<b>Hive Metastore</b><br/><i>Schema catalog</i>"]
|
||||
trino <-->|"table metadata"| hive
|
||||
hive -->|"location refs"| minio_lake
|
||||
|
||||
%% ── Visualization ─────────────────────────────────────────────
|
||||
superset["<b>Superset</b><br/><i>Dashboards &<br/>SQL Lab</i>"]
|
||||
dashboard["<b>React Dashboard</b><br/><i>frontend</i><br/>Charts, portfolio,<br/>recommendations"]
|
||||
query_api["<b>Query API</b><br/><i>services.api.app</i>"]
|
||||
|
||||
trino --> superset
|
||||
trino --> query_api
|
||||
query_api --> dashboard
|
||||
```
|
||||
|
||||
## Report Generation
|
||||
|
||||
The scheduler manages report generation as a sub-loop, enqueuing daily and weekly report jobs to a dedicated queue and consuming them inline.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
scheduler["<b>Scheduler</b><br/><i>report schedule check</i><br/>daily @ 16:30 ET<br/>weekly @ Saturday"]
|
||||
|
||||
q_report[["stonks:queue:report_generation"]]
|
||||
scheduler -->|"rpush<br/>(daily/weekly)"| q_report
|
||||
|
||||
scheduler_consumer["<b>Scheduler</b><br/><i>report consumer loop</i><br/>pops up to 5 jobs/cycle"]
|
||||
q_report -->|"lpop"| scheduler_consumer
|
||||
|
||||
generator["<b>Report Generator</b><br/><i>services.reporting.generator</i>"]
|
||||
scheduler_consumer -->|"process_report_job()"| generator
|
||||
|
||||
pg_reports[("PostgreSQL<br/><i>trading_reports</i>")]
|
||||
generator -->|"persist report"| pg_reports
|
||||
```
|
||||
|
||||
## Complete Queue Topology
|
||||
|
||||
| Queue | Full Key | Producer(s) | Consumer |
|
||||
|-------|----------|-------------|----------|
|
||||
| Ingestion | `stonks:queue:ingestion` | Scheduler (company, macro, global market sources) | Ingestion Worker |
|
||||
| Parsing | `stonks:queue:parsing` | Ingestion Worker (news, filings, web_scrape, macro) | Parser Worker |
|
||||
| Extraction | `stonks:queue:extraction` | Parser (standard docs), Scheduler (stale recovery) | Extractor Worker |
|
||||
| Macro Classification | `stonks:queue:macro_classification` | Parser (macro_event docs), Scheduler (stale/failed recovery) | Extractor Worker |
|
||||
| Aggregation | `stonks:queue:aggregation` | Extractor Worker (per ticker), Scheduler (periodic, all tickers) | Aggregation Worker |
|
||||
| Recommendation | `stonks:queue:recommendation` | Aggregation Worker (ticker + window, 5 min dedup TTL) | Recommendation Worker |
|
||||
| Broker Orders | `stonks:queue:broker_orders` | Trading Engine (act decisions), Trading API (manual overrides) | Broker Adapter |
|
||||
| Lake Publish | `stonks:queue:lake_publish` | Various services (via `enqueue_lake_job()`) | Lake Publisher |
|
||||
| Report Generation | `stonks:queue:report_generation` | Scheduler (daily/weekly triggers) | Scheduler (inline consumer) |
|
||||
|
||||
Dead-letter queues follow the pattern `stonks:dlq:<queue_name>` and are populated when a job exhausts its retry budget.
|
||||
|
||||
## Data Store Summary
|
||||
|
||||
| Store | Role | Key Tables / Buckets |
|
||||
|-------|------|---------------------|
|
||||
| **PostgreSQL** | Structured operational data | `documents`, `document_intelligence`, `document_impact_records`, `document_company_mentions`, `global_events`, `macro_impact_records`, `exposure_profiles`, `competitive_signal_records`, `competitor_relationships`, `trend_windows`, `trend_history`, `trend_projections`, `recommendations`, `recommendation_evidence`, `risk_evaluations`, `orders`, `order_events`, `positions`, `portfolio_snapshots`, `trading_decisions`, `circuit_breaker_events`, `reserve_pool_ledger`, `risk_tier_history`, `broker_accounts`, `ingestion_runs`, `sources`, `companies`, `company_aliases`, `ai_agents`, `agent_variants`, `agent_performance_log`, `risk_configs`, `trading_reports` |
|
||||
| **Redis** | Queues, dedup markers, rate limits, circuit breaker state, pipeline toggle | `stonks:queue:*` (9 queues), `stonks:dedupe:*`, `stonks:dedupe:trading:*`, `stonks:ratelimit:*`, `stonks:trading:circuit_breaker:*`, `stonks:trading:notification_rate:*`, `stonks:order_idempotency:*`, `stonks:lock:*`, `stonks:cache:*`, `stonks:retry:*`, `stonks:rec_dedup:*`, `stonks:pipeline:enabled`, `stonks:dlq:*` |
|
||||
| **MinIO** | Object storage for raw artifacts, normalized text, and analytical Parquet files | Raw artifacts bucket, normalized text bucket, parser output bucket, lakehouse bucket (partitioned Parquet: documents, extractions, market bars/quotes, orders, fills, positions, PnL, global events, macro impacts, trend projections, competitive signals, competitor relationships, recommendations) |
|
||||
|
||||
## External Integration Points
|
||||
|
||||
| Integration | Service | Protocol | Purpose |
|
||||
|-------------|---------|----------|---------|
|
||||
| **Polygon.io** | Ingestion (via PolygonNewsAdapter, PolygonMarketAdapter) | HTTPS REST | News articles, market bars, grouped daily data, intraday bars |
|
||||
| **SEC EDGAR** | Ingestion (via SECEdgarAdapter) | HTTPS REST | 10-K, 10-Q filings |
|
||||
| **Macro News** | Ingestion (via MacroNewsAdapter) | HTTPS REST | Geopolitical and economic event articles |
|
||||
| **Ollama / vLLM** | Extractor, Recommendation | HTTP `/api/generate` | LLM inference for document extraction (document-extractor agent), event classification (event-classifier agent), thesis rewriting (thesis-rewriter agent). Model and variant selected via `AgentConfigResolver` with 60s TTL cache. |
|
||||
| **Alpaca** | Broker Adapter | HTTPS REST | Paper/live trading: order submission, position sync, account state, order status polling |
|
||||
| **AWS SNS** | Trading Engine (notifications) | boto3 SDK | SMS alerts for circuit breaker trips, order fills, stop-loss triggers |
|
||||
| **Gmail** | Trading Engine (notifications) | SMTP (port 587 STARTTLS) | Email alerts for trading events |
|
||||
| **Trino** | Query API, Superset | HTTP | SQL queries over lakehouse Parquet files via Hive Metastore |
|
||||
|
||||
## Pipeline Toggle
|
||||
|
||||
The pipeline can be paused globally via the Redis key `stonks:pipeline:enabled`. When set to `"0"`, all queue workers (ingestion, parser, extractor, aggregation, recommendation, broker adapter, lake publisher) enter a sleep loop and stop processing jobs. The scheduler also skips scheduling cycles when the toggle is off. The toggle can be set via the Query API's pipeline control endpoints.
|
||||
|
||||
Setting `PIPELINE_DEFAULT_OFF=true` on the scheduler initializes the toggle to OFF on first boot, useful for staged deployments where you want to verify infrastructure before enabling the pipeline.
|
||||
@@ -0,0 +1,323 @@
|
||||
# Docker Compose Architecture — Stonks Oracle
|
||||
|
||||
This document describes the Docker Compose deployment topology for Stonks Oracle, derived from the `docker-compose.yml` file at the repository root.
|
||||
|
||||
All containers run on a single Docker network created by Compose. Infrastructure services (PostgreSQL, Redis, MinIO, Ollama, Trino, Hive Metastore, Superset) start first, and application services wait for their dependencies via `depends_on` with health check conditions.
|
||||
|
||||
## Container Topology Diagram
|
||||
|
||||
```mermaid
|
||||
graph TB
|
||||
%% ── Host machine ──────────────────────────────────────────────
|
||||
host((Host Machine))
|
||||
|
||||
%% ── .env file ─────────────────────────────────────────────────
|
||||
envfile[".env file<br/><i>MARKET_DATA_API_KEY</i><br/><i>BROKER_API_KEY</i><br/><i>BROKER_API_SECRET</i><br/><i>BROKER_BASE_URL</i>"]
|
||||
|
||||
%% ── Docker Compose default network ────────────────────────────
|
||||
subgraph network ["Docker Compose Network (default)"]
|
||||
direction TB
|
||||
|
||||
%% ── Infrastructure Containers ─────────────────────────────
|
||||
subgraph infra ["Infrastructure Containers"]
|
||||
direction LR
|
||||
postgres[("postgres<br/><i>postgres:16-alpine</i><br/>host :5432 → :5432")]
|
||||
redis[("redis<br/><i>redis:7-alpine</i><br/>host :6379 → :6379")]
|
||||
minio[("minio<br/><i>minio/minio:latest</i><br/>host :9000 → :9000<br/>host :9001 → :9001")]
|
||||
ollama[("ollama<br/><i>ollama/ollama:latest</i><br/>host :11434 → :11434")]
|
||||
end
|
||||
|
||||
subgraph infra_init ["Infrastructure Init"]
|
||||
minio_init["minio-init<br/><i>minio/mc:latest</i><br/><i>Creates buckets on startup</i>"]
|
||||
end
|
||||
|
||||
subgraph analytics ["Analytics Containers"]
|
||||
direction LR
|
||||
hive_metastore["hive-metastore<br/><i>apache/hive:4.0.0</i><br/>host :9083 → :9083"]
|
||||
trino["trino<br/><i>trinodb/trino:latest</i><br/>host :8080 → :8080"]
|
||||
superset["superset<br/><i>apache/superset:latest</i><br/>host :8088 → :8088"]
|
||||
end
|
||||
|
||||
%% ── Application Containers ────────────────────────────────
|
||||
|
||||
subgraph api_tier ["API Tier"]
|
||||
direction LR
|
||||
query_api["query-api<br/><i>docker/Dockerfile</i><br/><i>uvicorn services.api.app</i><br/>host :8004 → :8000"]
|
||||
symbol_registry["symbol-registry<br/><i>docker/Dockerfile</i><br/><i>uvicorn services.symbol_registry.app</i><br/>host :8001 → :8000"]
|
||||
end
|
||||
|
||||
subgraph frontend_tier ["Frontend Tier"]
|
||||
dashboard["dashboard<br/><i>frontend/Dockerfile</i><br/><i>nginx on :8080</i><br/>host :3000 → :8080"]
|
||||
end
|
||||
|
||||
subgraph trading_tier ["Trading Tier"]
|
||||
direction LR
|
||||
trading_engine["trading-engine<br/><i>docker/Dockerfile</i><br/><i>uvicorn services.trading.app</i><br/>host :8002 → :8000"]
|
||||
risk_engine["risk-engine<br/><i>docker/Dockerfile</i><br/><i>uvicorn services.risk.app</i><br/>host :8003 → :8000<br/><i>alias: risk</i>"]
|
||||
broker_adapter["broker-adapter<br/><i>docker/Dockerfile</i><br/><i>python -m services.adapters.broker_service</i><br/><i>no host port</i>"]
|
||||
end
|
||||
|
||||
subgraph orchestration_tier ["Orchestration Tier"]
|
||||
scheduler["scheduler<br/><i>docker/Dockerfile.scheduler</i><br/><i>no host port</i>"]
|
||||
end
|
||||
|
||||
subgraph processing_tier ["Processing Tier (pipeline workers)"]
|
||||
direction LR
|
||||
ingestion["ingestion<br/><i>docker/Dockerfile</i><br/><i>python -m services.ingestion.worker</i><br/><i>no host port</i>"]
|
||||
parser["parser<br/><i>docker/Dockerfile</i><br/><i>python -m services.parser.worker</i><br/><i>no host port</i>"]
|
||||
extractor["extractor<br/><i>docker/Dockerfile</i><br/><i>python -m services.extractor.main</i><br/><i>no host port</i>"]
|
||||
aggregation["aggregation<br/><i>docker/Dockerfile</i><br/><i>python -m services.aggregation.main</i><br/><i>no host port</i>"]
|
||||
recommendation["recommendation<br/><i>docker/Dockerfile</i><br/><i>python -m services.recommendation.main</i><br/><i>no host port</i>"]
|
||||
end
|
||||
|
||||
subgraph analytics_worker ["Analytics Worker"]
|
||||
lake_publisher["lake-publisher<br/><i>docker/Dockerfile</i><br/><i>python -m services.lake_publisher.jobs</i><br/><i>no host port</i>"]
|
||||
end
|
||||
end
|
||||
|
||||
%% ── Host port access ──────────────────────────────────────────
|
||||
host -->|":5432"| postgres
|
||||
host -->|":6379"| redis
|
||||
host -->|":9000 / :9001"| minio
|
||||
host -->|":11434"| ollama
|
||||
host -->|":8080"| trino
|
||||
host -->|":9083"| hive_metastore
|
||||
host -->|":8088"| superset
|
||||
host -->|":8001"| symbol_registry
|
||||
host -->|":8004"| query_api
|
||||
host -->|":8002"| trading_engine
|
||||
host -->|":8003"| risk_engine
|
||||
host -->|":3000"| dashboard
|
||||
|
||||
%% ── .env injection ────────────────────────────────────────────
|
||||
envfile -.->|"env_file: .env"| ingestion
|
||||
envfile -.->|"env_file: .env"| broker_adapter
|
||||
envfile -.->|"env_file: .env"| trading_engine
|
||||
|
||||
%% ── Styles ────────────────────────────────────────────────────
|
||||
classDef infraSvc fill:#95a5a6,stroke:#717d7e,color:#fff
|
||||
classDef analyticsSvc fill:#e74c3c,stroke:#a93226,color:#fff
|
||||
classDef apiSvc fill:#4a90d9,stroke:#2c5f8a,color:#fff
|
||||
classDef frontendSvc fill:#50c878,stroke:#2e7d46,color:#fff
|
||||
classDef tradingSvc fill:#e8a838,stroke:#b07d1a,color:#fff
|
||||
classDef orchSvc fill:#1abc9c,stroke:#148f77,color:#fff
|
||||
classDef processSvc fill:#9b59b6,stroke:#6c3483,color:#fff
|
||||
classDef initSvc fill:#bdc3c7,stroke:#7f8c8d,color:#333
|
||||
classDef envSvc fill:#f5f5dc,stroke:#999,color:#333
|
||||
|
||||
class postgres,redis,minio,ollama infraSvc
|
||||
class hive_metastore,trino,superset,lake_publisher analyticsSvc
|
||||
class query_api,symbol_registry apiSvc
|
||||
class dashboard frontendSvc
|
||||
class trading_engine,risk_engine,broker_adapter tradingSvc
|
||||
class scheduler orchSvc
|
||||
class ingestion,parser,extractor,aggregation,recommendation processSvc
|
||||
class minio_init initSvc
|
||||
class envfile envSvc
|
||||
```
|
||||
|
||||
## Dependency Graph
|
||||
|
||||
The following diagram shows `depends_on` relationships and health check conditions. Solid arrows indicate `condition: service_healthy` (the dependent waits for the health check to pass). Dashed arrows indicate `condition: service_started` (the dependent waits only for the container to start).
|
||||
|
||||
```mermaid
|
||||
graph LR
|
||||
%% ── Infrastructure health checks ──────────────────────────────
|
||||
postgres[("postgres<br/><i>pg_isready -U stonks</i>")]
|
||||
redis[("redis<br/><i>redis-cli ping</i>")]
|
||||
minio[("minio<br/><i>mc ready local</i>")]
|
||||
ollama[("ollama<br/><i>no health check</i>")]
|
||||
|
||||
%% ── Analytics dependencies ────────────────────────────────────
|
||||
hive["hive-metastore"] -->|started| minio
|
||||
trino["trino"] -->|started| minio
|
||||
trino -->|started| hive
|
||||
superset["superset"] -->|started| trino
|
||||
minio_init["minio-init"] -->|healthy| minio
|
||||
|
||||
%% ── Application depends_on (healthy) ──────────────────────────
|
||||
scheduler["scheduler"] -->|healthy| postgres
|
||||
scheduler -->|healthy| redis
|
||||
|
||||
symbol_registry["symbol-registry"] -->|healthy| postgres
|
||||
|
||||
ingestion["ingestion"] -->|healthy| postgres
|
||||
ingestion -->|healthy| redis
|
||||
ingestion -->|healthy| minio
|
||||
|
||||
parser["parser"] -->|healthy| postgres
|
||||
parser -->|healthy| redis
|
||||
|
||||
extractor["extractor"] -->|healthy| postgres
|
||||
extractor -->|healthy| redis
|
||||
extractor -.->|started| ollama
|
||||
|
||||
aggregation["aggregation"] -->|healthy| postgres
|
||||
aggregation -->|healthy| redis
|
||||
|
||||
recommendation["recommendation"] -->|healthy| postgres
|
||||
recommendation -->|healthy| redis
|
||||
|
||||
trading_engine["trading-engine"] -->|healthy| postgres
|
||||
trading_engine -->|healthy| redis
|
||||
|
||||
risk_engine["risk-engine"] -->|healthy| postgres
|
||||
|
||||
broker_adapter["broker-adapter"] -->|healthy| postgres
|
||||
broker_adapter -->|healthy| redis
|
||||
|
||||
lake_publisher["lake-publisher"] -->|healthy| postgres
|
||||
lake_publisher -->|healthy| minio
|
||||
|
||||
query_api["query-api"] -->|healthy| postgres
|
||||
query_api -->|healthy| redis
|
||||
query_api -->|healthy| minio
|
||||
|
||||
dashboard["dashboard"] -->|healthy| query_api
|
||||
|
||||
%% ── Styles ────────────────────────────────────────────────────
|
||||
classDef infraSvc fill:#95a5a6,stroke:#717d7e,color:#fff
|
||||
classDef appSvc fill:#4a90d9,stroke:#2c5f8a,color:#fff
|
||||
classDef analyticsSvc fill:#e74c3c,stroke:#a93226,color:#fff
|
||||
classDef initSvc fill:#bdc3c7,stroke:#7f8c8d,color:#333
|
||||
|
||||
class postgres,redis,minio,ollama infraSvc
|
||||
class scheduler,symbol_registry,ingestion,parser,extractor,aggregation,recommendation,trading_engine,risk_engine,broker_adapter,lake_publisher,query_api,dashboard appSvc
|
||||
class hive,trino,superset analyticsSvc
|
||||
class minio_init initSvc
|
||||
```
|
||||
|
||||
## Named Volumes
|
||||
|
||||
Docker Compose defines five named volumes for persistent data:
|
||||
|
||||
```mermaid
|
||||
graph LR
|
||||
pgdata["📦 pgdata"]
|
||||
miniodata["📦 miniodata"]
|
||||
ollama_models["📦 ollama_models"]
|
||||
hive_data["📦 hive_data"]
|
||||
superset_data["📦 superset_data"]
|
||||
|
||||
pgdata -->|"/var/lib/postgresql/data"| postgres[("postgres")]
|
||||
miniodata -->|"/data"| minio[("minio")]
|
||||
ollama_models -->|"/root/.ollama"| ollama[("ollama")]
|
||||
hive_data -->|"/opt/hive/data"| hive["hive-metastore"]
|
||||
superset_data -->|"/app/superset_home"| superset["superset"]
|
||||
|
||||
classDef volStyle fill:#f5f5dc,stroke:#999,color:#333
|
||||
classDef svcStyle fill:#95a5a6,stroke:#717d7e,color:#fff
|
||||
|
||||
class pgdata,miniodata,ollama_models,hive_data,superset_data volStyle
|
||||
class postgres,minio,ollama,hive,superset svcStyle
|
||||
```
|
||||
|
||||
| Volume | Mount Point | Container | Purpose |
|
||||
|--------|-------------|-----------|---------|
|
||||
| `pgdata` | `/var/lib/postgresql/data` | postgres | PostgreSQL database files |
|
||||
| `miniodata` | `/data` | minio | MinIO object storage data |
|
||||
| `ollama_models` | `/root/.ollama` | ollama | Downloaded LLM model weights |
|
||||
| `hive_data` | `/opt/hive/data` | hive-metastore | Hive Metastore embedded Derby DB |
|
||||
| `superset_data` | `/app/superset_home` | superset | Superset configuration and metadata |
|
||||
|
||||
### Bind Mounts
|
||||
|
||||
In addition to named volumes, several containers use bind mounts for configuration files:
|
||||
|
||||
| Host Path | Mount Point | Container | Mode |
|
||||
|-----------|-------------|-----------|------|
|
||||
| `./infra/migrations/` | `/docker-entrypoint-initdb.d` | postgres | rw (init scripts) |
|
||||
| `./infra/trino/catalog/` | `/etc/trino/catalog` | trino | rw |
|
||||
| `./infra/hive/core-site.xml` | `/opt/hive/conf/core-site.xml` | hive-metastore | ro |
|
||||
| `./infra/hive/metastore-site.xml` | `/opt/hive/conf/metastore-site.xml` | hive-metastore | ro |
|
||||
|
||||
## Host Port Mappings
|
||||
|
||||
Services accessible from the host machine:
|
||||
|
||||
| Host Port | Container | Container Port | Service |
|
||||
|-----------|-----------|----------------|---------|
|
||||
| 5432 | postgres | 5432 | PostgreSQL database |
|
||||
| 6379 | redis | 6379 | Redis cache and queues |
|
||||
| 9000 | minio | 9000 | MinIO S3 API |
|
||||
| 9001 | minio | 9001 | MinIO web console |
|
||||
| 11434 | ollama | 11434 | Ollama LLM API |
|
||||
| 8080 | trino | 8080 | Trino query engine |
|
||||
| 9083 | hive-metastore | 9083 | Hive Metastore thrift |
|
||||
| 8088 | superset | 8088 | Superset dashboard |
|
||||
| 8001 | symbol-registry | 8000 | Symbol Registry API |
|
||||
| 8002 | trading-engine | 8000 | Trading Engine API |
|
||||
| 8003 | risk-engine | 8000 | Risk Engine API |
|
||||
| 8004 | query-api | 8000 | Query API |
|
||||
| 3000 | dashboard | 8080 | React dashboard (nginx) |
|
||||
|
||||
Services without host port mappings (internal only): scheduler, ingestion, parser, extractor, aggregation, recommendation, broker-adapter, lake-publisher, minio-init.
|
||||
|
||||
## Environment Configuration
|
||||
|
||||
### Shared Environment (`x-app-env` YAML anchor)
|
||||
|
||||
All 13 application services and the scheduler receive these environment variables via the `x-app-env` anchor:
|
||||
|
||||
| Variable | Value | Purpose |
|
||||
|----------|-------|---------|
|
||||
| `POSTGRES_HOST` | `postgres` | Docker Compose service name for PostgreSQL |
|
||||
| `POSTGRES_PORT` | `5432` | PostgreSQL port |
|
||||
| `POSTGRES_DB` | `stonks` | Database name |
|
||||
| `POSTGRES_USER` | `stonks` | Database user |
|
||||
| `POSTGRES_PASSWORD` | `stonks_dev` | Database password (dev default) |
|
||||
| `REDIS_HOST` | `redis` | Docker Compose service name for Redis |
|
||||
| `REDIS_PORT` | `6379` | Redis port |
|
||||
| `MINIO_ENDPOINT` | `minio:9000` | Docker Compose service name for MinIO |
|
||||
| `MINIO_ACCESS_KEY` | `minioadmin` | MinIO access key |
|
||||
| `MINIO_SECRET_KEY` | `minioadmin` | MinIO secret key |
|
||||
| `OLLAMA_BASE_URL` | `http://ollama:11434` | Docker Compose service name for Ollama |
|
||||
|
||||
### `.env` File (API Keys)
|
||||
|
||||
Three services load additional secrets from the `.env` file in the repository root via `env_file: .env`:
|
||||
|
||||
| Variable | Required By | Purpose |
|
||||
|----------|-------------|---------|
|
||||
| `MARKET_DATA_API_KEY` | ingestion | Polygon.io market data API key |
|
||||
| `BROKER_API_KEY` | broker-adapter, trading-engine | Alpaca broker API key |
|
||||
| `BROKER_API_SECRET` | broker-adapter, trading-engine | Alpaca broker API secret |
|
||||
| `BROKER_BASE_URL` | broker-adapter, trading-engine | Alpaca API base URL (default: `https://paper-api.alpaca.markets`) |
|
||||
|
||||
## Health Check Summary
|
||||
|
||||
| Container | Health Check Command | Interval | Timeout | Retries | Start Period |
|
||||
|-----------|---------------------|----------|---------|---------|--------------|
|
||||
| postgres | `pg_isready -U stonks` | 5s | — | 5 | — |
|
||||
| redis | `redis-cli ping` | 5s | — | 5 | — |
|
||||
| minio | `mc ready local` | 5s | — | 5 | — |
|
||||
| symbol-registry | `curl -f http://localhost:8000/health` | 10s | 5s | 3 | 15s |
|
||||
| query-api | `curl -f http://localhost:8000/health` | 10s | 5s | 3 | 15s |
|
||||
| trading-engine | `curl -f http://localhost:8000/health` | 10s | 5s | 3 | 15s |
|
||||
| risk-engine | `curl -f http://localhost:8000/health` | 10s | 5s | 3 | 15s |
|
||||
| dashboard | `curl -f http://localhost:8080/` | 10s | 5s | 3 | 10s |
|
||||
| scheduler | `pgrep -f 'python -m services.scheduler.app'` | 10s | 5s | 3 | 15s |
|
||||
| ingestion | `pgrep -f 'python -m services.ingestion.worker'` | 10s | 5s | 3 | 15s |
|
||||
| parser | `pgrep -f 'python -m services.parser.worker'` | 10s | 5s | 3 | 15s |
|
||||
| extractor | `pgrep -f 'python -m services.extractor.main'` | 10s | 5s | 3 | 15s |
|
||||
| aggregation | `pgrep -f 'python -m services.aggregation.main'` | 10s | 5s | 3 | 15s |
|
||||
| recommendation | `pgrep -f 'python -m services.recommendation.main'` | 10s | 5s | 3 | 15s |
|
||||
| broker-adapter | `pgrep -f 'python -m services.adapters.broker_service'` | 10s | 5s | 3 | 15s |
|
||||
| lake-publisher | `pgrep -f 'python -m services.lake_publisher.jobs'` | 10s | 5s | 3 | 15s |
|
||||
|
||||
Infrastructure services (ollama, trino, hive-metastore, superset) do not define health checks in docker-compose.yml. Application services that depend on ollama use `condition: service_started` instead of `condition: service_healthy`.
|
||||
|
||||
## Internal Network Connectivity
|
||||
|
||||
All containers share the default Docker Compose network. Services reference each other by their Compose service name as the hostname:
|
||||
|
||||
| Hostname | Resolved To | Used By |
|
||||
|----------|-------------|---------|
|
||||
| `postgres` | PostgreSQL container | All 13 app services, superset |
|
||||
| `redis` | Redis container | scheduler, ingestion, parser, extractor, aggregation, recommendation, trading-engine, broker-adapter, query-api |
|
||||
| `minio` | MinIO container | ingestion, lake-publisher, query-api (via `minio:9000`) |
|
||||
| `ollama` | Ollama container | extractor (via `http://ollama:11434`) |
|
||||
| `hive-metastore` | Hive Metastore container | trino (thrift://hive-metastore:9083) |
|
||||
| `trino` | Trino container | superset (trino:8080) |
|
||||
| `query-api` | Query API container | dashboard (nginx proxy upstream) |
|
||||
| `risk` | risk-engine container (network alias) | trading-engine (risk evaluation calls) |
|
||||
@@ -0,0 +1,380 @@
|
||||
# Kubernetes Architecture — Stonks Oracle
|
||||
|
||||
This document describes the Kubernetes deployment topology for Stonks Oracle, derived from the Helm chart at `infra/helm/stonks-oracle/`.
|
||||
|
||||
All application workloads deploy to the `stonks-oracle` namespace. External cluster services (PostgreSQL, Redis, MinIO, Ollama) run in their own namespaces and are referenced via cross-namespace DNS.
|
||||
|
||||
## Deployment Diagram
|
||||
|
||||
```mermaid
|
||||
graph TB
|
||||
%% ── External traffic ──────────────────────────────────────────
|
||||
internet((Internet))
|
||||
|
||||
subgraph traefik ["kube-system · Traefik Ingress Controller"]
|
||||
direction LR
|
||||
ing_dash["stonks.celestium.life"]
|
||||
ing_api["stonks-api.celestium.life"]
|
||||
ing_reg["stonks-registry.celestium.life"]
|
||||
ing_trade["stonks-trading.celestium.life"]
|
||||
ing_superset["stonks-dash.celestium.life"]
|
||||
ing_trino["stonks-trino.celestium.life"]
|
||||
end
|
||||
|
||||
internet --> traefik
|
||||
|
||||
%% ── stonks-oracle namespace ───────────────────────────────────
|
||||
subgraph ns ["stonks-oracle namespace"]
|
||||
direction TB
|
||||
|
||||
%% ── API Tier (ingress-facing) ─────────────────────────────
|
||||
subgraph api_tier ["API Tier · tier: api"]
|
||||
direction LR
|
||||
query_api["query-api<br/><i>Deployment · 1 replica</i><br/>:8000<br/><i>readiness: /docs</i>"]
|
||||
symbol_registry["symbol-registry<br/><i>Deployment · 1 replica</i><br/>:8000<br/><i>readiness: /docs · liveness: /docs</i>"]
|
||||
end
|
||||
|
||||
%% ── Frontend Tier ─────────────────────────────────────────
|
||||
subgraph frontend_tier ["Frontend Tier · tier: frontend"]
|
||||
dashboard["dashboard<br/><i>Deployment · 1 replica</i><br/>:8080<br/><i>nginx-unprivileged</i><br/><i>readiness: / · liveness: /</i>"]
|
||||
end
|
||||
|
||||
%% ── Trading Tier ──────────────────────────────────────────
|
||||
subgraph trading_tier ["Trading Tier · tier: trading"]
|
||||
direction LR
|
||||
trading_engine["trading-engine<br/><i>Deployment · 1 replica</i><br/>:8000<br/><i>readiness: /ready · liveness: /health</i>"]
|
||||
risk_engine["risk-engine<br/><i>Deployment · 1 replica</i><br/>:8000"]
|
||||
broker_adapter["broker-adapter<br/><i>Deployment · 1 replica</i><br/><i>queue-driven worker · pipeline-gated</i>"]
|
||||
end
|
||||
|
||||
%% ── Orchestration Tier ────────────────────────────────────
|
||||
subgraph orchestration_tier ["Orchestration Tier · tier: orchestration"]
|
||||
scheduler["scheduler<br/><i>Deployment · 1 replica · pipeline-gated</i><br/><i>init: migrations → seed → backfill</i>"]
|
||||
end
|
||||
|
||||
%% ── Ingestion Tier ────────────────────────────────────────
|
||||
subgraph ingestion_tier ["Ingestion Tier · tier: ingestion"]
|
||||
ingestion["ingestion<br/><i>Deployment · 1 replica · pipeline-gated</i><br/><i>queue-driven worker</i>"]
|
||||
end
|
||||
|
||||
%% ── Processing Tier (pipeline workers) ────────────────────
|
||||
subgraph processing_tier ["Processing Tier · tier: processing"]
|
||||
direction LR
|
||||
parser["parser<br/><i>Deployment · 2 replicas · pipeline-gated</i>"]
|
||||
extractor["extractor<br/><i>Deployment · 1 replica · pipeline-gated</i>"]
|
||||
aggregation["aggregation<br/><i>Deployment · 4 replicas · pipeline-gated</i>"]
|
||||
recommendation["recommendation<br/><i>Deployment · 1 replica · pipeline-gated</i>"]
|
||||
end
|
||||
|
||||
%% ── Analytics Tier ────────────────────────────────────────
|
||||
subgraph analytics_tier ["Analytics Tier · tier: analytics"]
|
||||
direction LR
|
||||
lake_publisher["lake-publisher<br/><i>Deployment · 1 replica · pipeline-gated</i><br/><i>queue-driven worker</i>"]
|
||||
hive_metastore["hive-metastore<br/><i>Deployment · 1 replica</i><br/>:9083<br/><i>apache/hive:4.0.0</i><br/><i>PVC: hive-metastore-data</i>"]
|
||||
trino["trino<br/><i>Deployment · 1 replica</i><br/>:8080<br/><i>trinodb/trino:latest</i><br/><i>readiness: /v1/info</i>"]
|
||||
end
|
||||
|
||||
%% ── Superset (tier: dashboard in template) ────────────────
|
||||
subgraph superset_block ["Superset · tier: dashboard"]
|
||||
superset["superset<br/><i>Deployment · 1 replica</i><br/>:8088<br/><i>custom image</i><br/><i>PVC: superset-data</i><br/><i>readiness: /health</i>"]
|
||||
end
|
||||
|
||||
%% ── Helm Secrets ──────────────────────────────────────────
|
||||
subgraph secrets_block ["Helm-Managed Secrets"]
|
||||
direction LR
|
||||
sec_core["stonks-core-secrets<br/><i>POSTGRES_PASSWORD</i><br/><i>MINIO_ACCESS_KEY</i><br/><i>MINIO_SECRET_KEY</i><br/><i>REDIS_PASSWORD</i>"]
|
||||
sec_broker["stonks-broker-secrets<br/><i>BROKER_API_KEY</i><br/><i>BROKER_API_SECRET</i><br/><i>BROKER_BASE_URL</i>"]
|
||||
sec_market["stonks-market-secrets<br/><i>MARKET_DATA_API_KEY</i>"]
|
||||
sec_gmail["stonks-gmail-secrets<br/><i>GMAIL_SENDER</i><br/><i>GMAIL_RECIPIENT</i><br/><i>GMAIL_APP_PASSWORD</i>"]
|
||||
sec_dashboard["stonks-dashboard-secrets<br/><i>SUPERSET_SECRET_KEY</i><br/><i>SUPERSET_ADMIN_PASSWORD</i>"]
|
||||
end
|
||||
|
||||
%% ── ConfigMap ─────────────────────────────────────────────
|
||||
configmap["stonks-config<br/><i>ConfigMap</i><br/><i>All env vars from values.yaml config block</i>"]
|
||||
end
|
||||
|
||||
%% ── External Cluster Services ─────────────────────────────────
|
||||
subgraph pg_ns ["postgresql-service namespace"]
|
||||
postgres[("PostgreSQL<br/>postgresql-rw:5432")]
|
||||
end
|
||||
|
||||
subgraph redis_ns ["redis-service namespace"]
|
||||
redis[("Redis<br/>redis-master:6379")]
|
||||
end
|
||||
|
||||
subgraph minio_ns ["minio-service namespace"]
|
||||
minio[("MinIO<br/>minio:80")]
|
||||
end
|
||||
|
||||
subgraph ollama_ns ["ollama-service namespace"]
|
||||
ollama[("Ollama<br/>ollama:11434<br/><i>GPU: 4070 Ti Super 16GB</i>")]
|
||||
end
|
||||
|
||||
%% ── Ingress Routes ────────────────────────────────────────────
|
||||
ing_dash -->|":8080"| dashboard
|
||||
ing_api -->|":8000"| query_api
|
||||
ing_reg -->|":8000"| symbol_registry
|
||||
ing_trade -->|":8000"| trading_engine
|
||||
ing_superset -->|":8088"| superset
|
||||
ing_trino -->|":8080"| trino
|
||||
|
||||
%% ── Dashboard → Backend APIs ──────────────────────────────────
|
||||
dashboard -.->|"/api/ proxy"| query_api
|
||||
dashboard -.->|"/registry/ proxy"| symbol_registry
|
||||
dashboard -.->|"/risk/ proxy"| risk_engine
|
||||
|
||||
%% ── Pipeline data flow (via Redis queues) ─────────────────────
|
||||
scheduler -->|"enqueue jobs"| redis
|
||||
ingestion -->|"stonks:queue:parsing"| redis
|
||||
parser -->|"stonks:queue:extraction"| redis
|
||||
extractor -->|"stonks:queue:aggregation"| redis
|
||||
aggregation -->|"stonks:queue:recommendation"| redis
|
||||
recommendation -->|"stonks:queue:trading_decisions"| redis
|
||||
trading_engine -->|"stonks:queue:broker_orders"| redis
|
||||
broker_adapter -->|"read orders"| redis
|
||||
lake_publisher -->|"stonks:queue:lake_publish"| redis
|
||||
|
||||
%% ── External service connections ──────────────────────────────
|
||||
scheduler --> postgres
|
||||
scheduler --> redis
|
||||
ingestion --> postgres
|
||||
ingestion --> redis
|
||||
ingestion --> minio
|
||||
parser --> postgres
|
||||
parser --> redis
|
||||
extractor --> postgres
|
||||
extractor --> redis
|
||||
extractor --> ollama
|
||||
aggregation --> postgres
|
||||
aggregation --> redis
|
||||
recommendation --> postgres
|
||||
recommendation --> redis
|
||||
trading_engine --> postgres
|
||||
trading_engine --> redis
|
||||
risk_engine --> postgres
|
||||
broker_adapter --> postgres
|
||||
broker_adapter --> redis
|
||||
lake_publisher --> postgres
|
||||
lake_publisher --> minio
|
||||
query_api --> postgres
|
||||
query_api --> redis
|
||||
query_api --> minio
|
||||
symbol_registry --> postgres
|
||||
|
||||
%% ── Analytics plane connections ───────────────────────────────
|
||||
lake_publisher -->|"Parquet → s3a://stonks-lakehouse"| minio
|
||||
hive_metastore -->|"s3a:// catalog"| minio
|
||||
trino -->|"thrift://hive-metastore:9083"| hive_metastore
|
||||
superset -->|"trino:8080"| trino
|
||||
query_api -->|"trino:8080"| trino
|
||||
superset --> postgres
|
||||
superset --> redis
|
||||
|
||||
%% ── Trading tier external egress ──────────────────────────────
|
||||
trading_engine -->|"HTTPS :443<br/>Alpaca API"| internet
|
||||
trading_engine -->|"SMTP :587<br/>Gmail notifications"| internet
|
||||
broker_adapter -->|"HTTPS :443<br/>Alpaca API"| internet
|
||||
ingestion -->|"HTTPS :443<br/>Polygon.io / News APIs"| internet
|
||||
|
||||
%% ── Secret consumption ────────────────────────────────────────
|
||||
sec_core -.-> query_api
|
||||
sec_core -.-> symbol_registry
|
||||
sec_core -.-> scheduler
|
||||
sec_core -.-> ingestion
|
||||
sec_core -.-> parser
|
||||
sec_core -.-> extractor
|
||||
sec_core -.-> aggregation
|
||||
sec_core -.-> recommendation
|
||||
sec_core -.-> trading_engine
|
||||
sec_core -.-> risk_engine
|
||||
sec_core -.-> broker_adapter
|
||||
sec_core -.-> lake_publisher
|
||||
sec_core -.-> hive_metastore
|
||||
sec_core -.-> trino
|
||||
sec_core -.-> superset
|
||||
|
||||
sec_broker -.-> ingestion
|
||||
sec_broker -.-> trading_engine
|
||||
sec_broker -.-> risk_engine
|
||||
sec_broker -.-> broker_adapter
|
||||
|
||||
sec_market -.-> ingestion
|
||||
sec_market -.-> query_api
|
||||
|
||||
sec_gmail -.-> trading_engine
|
||||
|
||||
sec_dashboard -.-> superset
|
||||
|
||||
configmap -.-> query_api
|
||||
configmap -.-> symbol_registry
|
||||
configmap -.-> scheduler
|
||||
configmap -.-> ingestion
|
||||
configmap -.-> parser
|
||||
configmap -.-> extractor
|
||||
configmap -.-> aggregation
|
||||
configmap -.-> recommendation
|
||||
configmap -.-> trading_engine
|
||||
configmap -.-> risk_engine
|
||||
configmap -.-> broker_adapter
|
||||
configmap -.-> lake_publisher
|
||||
configmap -.-> superset
|
||||
|
||||
%% ── Styles ────────────────────────────────────────────────────
|
||||
classDef apiSvc fill:#4a90d9,stroke:#2c5f8a,color:#fff
|
||||
classDef frontendSvc fill:#50c878,stroke:#2e7d46,color:#fff
|
||||
classDef tradingSvc fill:#e8a838,stroke:#b07d1a,color:#fff
|
||||
classDef processSvc fill:#9b59b6,stroke:#6c3483,color:#fff
|
||||
classDef orchSvc fill:#1abc9c,stroke:#148f77,color:#fff
|
||||
classDef ingestionSvc fill:#e67e22,stroke:#bf6516,color:#fff
|
||||
classDef analyticsSvc fill:#e74c3c,stroke:#a93226,color:#fff
|
||||
classDef supersetSvc fill:#c0392b,stroke:#96281b,color:#fff
|
||||
classDef extSvc fill:#95a5a6,stroke:#717d7e,color:#fff
|
||||
classDef secretSvc fill:#f5f5dc,stroke:#999,color:#333
|
||||
classDef configSvc fill:#dfe6e9,stroke:#999,color:#333
|
||||
|
||||
class query_api,symbol_registry apiSvc
|
||||
class dashboard frontendSvc
|
||||
class trading_engine,risk_engine,broker_adapter tradingSvc
|
||||
class scheduler orchSvc
|
||||
class ingestion ingestionSvc
|
||||
class parser,extractor,aggregation,recommendation processSvc
|
||||
class lake_publisher,hive_metastore,trino analyticsSvc
|
||||
class superset supersetSvc
|
||||
class postgres,redis,minio,ollama extSvc
|
||||
class sec_core,sec_broker,sec_market,sec_gmail,sec_dashboard secretSvc
|
||||
class configmap configSvc
|
||||
```
|
||||
|
||||
## Network Policy Boundaries
|
||||
|
||||
The Helm chart deploys a **default-deny-ingress** policy that blocks all inbound traffic to pods in the `stonks-oracle` namespace. Each service that needs inbound connections has an explicit allow policy:
|
||||
|
||||
```mermaid
|
||||
graph LR
|
||||
subgraph netpol ["Network Policies — stonks-oracle namespace"]
|
||||
direction TB
|
||||
|
||||
deny["🔒 default-deny-ingress<br/><i>Blocks ALL ingress to all pods</i>"]
|
||||
|
||||
subgraph allows ["Explicit Allow Rules"]
|
||||
direction TB
|
||||
|
||||
np_dash["allow-dashboard-ingress<br/>dashboard :8080<br/>← kube-system (Traefik)"]
|
||||
|
||||
np_api["allow-query-api-ingress<br/>query-api :8000<br/>← kube-system (Traefik)<br/>← dashboard pod"]
|
||||
|
||||
np_reg["allow-symbol-registry-ingress<br/>symbol-registry :8000<br/>← kube-system (Traefik)<br/>← dashboard pod"]
|
||||
|
||||
np_trade["allow-trading-engine-ingress<br/>trading-engine :8000<br/>← kube-system (Traefik)<br/>← query-api pod<br/>← dashboard pod<br/><i>Egress: PostgreSQL :5432,</i><br/><i>Redis :6379, HTTPS :443, SMTP :587</i>"]
|
||||
|
||||
np_risk["allow-risk-engine-ingress<br/>risk-engine :8000<br/>← broker-adapter pod<br/>← query-api pod<br/>← dashboard pod"]
|
||||
|
||||
np_superset["allow-superset-ingress<br/>superset :8088<br/>← kube-system (Traefik)"]
|
||||
|
||||
np_trino["allow-trino-ingress<br/>trino :8080<br/>← superset pod<br/>← query-api pod<br/>← kube-system (Traefik)"]
|
||||
|
||||
np_hive["allow-hive-metastore-ingress<br/>hive-metastore :9083<br/>← trino pod<br/>← lake-publisher pod"]
|
||||
|
||||
np_broker["deny-broker-adapter-ingress<br/>broker-adapter<br/><i>No inbound traffic allowed</i>"]
|
||||
end
|
||||
end
|
||||
|
||||
style deny fill:#e74c3c,stroke:#c0392b,color:#fff
|
||||
style np_broker fill:#e74c3c,stroke:#c0392b,color:#fff
|
||||
style np_dash fill:#2ecc71,stroke:#27ae60,color:#fff
|
||||
style np_api fill:#2ecc71,stroke:#27ae60,color:#fff
|
||||
style np_reg fill:#2ecc71,stroke:#27ae60,color:#fff
|
||||
style np_trade fill:#f39c12,stroke:#d68910,color:#fff
|
||||
style np_risk fill:#f39c12,stroke:#d68910,color:#fff
|
||||
style np_superset fill:#2ecc71,stroke:#27ae60,color:#fff
|
||||
style np_trino fill:#2ecc71,stroke:#27ae60,color:#fff
|
||||
style np_hive fill:#3498db,stroke:#2980b9,color:#fff
|
||||
```
|
||||
|
||||
### Services Without Ingress Policies (Pipeline Workers)
|
||||
|
||||
The following services have **no inbound network policy** — they are queue-driven workers that only make outbound connections to PostgreSQL, Redis, MinIO, and Ollama. The default-deny-ingress policy blocks any unsolicited inbound traffic:
|
||||
|
||||
| Service | Tier | Behavior |
|
||||
|---------|------|----------|
|
||||
| scheduler | orchestration | Polls DB, enqueues to Redis. Runs migrations + seed + backfill as init containers |
|
||||
| ingestion | ingestion | Reads from `stonks:queue:ingestion`, writes to DB/MinIO/Redis. Egress to Polygon.io/News APIs |
|
||||
| parser | processing | Reads from `stonks:queue:parsing`, writes to DB/Redis |
|
||||
| extractor | processing | Reads from `stonks:queue:extraction`, calls Ollama, writes to DB/Redis |
|
||||
| aggregation | processing | Reads from `stonks:queue:aggregation`, writes to DB/Redis |
|
||||
| recommendation | processing | Reads from `stonks:queue:recommendation`, writes to DB/Redis |
|
||||
| lake-publisher | analytics | Reads from `stonks:queue:lake_publish`, writes Parquet to MinIO |
|
||||
|
||||
## Service Tier Summary
|
||||
|
||||
| Tier | Services | Ingress? | Replicas | Pipeline-Gated? | Notes |
|
||||
|------|----------|----------|----------|-----------------|-------|
|
||||
| **api** | query-api, symbol-registry | Yes (Traefik) | 1 each | No | FastAPI, readiness probes on `/docs` |
|
||||
| **frontend** | dashboard | Yes (Traefik) | 1 | No | nginx-unprivileged on :8080, proxies to API services |
|
||||
| **trading** | trading-engine, risk-engine, broker-adapter | trading-engine: Yes; risk-engine: internal only; broker-adapter: denied | 1 each | broker-adapter only | trading-engine has egress to Alpaca + Gmail |
|
||||
| **orchestration** | scheduler | No | 1 | Yes | Runs DB migrations + seed + backfill as init containers |
|
||||
| **ingestion** | ingestion | No | 1 | Yes | Fetches from external APIs (Polygon.io, news, filings) |
|
||||
| **processing** | parser, extractor, aggregation, recommendation | No | 2, 1, 4, 1 | Yes | Queue-driven pipeline workers |
|
||||
| **analytics** | lake-publisher, trino, hive-metastore | trino: Yes (Traefik); others: No | 1 each | lake-publisher only | trino + hive-metastore gated by `trino.enabled` / `hiveMetastore.enabled` |
|
||||
| **dashboard** (Superset) | superset | Yes (Traefik) | 1 | No | Gated by `superset.enabled`, custom image with trino + psycopg2 drivers |
|
||||
|
||||
## Secret Consumption Map
|
||||
|
||||
| Secret | Keys | Consumers |
|
||||
|--------|------|-----------|
|
||||
| `stonks-core-secrets` | POSTGRES_PASSWORD, MINIO_ACCESS_KEY, MINIO_SECRET_KEY, REDIS_PASSWORD | All 13 app services + hive-metastore (init), trino (init), superset |
|
||||
| `stonks-broker-secrets` | BROKER_API_KEY, BROKER_API_SECRET, BROKER_BASE_URL | ingestion, trading-engine, risk-engine, broker-adapter |
|
||||
| `stonks-market-secrets` | MARKET_DATA_API_KEY | ingestion, query-api |
|
||||
| `stonks-gmail-secrets` | GMAIL_SENDER, GMAIL_RECIPIENT, GMAIL_APP_PASSWORD | trading-engine |
|
||||
| `stonks-dashboard-secrets` | SUPERSET_SECRET_KEY, SUPERSET_ADMIN_PASSWORD | superset |
|
||||
|
||||
## Pipeline Toggle
|
||||
|
||||
Setting `pipelineEnabled: false` in `values.yaml` scales all services with `pipeline: true` to 0 replicas. This affects:
|
||||
|
||||
- scheduler, ingestion, parser, extractor, aggregation, recommendation, broker-adapter, lake-publisher
|
||||
|
||||
API-tier services (query-api, symbol-registry), trading-tier services (trading-engine, risk-engine), analytics services (trino, hive-metastore, superset), and the dashboard always run regardless of this toggle.
|
||||
|
||||
## External Cluster Services
|
||||
|
||||
These services run outside the `stonks-oracle` namespace and are referenced via cross-namespace DNS:
|
||||
|
||||
| Service | Namespace | DNS | Port | Notes |
|
||||
|---------|-----------|-----|------|-------|
|
||||
| PostgreSQL | `postgresql-service` | `postgresql-rw.postgresql-service.svc.cluster.local` | 5432 | CloudNativePG managed |
|
||||
| Redis | `redis-service` | `redis-master.redis-service.svc.cluster.local` | 6379 | Password in `stonks-core-secrets` |
|
||||
| MinIO | `minio-service` | `minio.minio-service.svc.cluster.local` | 80 | S3-compatible object store |
|
||||
| Ollama | `ollama-service` | `ollama.ollama-service.svc.cluster.local` | 11434 | LLM inference, GPU: 4070 Ti Super 16GB |
|
||||
|
||||
## Analytics Plane
|
||||
|
||||
The analytics stack runs within the `stonks-oracle` namespace:
|
||||
|
||||
1. **Lake Publisher** writes Parquet fact tables to MinIO at `s3a://stonks-lakehouse/warehouse`. Pipeline-gated — scales to 0 when `pipelineEnabled: false`.
|
||||
2. **Hive Metastore** (Apache Hive 4.0.0) manages table metadata, backed by embedded Derby DB with a PVC (`hive-metastore-data`) for persistence. Connects to MinIO for S3A filesystem access. Gated by `hiveMetastore.enabled`.
|
||||
3. **Trino** queries the lakehouse via Hive Metastore (`thrift://hive-metastore:9083`). Exposes two catalogs: `lakehouse` (Hive connector) and `iceberg` (Iceberg connector). Both connect to MinIO for data access. Gated by `trino.enabled`. Readiness probe on `/v1/info`.
|
||||
4. **Superset** connects to Trino for lakehouse queries and to PostgreSQL for its metadata DB. Uses Redis for caching. Exposed externally via Traefik ingress. Gated by `superset.enabled`. Uses custom image (`registry.celestium.life/stonks-oracle/superset:latest`) with trino + psycopg2 drivers. PVC (`superset-data`) for persistence.
|
||||
|
||||
## Ingress Routes
|
||||
|
||||
All ingress resources use the `traefik` IngressClass with TLS certificates issued by the `ca-issuer` ClusterIssuer:
|
||||
|
||||
| Domain | Backend Service | Port | TLS Secret |
|
||||
|--------|----------------|------|------------|
|
||||
| `stonks.celestium.life` | dashboard | 8080 | `stonks-dashboard-tls` |
|
||||
| `stonks-api.celestium.life` | query-api | 8000 | `stonks-api-tls` |
|
||||
| `stonks-registry.celestium.life` | symbol-registry | 8000 | `stonks-registry-tls` |
|
||||
| `stonks-trading.celestium.life` | trading-engine | 8000 | `stonks-trading-tls` |
|
||||
| `stonks-dash.celestium.life` | superset | 8088 | `stonks-dash-tls` |
|
||||
| `stonks-trino.celestium.life` | trino | 8080 | `stonks-trino-tls` |
|
||||
|
||||
## Deployment Stages
|
||||
|
||||
The Helm chart supports multiple deployment stages via value override files:
|
||||
|
||||
| Stage | Override File | Namespace | Key Differences |
|
||||
|-------|--------------|-----------|-----------------|
|
||||
| **Production** | `values.yaml` (base) | `stonks-oracle` | Full analytics stack, all services |
|
||||
| **Paper** | `values-paper.yaml` | `stonks-oracle` | `BROKER_MODE=paper`, `DEPLOY_STAGE=paper`, separate DB (`stonks_paper`), Redis DB 2, paper-specific ingress hostnames |
|
||||
| **Beta** | `values-beta.yaml` | `stonks-oracle-beta` | `DEPLOY_STAGE=beta`, `LOG_LEVEL=DEBUG`, separate DB (`stonks_beta`), Redis DB 1, analytics stack disabled, beta-specific ingress hostnames |
|
||||
@@ -0,0 +1,440 @@
|
||||
# Backup and Restore Guide
|
||||
|
||||
This guide documents every backup and restore script in the Stonks Oracle platform, their CLI options, storage locations, retention policies, and procedures for disaster recovery.
|
||||
|
||||
## Overview
|
||||
|
||||
Stonks Oracle provides two tiers of backup tooling:
|
||||
|
||||
| Tier | Scripts | Scope | Storage |
|
||||
|------|---------|-------|---------|
|
||||
| **Local (kubectl-based)** | `backup-db.sh`, `restore-db.sh`, `backup-redis.sh` | Individual data stores, streamed to the operator's machine | `~/backups/stonks-oracle/` (local filesystem) |
|
||||
| **Cluster (Kubernetes Job)** | `backup.sh`, `restore.sh` | Full platform (PostgreSQL + all MinIO buckets) | NFS share at `192.168.42.8:/volume1/Kubernetes/stonks` |
|
||||
|
||||
All scripts live in the `scripts/` directory and require `kubectl` access to the cluster.
|
||||
|
||||
---
|
||||
|
||||
## Local Backup Scripts
|
||||
|
||||
### `backup-db.sh` — PostgreSQL Database Backup
|
||||
|
||||
Creates a compressed `pg_dump` of the `stonks` database and optionally uploads it to MinIO.
|
||||
|
||||
**Usage:**
|
||||
|
||||
```bash
|
||||
./scripts/backup-db.sh # backup to local file
|
||||
./scripts/backup-db.sh --upload-minio # backup + upload to MinIO
|
||||
```
|
||||
|
||||
**CLI Arguments:**
|
||||
|
||||
| Argument | Required | Description |
|
||||
|----------|----------|-------------|
|
||||
| `--upload-minio` | No | Upload the backup file to the `stonks-backups` MinIO bucket after creating it |
|
||||
|
||||
**Environment Variables:**
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `BACKUP_DIR` | `~/backups/stonks-oracle` | Local directory where backup files are stored |
|
||||
|
||||
**What it captures:**
|
||||
|
||||
- Full `pg_dump` of the `stonks` database (all tables, data, sequences)
|
||||
- Dump flags: `--no-owner --no-privileges --clean --if-exists`
|
||||
- Output format: gzip-compressed SQL (`.sql.gz`)
|
||||
|
||||
**How it works:**
|
||||
|
||||
1. Runs `pg_dump` inside the PostgreSQL pod (`postgresql-1` in `postgresql-service` namespace) and streams the compressed output to the local machine
|
||||
2. Validates the backup is non-empty and counts tables as a sanity check
|
||||
3. If `--upload-minio` is specified, attempts to create the `stonks-backups` bucket (if it doesn't exist) and stages the file for upload
|
||||
4. Prunes old backups, keeping only the last 7 files matching `stonks-*.sql.gz`
|
||||
|
||||
**Storage:**
|
||||
|
||||
- Local path: `~/backups/stonks-oracle/stonks-<YYYYMMDD-HHMMSS>.sql.gz`
|
||||
- MinIO bucket (optional): `stonks-backups`
|
||||
|
||||
**Retention:** Keeps the last 7 backups. Older files matching `stonks-*.sql.gz` in the backup directory are automatically deleted.
|
||||
|
||||
---
|
||||
|
||||
### `backup-redis.sh` — Redis State Backup
|
||||
|
||||
Triggers a Redis `BGSAVE` and copies the RDB dump file to the local machine.
|
||||
|
||||
**Usage:**
|
||||
|
||||
```bash
|
||||
./scripts/backup-redis.sh
|
||||
```
|
||||
|
||||
**CLI Arguments:** None.
|
||||
|
||||
**Environment Variables:**
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `BACKUP_DIR` | `~/backups/stonks-oracle` | Local directory where the RDB file is stored |
|
||||
| `REDIS_PASSWORD` | `PSCh4ng3me!` | Redis authentication password |
|
||||
|
||||
**What it captures:**
|
||||
|
||||
- Redis RDB snapshot (`dump.rdb`) containing all in-memory state: deduplication markers, queue contents, rate-limit counters, cached values
|
||||
|
||||
**How it works:**
|
||||
|
||||
1. Triggers `BGSAVE` on the Redis master pod (`redis-master-0` in `redis-service` namespace)
|
||||
2. Waits 5 seconds for the background save to complete, then logs the `LASTSAVE` timestamp
|
||||
3. Copies the RDB file from the pod. Tries `/data/dump.rdb` first, then falls back to `/var/lib/redis/dump.rdb` and `/bitnami/redis/data/dump.rdb`
|
||||
4. Prints Redis keyspace statistics for verification
|
||||
|
||||
**Storage:**
|
||||
|
||||
- Local path: `~/backups/stonks-oracle/redis-<YYYYMMDD-HHMMSS>.rdb`
|
||||
|
||||
**Retention:** No automatic pruning. Old Redis backups accumulate and must be cleaned up manually.
|
||||
|
||||
---
|
||||
|
||||
### `restore-db.sh` — PostgreSQL Database Restore
|
||||
|
||||
Restores a `pg_dump` backup into the `stonks` database with full service scale-down/scale-up.
|
||||
|
||||
**Usage:**
|
||||
|
||||
```bash
|
||||
./scripts/restore-db.sh <backup-file.sql.gz>
|
||||
./scripts/restore-db.sh ~/backups/stonks-oracle/stonks-20260415-180000.sql.gz
|
||||
```
|
||||
|
||||
If called without arguments, lists available backups in `~/backups/stonks-oracle/`.
|
||||
|
||||
**CLI Arguments:**
|
||||
|
||||
| Argument | Required | Description |
|
||||
|----------|----------|-------------|
|
||||
| `<backup-file.sql.gz>` | Yes | Path to the gzip-compressed SQL backup file to restore |
|
||||
|
||||
**What it restores:**
|
||||
|
||||
- All tables, data, sequences, and indexes in the `stonks` database
|
||||
- Re-grants `ALL PRIVILEGES` to the `stonks` user on all tables and sequences after restore
|
||||
|
||||
**Service scale-down/scale-up procedure:**
|
||||
|
||||
1. **Terminates active connections** — Runs `pg_terminate_backend()` for all connections to the `stonks` database
|
||||
2. **Scales down all deployments** in the `stonks-oracle` namespace to 0 replicas to prevent reconnections
|
||||
3. **Waits 10 seconds** for pods to terminate
|
||||
4. **Restores the backup** using `psql --single-transaction` (piped from `zcat`)
|
||||
5. **Re-grants permissions** to the `stonks` user
|
||||
6. **Verifies** the restore by counting tables
|
||||
7. **Scales all deployments back to 1 replica**, then scales `ingestion` and `parser` to 2 replicas
|
||||
|
||||
**Data loss implications:**
|
||||
|
||||
> **WARNING:** This replaces ALL data in the `stonks` database with the backup contents. Any data written after the backup was taken is permanently lost. The script requires interactive confirmation — you must type `yes` to proceed.
|
||||
|
||||
---
|
||||
|
||||
## Cluster Backup Scripts (Kubernetes Jobs)
|
||||
|
||||
### `backup.sh` — Full Platform Backup (PostgreSQL + MinIO)
|
||||
|
||||
Runs a Kubernetes Job that backs up both PostgreSQL and all MinIO buckets to an NFS share.
|
||||
|
||||
**Usage:**
|
||||
|
||||
```bash
|
||||
bash scripts/backup.sh
|
||||
```
|
||||
|
||||
**CLI Arguments:** None.
|
||||
|
||||
**What it captures:**
|
||||
|
||||
- **PostgreSQL**: Full `pg_dump` in custom format (`-Fc`) as `stonks.pgdump`
|
||||
- **MinIO buckets** (8 buckets mirrored):
|
||||
- `stonks-raw-market` — Raw market data from Polygon.io
|
||||
- `stonks-raw-news` — Raw news articles
|
||||
- `stonks-raw-filings` — Raw SEC filings
|
||||
- `stonks-normalized` — Normalized documents
|
||||
- `stonks-llm-prompts` — LLM prompt logs
|
||||
- `stonks-llm-results` — LLM extraction results
|
||||
- `stonks-lakehouse` — Parquet fact tables for Trino
|
||||
- `stonks-audit` — Audit trail artifacts
|
||||
- **Manifest**: `manifest.json` with backup name, timestamp, and bucket list
|
||||
|
||||
**How it works:**
|
||||
|
||||
1. Deletes any previous `stonks-backup` Job
|
||||
2. Creates a Kubernetes Job using `postgres:18-alpine` with NFS volume mount and MinIO credentials from cluster secrets
|
||||
3. Inside the Job container:
|
||||
- Runs `pg_dump` with credentials from `stonks-config` ConfigMap and `stonks-core-secrets` Secret
|
||||
- Installs the MinIO client (`mc`) and mirrors each bucket to the NFS backup directory
|
||||
- Writes a `manifest.json` and updates the `latest` symlink
|
||||
4. Waits up to 600 seconds (10 minutes) for the Job to complete
|
||||
5. Job auto-cleans after 300 seconds (`ttlSecondsAfterFinished`)
|
||||
|
||||
**Storage:**
|
||||
|
||||
- NFS path: `192.168.42.8:/volume1/Kubernetes/stonks/<backup-name>/`
|
||||
- Directory structure:
|
||||
```
|
||||
stonks-backup-YYYYMMDD-HHMMSS/
|
||||
├── stonks.pgdump # PostgreSQL custom-format dump
|
||||
├── manifest.json # Backup metadata
|
||||
└── minio/
|
||||
├── stonks-raw-market/ # Mirrored bucket contents
|
||||
├── stonks-raw-news/
|
||||
├── stonks-raw-filings/
|
||||
├── stonks-normalized/
|
||||
├── stonks-llm-prompts/
|
||||
├── stonks-llm-results/
|
||||
├── stonks-lakehouse/
|
||||
└── stonks-audit/
|
||||
```
|
||||
- A `latest` symlink always points to the most recent backup
|
||||
|
||||
**Retention:** No automatic pruning on NFS. Old backups must be cleaned up manually.
|
||||
|
||||
---
|
||||
|
||||
### `restore.sh` — Full Platform Restore (PostgreSQL + MinIO)
|
||||
|
||||
Runs a Kubernetes Job that restores both PostgreSQL and MinIO buckets from an NFS backup.
|
||||
|
||||
**Usage:**
|
||||
|
||||
```bash
|
||||
bash scripts/restore.sh # restore from "latest" symlink
|
||||
bash scripts/restore.sh <backup-name> # restore a specific backup
|
||||
```
|
||||
|
||||
**CLI Arguments:**
|
||||
|
||||
| Argument | Required | Description |
|
||||
|----------|----------|-------------|
|
||||
| `<backup-name>` | No | Name of the backup directory on NFS. Defaults to `latest` (symlink to most recent backup) |
|
||||
|
||||
**What it restores:**
|
||||
|
||||
- **PostgreSQL**: Full database restore using `pg_restore --clean --if-exists --no-owner --no-acl`
|
||||
- **MinIO buckets**: All 8 buckets mirrored back with `mc mirror --overwrite`
|
||||
|
||||
**How it works:**
|
||||
|
||||
1. Prints a warning and gives 5 seconds to abort (Ctrl+C)
|
||||
2. Deletes any previous `stonks-restore` Job
|
||||
3. Creates a Kubernetes Job that:
|
||||
- Validates the backup exists (`stonks.pgdump` file present)
|
||||
- Restores PostgreSQL using `pg_restore` with `--clean` (drops and recreates objects)
|
||||
- Installs `mc` and mirrors each bucket back from NFS to MinIO
|
||||
- Verifies the restore by querying row counts for key tables (companies, documents, intelligence, impacts, trends, recommendations)
|
||||
4. Waits up to 600 seconds for the Job to complete
|
||||
|
||||
**Data loss implications:**
|
||||
|
||||
> **WARNING:** This will DROP and recreate all objects in the `stonks` database. All MinIO bucket contents are overwritten. Any data written after the backup was taken is permanently lost. The script provides a 5-second abort window before proceeding.
|
||||
|
||||
**Post-restore steps:**
|
||||
|
||||
After the restore completes, restart all services to pick up the restored state:
|
||||
|
||||
```bash
|
||||
kubectl rollout restart deployment -n stonks-oracle --all
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## MinIO Upload Option (`--upload-minio`)
|
||||
|
||||
The `backup-db.sh` script supports `--upload-minio` for off-host storage of database backups. When enabled:
|
||||
|
||||
1. The script connects to MinIO through an ingestion pod in the `stonks-oracle` namespace
|
||||
2. Creates the `stonks-backups` bucket if it doesn't already exist
|
||||
3. Stages the backup file for upload
|
||||
|
||||
This provides a second copy of the database backup on object storage, separate from the operator's local filesystem. The full cluster backup (`backup.sh`) stores backups on NFS and does not use this flag — it backs up MinIO bucket *contents* rather than uploading database dumps *to* MinIO.
|
||||
|
||||
---
|
||||
|
||||
## Full Nuke and Rebuild Procedure
|
||||
|
||||
When a complete platform reset is needed (corrupted state, major schema changes, fresh start), follow this procedure:
|
||||
|
||||
### Step 1: Tear Down Services
|
||||
|
||||
```bash
|
||||
bash ~/sources/kube/stonks-oracle/runmelast.sh
|
||||
```
|
||||
|
||||
This runs from `gremlin-1` and performs a Helm uninstall, cleaning up all Kubernetes resources in the `stonks-oracle` namespace. Database, MinIO, and Redis data are preserved (they run in separate namespaces).
|
||||
|
||||
### Step 2: Terminate Database Connections
|
||||
|
||||
```bash
|
||||
kubectl exec -n postgresql-service postgresql-1 -c postgres -- \
|
||||
psql -U postgres -c \
|
||||
"SELECT pg_terminate_backend(pid) FROM pg_stat_activity WHERE datname = 'stonks' AND pid <> pg_backend_pid();"
|
||||
```
|
||||
|
||||
### Step 3: Drop the Database
|
||||
|
||||
```bash
|
||||
kubectl exec -n postgresql-service postgresql-1 -c postgres -- \
|
||||
psql -U postgres -c "DROP DATABASE IF EXISTS stonks;"
|
||||
```
|
||||
|
||||
### Step 4: Flush Redis
|
||||
|
||||
Clear all `stonks:*` keys to reset deduplication markers, queue contents, and cached state:
|
||||
|
||||
```bash
|
||||
kubectl exec -n redis-service redis-master-0 -- \
|
||||
redis-cli -a 'PSCh4ng3me!' --scan --pattern 'stonks:*' | \
|
||||
xargs -L 100 kubectl exec -n redis-service redis-master-0 -- \
|
||||
redis-cli -a 'PSCh4ng3me!' DEL
|
||||
```
|
||||
|
||||
### Step 5: Redeploy
|
||||
|
||||
```bash
|
||||
bash ~/sources/kube/stonks-oracle/runmefirst.sh
|
||||
```
|
||||
|
||||
This runs from `gremlin-1` and performs:
|
||||
- Database creation and migration (all `infra/migrations/*.sql` files applied in order)
|
||||
- Helm install with secrets injected via `--set` flags
|
||||
- Rolling restart of all deployments
|
||||
|
||||
### Step 6: Re-seed the Symbol Registry
|
||||
|
||||
```bash
|
||||
POSTGRES_HOST=postgresql-rw.postgresql-service.svc.cluster.local \
|
||||
POSTGRES_PASSWORD='St0nks0racl3!' \
|
||||
POSTGRES_USER=stonks \
|
||||
POSTGRES_DB=stonks \
|
||||
.venv/bin/python -m services.symbol_registry.seed
|
||||
```
|
||||
|
||||
This populates the 50 tracked companies across 10 sectors and 46 competitor relationships.
|
||||
|
||||
---
|
||||
|
||||
## Recommended Backup Schedules
|
||||
|
||||
### Daily Database Backup (cron)
|
||||
|
||||
Run `backup-db.sh` daily on a machine with `kubectl` access. The built-in retention keeps the last 7 backups automatically.
|
||||
|
||||
```cron
|
||||
# Daily database backup at 2:00 AM
|
||||
0 2 * * * /path/to/stonks-oracle/scripts/backup-db.sh --upload-minio >> /var/log/stonks-backup.log 2>&1
|
||||
```
|
||||
|
||||
### Weekly Full Backup (cron)
|
||||
|
||||
Run the full cluster backup weekly to capture both PostgreSQL and MinIO data on NFS:
|
||||
|
||||
```cron
|
||||
# Weekly full backup (PostgreSQL + MinIO) on Sundays at 3:00 AM
|
||||
0 3 * * 0 /path/to/stonks-oracle/scripts/backup.sh >> /var/log/stonks-full-backup.log 2>&1
|
||||
```
|
||||
|
||||
### Redis Backup Before Deployments
|
||||
|
||||
Redis state is transient (queues, dedup markers, caches) and rebuilds naturally. Back up Redis before major deployments or database resets as a precaution:
|
||||
|
||||
```bash
|
||||
./scripts/backup-redis.sh
|
||||
```
|
||||
|
||||
### Kubernetes CronJobs
|
||||
|
||||
For fully automated in-cluster backups, create a CronJob based on the same Job spec used by `backup.sh`:
|
||||
|
||||
```yaml
|
||||
apiVersion: batch/v1
|
||||
kind: CronJob
|
||||
metadata:
|
||||
name: stonks-backup
|
||||
namespace: stonks-oracle
|
||||
spec:
|
||||
schedule: "0 2 * * *" # Daily at 2:00 AM UTC
|
||||
concurrencyPolicy: Forbid
|
||||
successfulJobsHistoryLimit: 3
|
||||
failedJobsHistoryLimit: 3
|
||||
jobTemplate:
|
||||
spec:
|
||||
ttlSecondsAfterFinished: 3600
|
||||
backoffLimit: 1
|
||||
template:
|
||||
spec:
|
||||
restartPolicy: Never
|
||||
volumes:
|
||||
- name: nfs-backup
|
||||
nfs:
|
||||
server: 192.168.42.8
|
||||
path: /volume1/Kubernetes/stonks
|
||||
containers:
|
||||
- name: backup
|
||||
image: postgres:18-alpine
|
||||
volumeMounts:
|
||||
- name: nfs-backup
|
||||
mountPath: /backup
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
name: stonks-config
|
||||
- secretRef:
|
||||
name: stonks-core-secrets
|
||||
env:
|
||||
- name: MINIO_ACCESS_KEY
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: stonks-core-secrets
|
||||
key: MINIO_ACCESS_KEY
|
||||
- name: MINIO_SECRET_KEY
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: stonks-core-secrets
|
||||
key: MINIO_SECRET_KEY
|
||||
command: ["sh", "-c"]
|
||||
args:
|
||||
- |
|
||||
set -e
|
||||
apk add --no-cache curl ca-certificates
|
||||
STAMP="stonks-backup-$(date +%Y%m%d-%H%M%S)"
|
||||
DIR="/backup/${STAMP}"
|
||||
mkdir -p "${DIR}/minio"
|
||||
|
||||
# PostgreSQL backup
|
||||
PGPASSWORD="${POSTGRES_PASSWORD}" pg_dump \
|
||||
-h "${POSTGRES_HOST}" -p "${POSTGRES_PORT}" \
|
||||
-U "${POSTGRES_USER}" -d "${POSTGRES_DB}" \
|
||||
--no-owner --no-acl -Fc \
|
||||
-f "${DIR}/stonks.pgdump"
|
||||
|
||||
# MinIO backup
|
||||
curl -sL https://dl.min.io/client/mc/release/linux-amd64/mc -o /usr/local/bin/mc
|
||||
chmod +x /usr/local/bin/mc
|
||||
mc alias set backup "http://${MINIO_ENDPOINT}" "${MINIO_ACCESS_KEY}" "${MINIO_SECRET_KEY}" --api S3v4
|
||||
|
||||
for bucket in stonks-raw-market stonks-raw-news stonks-raw-filings stonks-normalized stonks-llm-prompts stonks-llm-results stonks-lakehouse stonks-audit; do
|
||||
mc mirror "backup/${bucket}" "${DIR}/minio/${bucket}/" 2>/dev/null || true
|
||||
done
|
||||
|
||||
ln -sfn "${STAMP}" /backup/latest
|
||||
echo "Backup complete: ${DIR}"
|
||||
```
|
||||
|
||||
### Recommended Schedule Summary
|
||||
|
||||
| What | Frequency | Script | Retention |
|
||||
|------|-----------|--------|-----------|
|
||||
| Database only | Daily | `backup-db.sh --upload-minio` | Last 7 (auto-pruned) |
|
||||
| Full platform (DB + MinIO) | Weekly | `backup.sh` | Manual cleanup on NFS |
|
||||
| Redis snapshot | Before deployments | `backup-redis.sh` | Manual cleanup |
|
||||
@@ -0,0 +1,896 @@
|
||||
# Docker Deployment Guide
|
||||
|
||||
This guide covers running the full Stonks Oracle platform locally using Docker Compose. It documents every service, environment variable, volume mount, health check, and operational command.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Docker Engine 24+ and Docker Compose v2
|
||||
- NVIDIA GPU with drivers and NVIDIA Container Toolkit (for Ollama LLM inference)
|
||||
- At least 16 GB RAM (Ollama + Trino + all services)
|
||||
- API keys for Polygon.io and Alpaca (optional — platform runs in degraded mode without them)
|
||||
|
||||
## Quick Start
|
||||
|
||||
```bash
|
||||
# 1. Clone the repository
|
||||
git clone <repo-url> && cd stonks-oracle
|
||||
|
||||
# 2. Configure API keys (create .env in the repo root)
|
||||
cat > .env <<'EOF'
|
||||
MARKET_DATA_API_KEY=your_polygon_key
|
||||
BROKER_API_KEY=your_alpaca_key
|
||||
BROKER_API_SECRET=your_alpaca_secret
|
||||
BROKER_BASE_URL=https://paper-api.alpaca.markets
|
||||
EOF
|
||||
|
||||
# 3. Start everything
|
||||
docker compose up -d
|
||||
|
||||
# 4. Pull an LLM model into Ollama
|
||||
docker compose exec ollama ollama pull qwen3.5:9b-fast
|
||||
|
||||
# 5. Seed the database
|
||||
docker compose exec scheduler python -m services.symbol_registry.seed
|
||||
|
||||
# 6. Verify all services are healthy
|
||||
docker compose ps
|
||||
|
||||
# 7. Access the dashboard
|
||||
open http://localhost:3000
|
||||
```
|
||||
|
||||
### Automated Deployment
|
||||
|
||||
The `deploy-docker.sh` script automates the full deployment to a remote host via SSH, including prerequisite installation, repository sync, environment configuration, image builds, service startup, database seeding, and Ollama model pulling:
|
||||
|
||||
```bash
|
||||
# Deploy with defaults (GPU-accelerated Docker Ollama)
|
||||
bash deploy-docker.sh
|
||||
|
||||
# Specify a custom Ollama model
|
||||
bash deploy-docker.sh --ollama-model qwen3.6
|
||||
|
||||
# Deploy to a different host
|
||||
bash deploy-docker.sh --host user@myserver --dir /opt/stonks
|
||||
```
|
||||
|
||||
| Flag | Default | Description |
|
||||
|------|---------|-------------|
|
||||
| `--host` | `celes@192.168.42.254` | SSH target (`USER@HOST`) |
|
||||
| `--ollama-url` | (auto — Docker container) | Ollama API URL |
|
||||
| `--ollama-model` | `qwen3.5:9b-fast` | Ollama model to pull |
|
||||
| `--dir` | `~/stonks-oracle` | Remote install directory |
|
||||
|
||||
The script detects the target OS and package manager (apt, dnf, yum, pacman, zypper) and installs Docker, NVIDIA drivers, and the NVIDIA Container Toolkit as needed. It also handles WSL environments and firewall configuration.
|
||||
|
||||
---
|
||||
|
||||
## Service Inventory
|
||||
|
||||
### Infrastructure Services
|
||||
|
||||
| Service | Image | Ports | Volumes | Purpose |
|
||||
|---------|-------|-------|---------|---------|
|
||||
| `postgres` | `postgres:16-alpine` | `5432:5432` | `pgdata` → `/var/lib/postgresql/data`, `./infra/migrations` → `/docker-entrypoint-initdb.d` | Primary database; migrations auto-applied on first start |
|
||||
| `redis` | `redis:7-alpine` | `6379:6379` | — | Queue broker, caching, deduplication |
|
||||
| `minio` | `minio/minio:latest` | `9000:9000` (API), `9001:9001` (console) | `miniodata` → `/data` | Object storage for raw artifacts and lakehouse |
|
||||
| `minio-init` | `minio/mc:latest` | — | — | One-shot init container that creates required buckets |
|
||||
| `ollama` | `ollama/ollama:latest` | `11434:11434` | `ollama_models` → `/root/.ollama` | LLM inference server for extraction and classification |
|
||||
| `trino` | `trinodb/trino:latest` | `8080:8080` | `./infra/trino/catalog` → `/etc/trino/catalog` | SQL query engine over the lakehouse |
|
||||
| `hive-metastore` | `apache/hive:4.0.0` | `9083:9083` | `hive_data` → `/opt/hive/data`, `./infra/hive/core-site.xml` → `/opt/hive/conf/core-site.xml`, `./infra/hive/metastore-site.xml` → `/opt/hive/conf/metastore-site.xml` | Iceberg/Hive metadata catalog for Trino |
|
||||
| `superset` | `apache/superset:latest` | `8088:8088` | `superset_data` → `/app/superset_home` | BI dashboards over Trino |
|
||||
|
||||
### Application Services
|
||||
|
||||
| Service | Dockerfile | `SERVICE_CMD` / Command | Ports | Depends On |
|
||||
|---------|-----------|------------------------|-------|------------|
|
||||
| `scheduler` | `docker/Dockerfile.scheduler` | `python -m services.scheduler.app` | — | postgres (healthy), redis (healthy) |
|
||||
| `symbol-registry` | `docker/Dockerfile` | `uvicorn services.symbol_registry.app:app --host 0.0.0.0 --port 8000` | `8001:8000` | postgres (healthy) |
|
||||
| `ingestion` | `docker/Dockerfile` | `python -m services.ingestion.worker` | — | postgres (healthy), redis (healthy), minio (healthy) |
|
||||
| `parser` | `docker/Dockerfile` | `python -m services.parser.worker` | — | postgres (healthy), redis (healthy) |
|
||||
| `extractor` | `docker/Dockerfile` | `python -m services.extractor.main` | — | postgres (healthy), redis (healthy), ollama (started) |
|
||||
| `aggregation` | `docker/Dockerfile` | `python -m services.aggregation.main` | — | postgres (healthy), redis (healthy) |
|
||||
| `recommendation` | `docker/Dockerfile` | `python -m services.recommendation.main` | — | postgres (healthy), redis (healthy) |
|
||||
| `trading-engine` | `docker/Dockerfile` | `uvicorn services.trading.app:app --host 0.0.0.0 --port 8000` | `8002:8000` | postgres (healthy), redis (healthy) |
|
||||
| `risk-engine` | `docker/Dockerfile` | `uvicorn services.risk.app:app --host 0.0.0.0 --port 8000` | `8003:8000` | postgres (healthy) |
|
||||
| `broker-adapter` | `docker/Dockerfile` | `python -m services.adapters.broker_service` | — | postgres (healthy), redis (healthy) |
|
||||
| `lake-publisher` | `docker/Dockerfile` | `python -m services.lake_publisher.jobs` | — | postgres (healthy), minio (healthy) |
|
||||
| `query-api` | `docker/Dockerfile` | `uvicorn services.api.app:app --host 0.0.0.0 --port 8000` | `8004:8000` | postgres (healthy), redis (healthy), minio (healthy) |
|
||||
| `dashboard` | `frontend/Dockerfile` | nginx (built-in) | `3000:8080` | query-api (healthy) |
|
||||
|
||||
The `risk-engine` service has a Docker network alias of `risk` so the dashboard's nginx reverse proxy can resolve it as `http://risk:8000`.
|
||||
|
||||
### Port Summary
|
||||
|
||||
| Port | Service | Protocol |
|
||||
|------|---------|----------|
|
||||
| 3000 | Dashboard (React UI) | HTTP |
|
||||
| 5432 | PostgreSQL | TCP |
|
||||
| 6379 | Redis | TCP |
|
||||
| 8001 | Symbol Registry API | HTTP |
|
||||
| 8002 | Trading Engine API | HTTP |
|
||||
| 8003 | Risk Engine API | HTTP |
|
||||
| 8004 | Query API | HTTP |
|
||||
| 8080 | Trino | HTTP |
|
||||
| 8088 | Superset | HTTP |
|
||||
| 9000 | MinIO API | HTTP |
|
||||
| 9001 | MinIO Console | HTTP |
|
||||
| 9083 | Hive Metastore | Thrift |
|
||||
| 11434 | Ollama | HTTP |
|
||||
|
||||
---
|
||||
|
||||
## Environment Variables
|
||||
|
||||
### Shared Application Environment (`x-app-env`)
|
||||
|
||||
All application services inherit these variables via the `x-app-env` YAML anchor:
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `POSTGRES_HOST` | `postgres` | PostgreSQL hostname (Docker service name) |
|
||||
| `POSTGRES_PORT` | `5432` | PostgreSQL port |
|
||||
| `POSTGRES_DB` | `stonks` | Database name |
|
||||
| `POSTGRES_USER` | `stonks` | Database user |
|
||||
| `POSTGRES_PASSWORD` | `stonks_dev` | Database password |
|
||||
| `REDIS_HOST` | `redis` | Redis hostname (Docker service name) |
|
||||
| `REDIS_PORT` | `6379` | Redis port |
|
||||
| `MINIO_ENDPOINT` | `minio:9000` | MinIO API endpoint |
|
||||
| `MINIO_ACCESS_KEY` | `minioadmin` | MinIO access key |
|
||||
| `MINIO_SECRET_KEY` | `minioadmin` | MinIO secret key |
|
||||
| `OLLAMA_BASE_URL` | `http://ollama:11434` | Ollama LLM server URL |
|
||||
|
||||
### `.env` File
|
||||
|
||||
The `.env` file is loaded by `ingestion`, `broker-adapter`, and `trading-engine` via the `env_file` directive. Create it in the repository root:
|
||||
|
||||
```dotenv
|
||||
# Stonks Oracle — Environment Variables
|
||||
# Loaded by: ingestion, broker-adapter, trading-engine
|
||||
|
||||
# ── Required for live data ingestion ──
|
||||
MARKET_DATA_API_KEY=
|
||||
|
||||
# ── Required for paper/live trading ──
|
||||
BROKER_API_KEY=
|
||||
BROKER_API_SECRET=
|
||||
BROKER_BASE_URL=https://paper-api.alpaca.markets
|
||||
|
||||
# ── Trading engine settings (optional) ──
|
||||
TRADING_ENABLED=true
|
||||
TRADING_RISK_TIER=moderate
|
||||
TRADING_MAX_OPEN_POSITIONS=15
|
||||
|
||||
# ── LLM model (optional) ──
|
||||
OLLAMA_MODEL=qwen3.5:9b-fast
|
||||
|
||||
# ── Signal layers (optional) ──
|
||||
MACRO_ENABLED=true
|
||||
COMPETITIVE_ENABLED=true
|
||||
```
|
||||
|
||||
| Variable | Required | Default | Used By | Description |
|
||||
|----------|----------|---------|---------|-------------|
|
||||
| `MARKET_DATA_API_KEY` | No* | (empty) | ingestion | Polygon.io API key for market data fetching |
|
||||
| `BROKER_API_KEY` | No* | (empty) | broker-adapter, trading-engine | Alpaca API key |
|
||||
| `BROKER_API_SECRET` | No* | (empty) | broker-adapter, trading-engine | Alpaca API secret |
|
||||
| `BROKER_BASE_URL` | No | `https://paper-api.alpaca.markets` | broker-adapter, trading-engine | Alpaca API base URL |
|
||||
|
||||
*Services start without these keys but run in degraded mode — ingestion cannot fetch market data and the broker adapter cannot execute trades.
|
||||
|
||||
### Infrastructure Service Environment
|
||||
|
||||
**PostgreSQL** (`postgres`):
|
||||
|
||||
| Variable | Value | Description |
|
||||
|----------|-------|-------------|
|
||||
| `POSTGRES_DB` | `stonks` | Database created on first start |
|
||||
| `POSTGRES_USER` | `stonks` | Superuser for the database |
|
||||
| `POSTGRES_PASSWORD` | `stonks_dev` | Password for the database user |
|
||||
|
||||
**MinIO** (`minio`):
|
||||
|
||||
| Variable | Value | Description |
|
||||
|----------|-------|-------------|
|
||||
| `MINIO_ROOT_USER` | `minioadmin` | MinIO admin username |
|
||||
| `MINIO_ROOT_PASSWORD` | `minioadmin` | MinIO admin password |
|
||||
|
||||
**Trino** (`trino`):
|
||||
|
||||
| Variable | Value | Description |
|
||||
|----------|-------|-------------|
|
||||
| `MINIO_ACCESS_KEY` | `minioadmin` | Passed to Trino for MinIO catalog access |
|
||||
| `MINIO_SECRET_KEY` | `minioadmin` | Passed to Trino for MinIO catalog access |
|
||||
|
||||
**Hive Metastore** (`hive-metastore`):
|
||||
|
||||
| Variable | Value | Description |
|
||||
|----------|-------|-------------|
|
||||
| `SERVICE_NAME` | `metastore` | Tells Hive to run in metastore-only mode |
|
||||
| `DB_DRIVER` | `derby` | Embedded Derby database for metadata |
|
||||
|
||||
**Superset** (`superset`):
|
||||
|
||||
| Variable | Value | Description |
|
||||
|----------|-------|-------------|
|
||||
| `SUPERSET_SECRET_KEY` | `stonks-dev-secret-key-change-me` | Flask secret key (change in production) |
|
||||
| `ADMIN_USERNAME` | `admin` | Initial admin username |
|
||||
| `ADMIN_PASSWORD` | `admin` | Initial admin password |
|
||||
| `ADMIN_EMAIL` | `admin@stonks.local` | Initial admin email |
|
||||
|
||||
### Additional Configuration Variables
|
||||
|
||||
All application services support additional environment variables loaded via `services/shared/config.py`. These can be added to individual service `environment` blocks or to the `x-app-env` anchor as needed:
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `REDIS_DB` | `0` | Redis database number |
|
||||
| `REDIS_PASSWORD` | (none) | Redis password (not needed in Docker Compose) |
|
||||
| `MINIO_SECURE` | `false` | Use HTTPS for MinIO |
|
||||
| `OLLAMA_MODEL` | `qwen3.5:9b` | Default LLM model for extraction |
|
||||
| `OLLAMA_TIMEOUT` | `120` | Ollama request timeout (seconds) |
|
||||
| `OLLAMA_MAX_RETRIES` | `2` | Max retries for Ollama requests |
|
||||
| `OLLAMA_RETRY_BASE_DELAY` | `1.0` | Base delay between retries (seconds) |
|
||||
| `OLLAMA_RETRY_MAX_DELAY` | `10.0` | Maximum delay between retries (seconds) |
|
||||
| `OLLAMA_RETRY_BACKOFF_MULTIPLIER` | `2.0` | Backoff multiplier for retries |
|
||||
| `VLLM_BASE_URL` | `http://192.168.42.254:8000` | vLLM server URL (if using vLLM instead of Ollama) |
|
||||
| `VLLM_MODEL` | `RedHatAI/Qwen3.6-35B-A3B-NVFP4` | vLLM model name |
|
||||
| `VLLM_TIMEOUT` | `120` | vLLM request timeout (seconds) |
|
||||
| `VLLM_MAX_RETRIES` | `2` | Max retries for vLLM requests |
|
||||
| `VLLM_TEMPERATURE` | `0.7` | vLLM sampling temperature |
|
||||
| `VLLM_MAX_TOKENS` | `4096` | vLLM max output tokens |
|
||||
| `VLLM_API_KEY` | (empty) | vLLM API key (if required) |
|
||||
| `TRINO_HOST` | `localhost` | Trino hostname |
|
||||
| `TRINO_PORT` | `8080` | Trino port |
|
||||
| `TRINO_CATALOG` | `lakehouse` | Trino catalog name |
|
||||
| `TRINO_SCHEMA` | `stonks` | Trino schema name |
|
||||
| `TRINO_ICEBERG_CATALOG` | `iceberg` | Trino Iceberg catalog name |
|
||||
| `MARKET_DATA_BASE_URL` | `https://api.polygon.io` | Polygon.io base URL |
|
||||
| `MARKET_DATA_PROVIDER` | `polygon` | Market data provider |
|
||||
| `BROKER_MODE` | `paper` | Broker mode: `paper` or `live` |
|
||||
| `BROKER_PROVIDER` | `alpaca` | Broker provider |
|
||||
| `TRADING_ENABLED` | `false` | Enable autonomous trading engine |
|
||||
| `TRADING_RISK_TIER` | `moderate` | Risk tier: `conservative`, `moderate`, `aggressive` |
|
||||
| `TRADING_POLLING_INTERVAL_SECONDS` | `60` | Recommendation polling interval |
|
||||
| `TRADING_MAX_OPEN_POSITIONS` | `10` | Maximum concurrent open positions |
|
||||
| `TRADING_RESERVE_SIPHON_PCT` | `0.20` | Percentage of profits siphoned to reserve pool |
|
||||
| `TRADING_STOP_LOSS_CHECK_INTERVAL_SECONDS` | `300` | Stop-loss check interval |
|
||||
| `TRADING_FAST_STOP_LOSS_INTERVAL_SECONDS` | `60` | Fast stop-loss check interval |
|
||||
| `TRADING_GRADUAL_ENTRY_TRANCHES` | `3` | Number of tranches for gradual entry |
|
||||
| `TRADING_GRADUAL_ENTRY_THRESHOLD_DOLLARS` | `30.0` | Dollar threshold for gradual entry |
|
||||
| `TRADING_ABSOLUTE_POSITION_CAP` | `50.0` | Maximum position size (dollars) |
|
||||
| `TRADING_ACTIVE_POOL_MINIMUM` | `100.0` | Minimum active pool balance |
|
||||
| `TRADING_EMERGENCY_DRAWDOWN_THRESHOLD_PCT` | `0.40` | Emergency drawdown threshold |
|
||||
| `TRADING_RESERVE_HIGH_WATER_PCT` | `0.30` | Reserve high-water mark percentage |
|
||||
| `TRADING_MICRO_TRADING_ENABLED` | `false` | Enable micro-trading mode |
|
||||
| `TRADING_MICRO_TRADING_INTERVAL_SECONDS` | `300` | Micro-trading polling interval |
|
||||
| `TRADING_MICRO_TRADING_ALLOCATION_CAP_PCT` | `0.03` | Micro-trading allocation cap |
|
||||
| `TRADING_MICRO_TRADING_MAX_DAILY` | `10` | Max micro-trades per day |
|
||||
| `TRADING_MICRO_TRADING_MAX_HOLD_MINUTES` | `120` | Max micro-trade hold time |
|
||||
| `TRADING_SNS_TOPIC_ARN` | (empty) | AWS SNS topic ARN for notifications |
|
||||
| `TRADING_SNS_PHONE_NUMBER` | (empty) | Phone number for SNS notifications |
|
||||
| `TRADING_GMAIL_SENDER` | (empty) | Gmail sender address for notifications |
|
||||
| `TRADING_GMAIL_RECIPIENT` | (empty) | Gmail recipient address for notifications |
|
||||
| `MACRO_ENABLED` | `true` | Enable macro signal layer |
|
||||
| `MACRO_SIGNAL_WEIGHT` | `0.3` | Relative weight of macro vs company signals |
|
||||
| `MACRO_CONFIDENCE_THRESHOLD` | `0.4` | Minimum confidence for macro event inclusion |
|
||||
| `MACRO_SHORT_TERM_STALENESS_HOURS` | `48` | Hours before short-term events get accelerated decay |
|
||||
| `PROJECTION_CONFIDENCE_THRESHOLD` | `0.3` | Minimum confidence for projections to influence recommendations |
|
||||
| `COMPETITIVE_ENABLED` | `true` | Enable competitive signal layer |
|
||||
| `COMPETITIVE_SIGNAL_WEIGHT` | `0.2` | Relative weight of competitive signals |
|
||||
| `COMPETITIVE_PATTERN_CONFIDENCE_THRESHOLD` | `0.3` | Minimum confidence for pattern inclusion |
|
||||
| `COMPETITIVE_PROPAGATION_STRENGTH_THRESHOLD` | `0.2` | Minimum strength for signal propagation |
|
||||
| `COMPETITIVE_ROUTINE_LOOKBACK_DAYS` | `180` | Lookback window for routine patterns |
|
||||
| `COMPETITIVE_MAJOR_DECISION_LOOKBACK_DAYS` | `365` | Lookback window for major decisions |
|
||||
| `COMPETITIVE_MIN_PATTERN_SAMPLES` | `3` | Minimum samples for pattern matching |
|
||||
| `COMPETITIVE_MAJOR_DECISION_WEIGHT_MULTIPLIER` | `1.3` | Weight multiplier for major decision patterns |
|
||||
| `COMPETITIVE_STALENESS_WINDOW_DAYS` | `180` | Window for staleness decay on competitive signals |
|
||||
| `COMPETITIVE_STALENESS_RECENT_DAYS` | `90` | Days within which signals are considered recent |
|
||||
| `COMPETITIVE_STALENESS_DECAY_PENALTY` | `0.5` | Decay penalty for stale competitive signals |
|
||||
| `COMPETITIVE_PROPAGATION_FAILURE_THRESHOLD` | `5` | Consecutive propagation failures before operator alert |
|
||||
| `ALERT_SOURCE_FAILURE_THRESHOLD` | `3` | Consecutive source failures before alert fires |
|
||||
| `ALERT_SOURCE_FAILURE_WINDOW_HOURS` | `6` | Lookback window for source failure alerting |
|
||||
| `ALERT_SCHEMA_FAILURE_RATE_THRESHOLD` | `0.3` | Extraction failure rate (30%) that triggers alert |
|
||||
| `ALERT_SCHEMA_FAILURE_WINDOW_HOURS` | `1` | Lookback window for schema failure spike |
|
||||
| `ALERT_LAKE_LAG_THRESHOLD_MINUTES` | `60` | Minutes since last lake publish before alert |
|
||||
| `ALERT_BROKER_ERROR_THRESHOLD` | `3` | Consecutive broker errors before alert |
|
||||
| `ALERT_BROKER_ERROR_WINDOW_HOURS` | `1` | Lookback window for broker error alerting |
|
||||
| `ALERT_CHECK_INTERVAL_SECONDS` | `120` | How often alerting rules are evaluated |
|
||||
| `RETENTION_RAW_MARKET_DAYS` | `90` | Retention period for raw market data (days) |
|
||||
| `RETENTION_RAW_NEWS_DAYS` | `180` | Retention period for raw news articles (days) |
|
||||
| `RETENTION_RAW_FILINGS_DAYS` | `365` | Retention period for raw SEC filings (days) |
|
||||
| `RETENTION_NORMALIZED_DAYS` | `180` | Retention period for normalized documents (days) |
|
||||
| `RETENTION_LLM_PROMPTS_DAYS` | `365` | Retention period for LLM prompt archives (days) |
|
||||
| `RETENTION_LLM_RESULTS_DAYS` | `365` | Retention period for LLM extraction results (days) |
|
||||
| `RETENTION_LAKEHOUSE_DAYS` | `730` | Retention period for lakehouse Parquet files (days) |
|
||||
| `RETENTION_AUDIT_DAYS` | `730` | Retention period for audit trail artifacts (days) |
|
||||
| `RETENTION_CLEANUP_INTERVAL_HOURS` | `24` | How often the retention cleanup worker runs |
|
||||
| `RETENTION_BATCH_SIZE` | `1000` | Number of objects processed per cleanup batch |
|
||||
| `LOG_LEVEL` | `INFO` | Logging level |
|
||||
| `JSON_LOGS` | `true` | Enable structured JSON logging |
|
||||
| `DEPLOY_STAGE` | (empty) | Deployment stage prefix for bucket names |
|
||||
|
||||
See `services/shared/config.py` for the complete list of all supported environment variables with their defaults.
|
||||
|
||||
---
|
||||
|
||||
## LLM Provider Configuration
|
||||
|
||||
Stonks Oracle supports two LLM backends: **Ollama** (local, self-hosted) and **vLLM** (high-performance inference server). The active provider is configured per-agent in the `ai_agents` database table, but the connection details come from environment variables.
|
||||
|
||||
### Option A: Bundled Ollama (default)
|
||||
|
||||
The `docker-compose.yml` includes an Ollama container with GPU passthrough via the NVIDIA Container Toolkit. On first start, pull a model:
|
||||
|
||||
```bash
|
||||
docker compose exec ollama ollama pull qwen3.5:9b-fast
|
||||
```
|
||||
|
||||
No additional configuration needed — services connect to `http://ollama:11434` by default.
|
||||
|
||||
The Ollama container requests all available NVIDIA GPUs via the `deploy.resources.reservations.devices` configuration. If no GPU is available, Ollama falls back to CPU inference (significantly slower).
|
||||
|
||||
### Option B: External Ollama
|
||||
|
||||
If Ollama is already running on the host (e.g. with GPU access), create a `docker-compose.override.yml`:
|
||||
|
||||
```yaml
|
||||
services:
|
||||
ollama:
|
||||
entrypoint: ["true"]
|
||||
restart: "no"
|
||||
ports: []
|
||||
extractor:
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
redis:
|
||||
condition: service_healthy
|
||||
environment:
|
||||
OLLAMA_BASE_URL: "http://host.docker.internal:11434"
|
||||
extra_hosts:
|
||||
- "host.docker.internal:host-gateway"
|
||||
recommendation:
|
||||
environment:
|
||||
OLLAMA_BASE_URL: "http://host.docker.internal:11434"
|
||||
extra_hosts:
|
||||
- "host.docker.internal:host-gateway"
|
||||
```
|
||||
|
||||
This disables the bundled Ollama container and routes services to the host's instance. Replace the port if your Ollama runs on a non-standard port. For a remote Ollama instance (not on localhost), replace `host.docker.internal` with the remote IP and remove the `extra_hosts` block.
|
||||
|
||||
### Option C: vLLM Server
|
||||
|
||||
For higher throughput or quantized models (e.g. `RedHatAI/Qwen3.6-35B-A3B-NVFP4`), point services at a vLLM server. Add to your `.env`:
|
||||
|
||||
```dotenv
|
||||
VLLM_BASE_URL=http://192.168.42.254:8000
|
||||
VLLM_MODEL=RedHatAI/Qwen3.6-35B-A3B-NVFP4
|
||||
VLLM_TIMEOUT=120
|
||||
VLLM_TEMPERATURE=0.7
|
||||
```
|
||||
|
||||
Then update the `ai_agents` table to use the vLLM provider:
|
||||
|
||||
```sql
|
||||
UPDATE ai_agents SET model_provider = 'vllm', model_name = 'RedHatAI/Qwen3.6-35B-A3B-NVFP4' WHERE active = true;
|
||||
```
|
||||
|
||||
Or use the API:
|
||||
|
||||
```bash
|
||||
curl -X PUT http://localhost:8004/api/admin/agents/document-extractor \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"model_provider": "vllm", "model_name": "RedHatAI/Qwen3.6-35B-A3B-NVFP4"}'
|
||||
```
|
||||
|
||||
### Option D: Mixed (Ollama + vLLM)
|
||||
|
||||
You can run different agents on different providers. For example, use vLLM for the high-volume extractor and Ollama for the thesis rewriter:
|
||||
|
||||
```sql
|
||||
UPDATE ai_agents SET model_provider = 'vllm', model_name = 'RedHatAI/Qwen3.6-35B-A3B-NVFP4' WHERE slug = 'document-extractor';
|
||||
UPDATE ai_agents SET model_provider = 'vllm', model_name = 'RedHatAI/Qwen3.6-35B-A3B-NVFP4' WHERE slug = 'event-classifier';
|
||||
UPDATE ai_agents SET model_provider = 'ollama', model_name = 'qwen3.5:9b-fast' WHERE slug = 'thesis-rewriter';
|
||||
```
|
||||
|
||||
Both `OLLAMA_BASE_URL` and `VLLM_BASE_URL` must be set in the environment for mixed mode.
|
||||
|
||||
### Automated Deployment
|
||||
|
||||
The `deploy-docker.sh` script handles LLM configuration automatically. It always uses the Docker Ollama container with GPU passthrough (NVIDIA Container Toolkit):
|
||||
|
||||
```bash
|
||||
# Deploy with defaults (Docker Ollama, GPU-accelerated)
|
||||
bash deploy-docker.sh
|
||||
|
||||
# Specify a custom model
|
||||
bash deploy-docker.sh --ollama-model qwen3.6
|
||||
|
||||
# Specify a different host and directory
|
||||
bash deploy-docker.sh --host user@myserver --dir /opt/stonks
|
||||
```
|
||||
|
||||
If an external Ollama URL is provided via `--ollama-url`, the script creates a `docker-compose.override.yml` that disables the bundled container and routes services to the external instance.
|
||||
|
||||
---
|
||||
|
||||
## Volume Mounts and Data Persistence
|
||||
|
||||
Docker Compose defines five named volumes for persistent data:
|
||||
|
||||
| Volume | Mounted By | Mount Path | Contents |
|
||||
|--------|-----------|------------|----------|
|
||||
| `pgdata` | postgres | `/var/lib/postgresql/data` | PostgreSQL database files |
|
||||
| `miniodata` | minio | `/data` | MinIO object storage (raw artifacts, lakehouse Parquet files) |
|
||||
| `ollama_models` | ollama | `/root/.ollama` | Downloaded LLM model weights |
|
||||
| `hive_data` | hive-metastore | `/opt/hive/data` | Hive metastore Derby database |
|
||||
| `superset_data` | superset | `/app/superset_home` | Superset configuration and metadata |
|
||||
|
||||
### Bind Mounts
|
||||
|
||||
In addition to named volumes, several services use bind mounts for configuration:
|
||||
|
||||
| Service | Host Path | Container Path | Mode | Purpose |
|
||||
|---------|-----------|---------------|------|---------|
|
||||
| postgres | `./infra/migrations` | `/docker-entrypoint-initdb.d` | rw | SQL migrations auto-applied on first start |
|
||||
| trino | `./infra/trino/catalog` | `/etc/trino/catalog` | rw | Trino catalog configuration (lakehouse, iceberg) |
|
||||
| hive-metastore | `./infra/hive/core-site.xml` | `/opt/hive/conf/core-site.xml` | ro | Hadoop core-site config for MinIO access |
|
||||
| hive-metastore | `./infra/hive/metastore-site.xml` | `/opt/hive/conf/metastore-site.xml` | ro | Hive metastore config |
|
||||
|
||||
### Resetting Data
|
||||
|
||||
To destroy all persistent data and start fresh:
|
||||
|
||||
```bash
|
||||
# Stop all containers and remove named volumes
|
||||
docker compose down -v
|
||||
```
|
||||
|
||||
This removes `pgdata`, `miniodata`, `ollama_models`, `hive_data`, and `superset_data`. The next `docker compose up` will re-initialize PostgreSQL with migrations, re-create MinIO buckets (via `minio-init`), and re-download Ollama models.
|
||||
|
||||
To reset only specific volumes:
|
||||
|
||||
```bash
|
||||
docker compose down
|
||||
docker volume rm stonks-oracle_pgdata # Reset database only
|
||||
docker compose up -d
|
||||
```
|
||||
|
||||
> **Note**: Volume names are prefixed with the project directory name (e.g., `stonks-oracle_pgdata`). Use `docker volume ls` to see exact names.
|
||||
|
||||
---
|
||||
|
||||
## Health Checks
|
||||
|
||||
Every service has a health check configured. Docker Compose uses these to enforce startup ordering via `depends_on` with `condition: service_healthy`.
|
||||
|
||||
### Infrastructure Health Checks
|
||||
|
||||
| Service | Test Command | Interval | Retries |
|
||||
|---------|-------------|----------|---------|
|
||||
| `postgres` | `pg_isready -U stonks` | 5s | 5 |
|
||||
| `redis` | `redis-cli ping` | 5s | 5 |
|
||||
| `minio` | `mc ready local` | 5s | 5 |
|
||||
|
||||
### Application Health Checks — FastAPI Services
|
||||
|
||||
FastAPI services (symbol-registry, trading-engine, risk-engine, query-api) use HTTP health endpoints:
|
||||
|
||||
| Service | Test Command | Interval | Timeout | Retries | Start Period |
|
||||
|---------|-------------|----------|---------|---------|-------------|
|
||||
| `symbol-registry` | `curl -f http://localhost:8000/health` | 10s | 5s | 3 | 15s |
|
||||
| `trading-engine` | `curl -f http://localhost:8000/health` | 10s | 5s | 3 | 15s |
|
||||
| `risk-engine` | `curl -f http://localhost:8000/health` | 10s | 5s | 3 | 15s |
|
||||
| `query-api` | `curl -f http://localhost:8000/health` | 10s | 5s | 3 | 15s |
|
||||
| `dashboard` | `curl -f http://localhost:8080/` | 10s | 5s | 3 | 10s |
|
||||
|
||||
### Application Health Checks — Worker Services
|
||||
|
||||
Worker services (no HTTP endpoint) use process liveness checks:
|
||||
|
||||
| Service | Test Command | Interval | Timeout | Retries | Start Period |
|
||||
|---------|-------------|----------|---------|---------|-------------|
|
||||
| `scheduler` | `pgrep -f 'python -m services.scheduler.app'` | 10s | 5s | 3 | 15s |
|
||||
| `ingestion` | `pgrep -f 'python -m services.ingestion.worker'` | 10s | 5s | 3 | 15s |
|
||||
| `parser` | `pgrep -f 'python -m services.parser.worker'` | 10s | 5s | 3 | 15s |
|
||||
| `extractor` | `pgrep -f 'python -m services.extractor.main'` | 10s | 5s | 3 | 15s |
|
||||
| `aggregation` | `pgrep -f 'python -m services.aggregation.main'` | 10s | 5s | 3 | 15s |
|
||||
| `recommendation` | `pgrep -f 'python -m services.recommendation.main'` | 10s | 5s | 3 | 15s |
|
||||
| `broker-adapter` | `pgrep -f 'python -m services.adapters.broker_service'` | 10s | 5s | 3 | 15s |
|
||||
| `lake-publisher` | `pgrep -f 'python -m services.lake_publisher.jobs'` | 10s | 5s | 3 | 15s |
|
||||
|
||||
### Verifying Service Health
|
||||
|
||||
```bash
|
||||
# Check all service statuses
|
||||
docker compose ps
|
||||
|
||||
# Check a specific service
|
||||
docker compose ps query-api
|
||||
|
||||
# Inspect health check details for a container
|
||||
docker inspect --format='{{json .State.Health}}' stonks-oracle-query-api-1 | python -m json.tool
|
||||
|
||||
# Wait for all services to be healthy
|
||||
docker compose up -d --wait
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Dockerfile Build Details
|
||||
|
||||
### `docker/Dockerfile` — Generic Python Service Image
|
||||
|
||||
Used by all application services except the scheduler. Accepts a `SERVICE_CMD` build argument that determines which service the container runs.
|
||||
|
||||
**Base image**: `python:3.12-slim` (via Harbor proxy cache in CI)
|
||||
|
||||
**Build arguments**:
|
||||
|
||||
| Argument | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `SERVICE_CMD` | `python -m services.scheduler.app` | The command executed when the container starts |
|
||||
| `CACHE_BUST` | (none) | Optional cache-busting argument to force rebuild of source layers |
|
||||
|
||||
**What gets copied**:
|
||||
- `requirements.txt` → pip dependencies installed
|
||||
- `services/` → all service source code
|
||||
- `scripts/` → operational scripts
|
||||
- `tests/` → test files (available for in-container testing)
|
||||
- `conftest.py` → pytest configuration
|
||||
|
||||
**Environment variables set**:
|
||||
- `PYTHONDONTWRITEBYTECODE=1` — no `.pyc` files
|
||||
- `PYTHONUNBUFFERED=1` — unbuffered stdout/stderr for log visibility
|
||||
- `PYTHONPATH=/app` — ensures `services.*` imports resolve
|
||||
|
||||
**System packages installed**: `gcc`, `libpq-dev` (PostgreSQL client library), `curl` (for health checks)
|
||||
|
||||
**Security**: Runs as non-root user `stonks` (UID 1000).
|
||||
|
||||
**How `SERVICE_CMD` works**: The `CMD` directive is `sh -c "${SERVICE_CMD}"`, so the build argument becomes the runtime command. Each service in `docker-compose.yml` overrides this via the `args.SERVICE_CMD` build parameter:
|
||||
|
||||
```yaml
|
||||
query-api:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: docker/Dockerfile
|
||||
args:
|
||||
SERVICE_CMD: "uvicorn services.api.app:app --host 0.0.0.0 --port 8000"
|
||||
```
|
||||
|
||||
### `docker/Dockerfile.scheduler` — Scheduler Image
|
||||
|
||||
A specialized variant of the generic Dockerfile used only by the `scheduler` service. Adds `postgresql-client` for running database migrations via `psql`.
|
||||
|
||||
**Additional contents**:
|
||||
- `infra/migrations/` → copied to `/app/infra/migrations/` for migration execution
|
||||
- `postgresql-client` system package installed
|
||||
|
||||
**Command**: Hardcoded `CMD ["python", "-m", "services.scheduler.app"]` (no `SERVICE_CMD` argument).
|
||||
|
||||
### `docker/Dockerfile.superset` — Custom Superset Image
|
||||
|
||||
Extends the official Apache Superset image with additional database drivers.
|
||||
|
||||
**Base image**: `apache/superset:latest` (via Harbor proxy cache in CI)
|
||||
|
||||
**Additional packages**: `trino[sqlalchemy]`, `psycopg2-binary`, `redis`
|
||||
|
||||
### `frontend/Dockerfile` — Dashboard Image
|
||||
|
||||
Multi-stage build for the React dashboard.
|
||||
|
||||
**Stage 1 — Build** (base: `node:24-alpine`):
|
||||
|
||||
| Build Argument | Default | Description |
|
||||
|---------------|---------|-------------|
|
||||
| `VITE_QUERY_API_URL` | `""` | Query API base URL (empty = use relative `/api/` proxy) |
|
||||
| `VITE_SYMBOL_REGISTRY_URL` | `""` | Symbol Registry base URL (empty = use relative `/registry/` proxy) |
|
||||
| `VITE_RISK_ENGINE_URL` | `""` | Risk Engine base URL (empty = use relative `/risk/` proxy) |
|
||||
|
||||
**Stage 2 — Serve** (base: `nginxinc/nginx-unprivileged:alpine`):
|
||||
- Serves the built static files on port 8080
|
||||
- Uses `frontend/nginx.conf` for SPA fallback and API reverse proxying
|
||||
- Proxies `/api/` → `query-api:8000`, `/registry/` → `symbol-registry:8000`, `/risk/` → `risk:8000`, `/trading/` → `trading-engine:8000`
|
||||
- SSE stream endpoint (`/api/ops/pipeline/stream`) has buffering disabled for real-time delivery
|
||||
- Static assets under `/assets/` are cached with 1-year expiry
|
||||
|
||||
### Building Custom Images
|
||||
|
||||
To build a single service image locally:
|
||||
|
||||
```bash
|
||||
# Build the query-api image
|
||||
docker compose build query-api
|
||||
|
||||
# Build with a custom SERVICE_CMD
|
||||
docker build -t my-custom-service \
|
||||
--build-arg SERVICE_CMD="python -m services.my_service.main" \
|
||||
-f docker/Dockerfile .
|
||||
|
||||
# Build the dashboard with custom API URLs
|
||||
docker build -t my-dashboard \
|
||||
--build-arg VITE_QUERY_API_URL="https://api.example.com" \
|
||||
-f frontend/Dockerfile frontend/
|
||||
|
||||
# Rebuild all images
|
||||
docker compose build
|
||||
|
||||
# Rebuild without cache (force fresh build)
|
||||
docker compose build --no-cache
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Dependency Ordering
|
||||
|
||||
Docker Compose enforces startup order using `depends_on` with health check conditions. The dependency graph is:
|
||||
|
||||
```
|
||||
postgres (healthy) ──┬── scheduler
|
||||
├── symbol-registry
|
||||
├── ingestion
|
||||
├── parser
|
||||
├── extractor
|
||||
├── aggregation
|
||||
├── recommendation
|
||||
├── trading-engine
|
||||
├── risk-engine
|
||||
├── broker-adapter
|
||||
├── lake-publisher
|
||||
└── query-api
|
||||
|
||||
redis (healthy) ─────┬── scheduler
|
||||
├── ingestion
|
||||
├── parser
|
||||
├── extractor
|
||||
├── aggregation
|
||||
├── recommendation
|
||||
├── trading-engine
|
||||
├── broker-adapter
|
||||
└── query-api
|
||||
|
||||
minio (healthy) ─────┬── minio-init
|
||||
├── ingestion
|
||||
├── lake-publisher
|
||||
└── query-api
|
||||
|
||||
ollama (started) ────── extractor
|
||||
|
||||
minio ───────────────── trino
|
||||
hive-metastore ─────── trino
|
||||
trino ──────────────── superset (via depends_on)
|
||||
|
||||
query-api (healthy) ── dashboard
|
||||
```
|
||||
|
||||
Services with `condition: service_healthy` wait until the dependency's health check passes. The `extractor` depends on `ollama` with `condition: service_started` (no health check — Ollama may take time to load models).
|
||||
|
||||
---
|
||||
|
||||
## Operational Commands
|
||||
|
||||
### Starting Services
|
||||
|
||||
```bash
|
||||
# Start all services in the background
|
||||
docker compose up -d
|
||||
|
||||
# Start all services and wait for health checks
|
||||
docker compose up -d --wait
|
||||
|
||||
# Start only infrastructure (useful for local development)
|
||||
docker compose up -d postgres redis minio minio-init ollama
|
||||
|
||||
# Start a specific service and its dependencies
|
||||
docker compose up -d query-api
|
||||
```
|
||||
|
||||
### Stopping Services
|
||||
|
||||
```bash
|
||||
# Stop all services (preserves volumes)
|
||||
docker compose down
|
||||
|
||||
# Stop all services and remove volumes (full reset)
|
||||
docker compose down -v
|
||||
|
||||
# Stop a specific service
|
||||
docker compose stop trading-engine
|
||||
```
|
||||
|
||||
### Restarting Services
|
||||
|
||||
```bash
|
||||
# Restart a specific service
|
||||
docker compose restart query-api
|
||||
|
||||
# Restart with a fresh build
|
||||
docker compose up -d --build query-api
|
||||
|
||||
# Force recreate a service (picks up compose file changes)
|
||||
docker compose up -d --force-recreate query-api
|
||||
```
|
||||
|
||||
### Viewing Logs
|
||||
|
||||
```bash
|
||||
# Follow logs for all services
|
||||
docker compose logs -f
|
||||
|
||||
# Follow logs for a specific service
|
||||
docker compose logs -f query-api
|
||||
|
||||
# View last 50 lines of a service's logs
|
||||
docker compose logs --tail=50 ingestion
|
||||
|
||||
# View logs for multiple services
|
||||
docker compose logs -f scheduler ingestion extractor
|
||||
```
|
||||
|
||||
### Scaling Replicas
|
||||
|
||||
```bash
|
||||
# Scale a worker service to 3 replicas
|
||||
docker compose up -d --scale ingestion=3
|
||||
|
||||
# Scale multiple services
|
||||
docker compose up -d --scale ingestion=3 --scale extractor=2
|
||||
|
||||
# Scale back to 1
|
||||
docker compose up -d --scale ingestion=1
|
||||
```
|
||||
|
||||
> **Note**: Scaling works best for worker services (ingestion, parser, extractor, aggregation, recommendation, broker-adapter, lake-publisher) that consume from Redis queues. Do not scale FastAPI services that expose host ports without adjusting port mappings.
|
||||
|
||||
### Inspecting Services
|
||||
|
||||
```bash
|
||||
# List all services and their status
|
||||
docker compose ps
|
||||
|
||||
# View resource usage
|
||||
docker compose top
|
||||
|
||||
# Execute a command inside a running container
|
||||
docker compose exec query-api python -c "from services.shared.config import load_config; print(load_config())"
|
||||
|
||||
# Open a shell in a container
|
||||
docker compose exec postgres psql -U stonks -d stonks
|
||||
|
||||
# Seed the database
|
||||
docker compose exec scheduler python -m services.symbol_registry.seed
|
||||
```
|
||||
|
||||
### Full Reset
|
||||
|
||||
```bash
|
||||
# Nuclear option: stop everything, remove volumes, rebuild, restart
|
||||
docker compose down -v
|
||||
docker compose build --no-cache
|
||||
docker compose up -d
|
||||
```
|
||||
|
||||
This destroys all data (database, object storage, model weights, metastore, Superset config) and starts from scratch. PostgreSQL migrations are re-applied automatically. MinIO buckets are re-created by `minio-init`. Ollama models must be re-downloaded.
|
||||
|
||||
---
|
||||
|
||||
## MinIO Bucket Initialization
|
||||
|
||||
The `minio-init` service runs once on startup and creates the required object storage buckets:
|
||||
|
||||
| Bucket | Purpose |
|
||||
|--------|---------|
|
||||
| `stonks-raw-market` | Raw market data from Polygon.io |
|
||||
| `stonks-raw-news` | Raw news articles |
|
||||
| `stonks-raw-filings` | Raw SEC filings |
|
||||
| `stonks-normalized` | Normalized/parsed documents |
|
||||
| `stonks-llm-prompts` | LLM prompt archives |
|
||||
| `stonks-llm-results` | LLM extraction results |
|
||||
| `stonks-lakehouse` | Parquet fact tables for Trino |
|
||||
| `stonks-audit` | Audit trail artifacts |
|
||||
|
||||
Access the MinIO console at `http://localhost:9001` (credentials: `minioadmin` / `minioadmin`).
|
||||
|
||||
---
|
||||
|
||||
## Dashboard Reverse Proxy
|
||||
|
||||
The dashboard container runs nginx with reverse proxy rules that route API requests to backend services using Docker Compose service names:
|
||||
|
||||
| Path | Proxied To | Service |
|
||||
|------|-----------|---------|
|
||||
| `/api/` | `http://query-api:8000` | Query API |
|
||||
| `/api/ops/pipeline/stream` | `http://query-api:8000` (SSE, no buffering) | Query API (real-time pipeline stream) |
|
||||
| `/registry/` | `http://symbol-registry:8000/` | Symbol Registry API |
|
||||
| `/risk/` | `http://risk:8000/` | Risk Engine (via network alias) |
|
||||
| `/trading/` | `http://trading-engine:8000/` | Trading Engine API |
|
||||
|
||||
The `risk-engine` service has a network alias of `risk` in `docker-compose.yml` so the nginx upstream resolves correctly.
|
||||
|
||||
All other paths serve the React SPA with `try_files` fallback to `index.html`. Static assets under `/assets/` are served with 1-year cache headers.
|
||||
|
||||
Security headers applied: `X-Frame-Options: SAMEORIGIN`, `X-Content-Type-Options: nosniff`, `Referrer-Policy: strict-origin-when-cross-origin`.
|
||||
|
||||
---
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### Service won't start
|
||||
|
||||
Check dependency health:
|
||||
|
||||
```bash
|
||||
docker compose ps postgres redis minio
|
||||
```
|
||||
|
||||
If infrastructure services are unhealthy, application services will wait indefinitely. Check infrastructure logs:
|
||||
|
||||
```bash
|
||||
docker compose logs postgres
|
||||
```
|
||||
|
||||
### Database migration errors
|
||||
|
||||
Migrations in `./infra/migrations/` are applied by PostgreSQL's `docker-entrypoint-initdb.d` mechanism, which only runs on first database initialization. If you need to re-run migrations:
|
||||
|
||||
```bash
|
||||
docker compose down -v # Remove pgdata volume
|
||||
docker compose up -d # Migrations re-applied on fresh init
|
||||
```
|
||||
|
||||
### Ollama model not available
|
||||
|
||||
The extractor service needs an LLM model loaded. Pull a model manually:
|
||||
|
||||
```bash
|
||||
# If using bundled Ollama container:
|
||||
docker compose exec ollama ollama pull qwen3.5:9b-fast
|
||||
|
||||
# If using host Ollama:
|
||||
ollama pull qwen3.5:9b-fast
|
||||
|
||||
# If using vLLM, ensure the model is loaded on the vLLM server
|
||||
curl http://your-vllm-host:8000/v1/models
|
||||
```
|
||||
|
||||
### Ollama port conflict (address already in use)
|
||||
|
||||
If Ollama is already running on the host, the bundled container will fail to bind port 11434. Use the external Ollama configuration described in the "LLM Provider Configuration" section above, or use `deploy-docker.sh` which handles this automatically.
|
||||
|
||||
### GPU not detected by Ollama container
|
||||
|
||||
Ensure the NVIDIA Container Toolkit is installed and Docker is configured:
|
||||
|
||||
```bash
|
||||
# Verify GPU passthrough works
|
||||
docker run --rm --gpus all nvidia/cuda:12.8.0-base-ubuntu24.04 nvidia-smi
|
||||
|
||||
# If it fails, reconfigure Docker runtime
|
||||
sudo nvidia-ctk runtime configure --runtime=docker
|
||||
sudo systemctl restart docker
|
||||
```
|
||||
|
||||
### Port conflicts
|
||||
|
||||
If a port is already in use, modify the host port mapping in `docker-compose.yml`:
|
||||
|
||||
```yaml
|
||||
query-api:
|
||||
ports:
|
||||
- "9004:8000" # Changed from 8004 to 9004
|
||||
```
|
||||
|
||||
### Container runs out of memory
|
||||
|
||||
The full stack requires at least 16 GB RAM. If services are being OOM-killed:
|
||||
|
||||
```bash
|
||||
# Check which containers are using the most memory
|
||||
docker stats --no-stream
|
||||
|
||||
# Reduce memory usage by stopping non-essential services
|
||||
docker compose stop trino hive-metastore superset
|
||||
```
|
||||
@@ -0,0 +1,915 @@
|
||||
# Stonks Oracle — Mathematical Reference
|
||||
|
||||
Every equation, formula, threshold, and constant used in the signal processing, aggregation, recommendation, and trading pipeline. Organized by pipeline stage.
|
||||
|
||||
Code references are provided so each formula can be traced to its implementation.
|
||||
|
||||
---
|
||||
|
||||
## 1. Signal Scoring
|
||||
|
||||
**Source:** `services/aggregation/scoring.py`
|
||||
|
||||
### 1.1 Combined Signal Weight
|
||||
|
||||
Each document signal receives a composite weight:
|
||||
|
||||
```
|
||||
W_combined = G_conf × W_recency × W_credibility × (1 + B_novelty) × M_context
|
||||
```
|
||||
|
||||
| Component | Symbol | Formula | Range |
|
||||
|---|---|---|---|
|
||||
| Confidence gate | G_conf | 1 if extraction_confidence ≥ 0.2, else 0 | {0, 1} |
|
||||
| Recency decay | W_recency | 2^(−t_age / t_half) | [0.01, 1.0] |
|
||||
| Credibility | W_credibility | clamp(credibility, 0.1, 1.0)^α | [0.1, 1.0] |
|
||||
| Novelty bonus | B_novelty | novelty_score × 0.25 | [0, 0.25] |
|
||||
| Market context | M_context | 1 + boost_vol + boost_vol_surge | [1.0, 1.45] |
|
||||
|
||||
### 1.2 Recency Decay
|
||||
|
||||
```
|
||||
W_recency = max( 2^(−t_age / t_half), 0.01 )
|
||||
```
|
||||
|
||||
where `t_age` is document age in hours and half-lives by window are:
|
||||
|
||||
| Window | t_half (hours) |
|
||||
|---|---|
|
||||
| intraday | 2 |
|
||||
| 1d | 12 |
|
||||
| 7d | 72 |
|
||||
| 30d | 240 |
|
||||
| 90d | 720 |
|
||||
|
||||
### 1.3 Credibility Weight
|
||||
|
||||
```
|
||||
W_credibility = clamp(c_raw, 0.1, 1.0)^α where α = 1.0 (default)
|
||||
```
|
||||
|
||||
α > 1 penalizes low-credibility sources more aggressively; α < 1 flattens the curve.
|
||||
|
||||
### 1.4 Market Context Multiplier
|
||||
|
||||
```
|
||||
boost_vol = min( ln(1 + max(σ − 1.0, 0)) × 0.15, 0.30 )
|
||||
|
||||
boost_surge = 0.15 if ΔV% > 50%, else 0
|
||||
|
||||
M_context = 1.0 + boost_vol + boost_surge
|
||||
```
|
||||
|
||||
where σ is price volatility and ΔV% is volume change percentage.
|
||||
|
||||
### 1.5 Weighted Sentiment Average
|
||||
|
||||
```
|
||||
S_avg = Σ(W_combined_i × impact_i × sentiment_i) / Σ(W_combined_i × impact_i)
|
||||
```
|
||||
|
||||
- sentiment_i ∈ {+1.0 (positive), −1.0 (negative), 0.0 (neutral/mixed)}
|
||||
- impact_i ∈ [0, 1] from extraction
|
||||
- Returns 0.0 when denominator = 0
|
||||
|
||||
---
|
||||
|
||||
## 1B. Probabilistic Signal Scoring (Feature-Flagged)
|
||||
|
||||
**Source:** `services/aggregation/scoring.py`
|
||||
**Active when:** `probabilistic_scoring_enabled = true` in `risk_configs.config` JSONB
|
||||
|
||||
When the probabilistic pipeline is enabled, the combined weight formula changes:
|
||||
|
||||
### 1B.1 Combined Signal Weight (Probabilistic)
|
||||
|
||||
```
|
||||
W_combined = G_sigmoid × W_recency(adaptive) × W_credibility × (1 + B_novelty) × R_info × F_accuracy × M_regime
|
||||
```
|
||||
|
||||
| Component | Symbol | Formula | Range |
|
||||
|---|---|---|---|
|
||||
| Sigmoid gate | G_sigmoid | σ(k·(x − midpoint)) = 1/(1+e^(−5·(x−0.5))) | (0, 1) |
|
||||
| Adaptive recency | W_recency | 2^(−t_age / τ_adaptive) | [0.01, 1.0] |
|
||||
| Credibility | W_credibility | same as heuristic | [0.1, 1.0] |
|
||||
| Novelty bonus | B_novelty | same as heuristic | [0, 0.25] |
|
||||
| Information gain | R_info | 1 + λ·(−log₂ P(event_type)) | [1.0, 3.0] |
|
||||
| Source accuracy | F_accuracy | 0.5 + accuracy_ratio (if samples ≥ 10, else 1.0) | [0.5, 1.5] |
|
||||
| Regime multiplier | M_regime | 1 + 0.15·|z_r| + 0.10·|z_v| | [1.0, 2.5] |
|
||||
|
||||
### 1B.2 Sigmoid Confidence Gate
|
||||
|
||||
Replaces the binary 0/1 gate with a smooth transition:
|
||||
|
||||
```
|
||||
G_sigmoid = σ(k·(x − m)) = 1 / (1 + e^(−k·(x−m)))
|
||||
```
|
||||
|
||||
Default: k = 5.0, m = 0.5. At x=0.5 → 0.5; at x=0.2 → ~0.18; at x=0.8 → ~0.82.
|
||||
|
||||
### 1B.3 Information Gain (Surprise Weighting)
|
||||
|
||||
```
|
||||
R_info = min(1 + λ·(−log₂ P(event_type)), 3.0)
|
||||
```
|
||||
|
||||
| Event Type | P(event_type) | R_info (λ=0.3) |
|
||||
|---|---|---|
|
||||
| earnings | 0.25 | 1.60 |
|
||||
| dividend | 0.15 | 1.84 |
|
||||
| product_launch | 0.10 | 2.00 |
|
||||
| regulatory | 0.08 | 2.07 |
|
||||
| management_change | 0.06 | 2.19 |
|
||||
| legal | 0.05 | 2.29 |
|
||||
| restructuring | 0.04 | 2.39 |
|
||||
| m_and_a | 0.03 | 2.56 |
|
||||
| unknown | 0.10 (default) | 2.00 |
|
||||
|
||||
### 1B.4 Adaptive Recency Decay
|
||||
|
||||
```
|
||||
τ_adaptive = τ_base × (1 + β_impact) × (1 + β_surprise) × (1 + β_market)
|
||||
```
|
||||
|
||||
| Factor | Formula | Range |
|
||||
|---|---|---|
|
||||
| β_impact | impact_score × 1.0 | [0, 1.0] |
|
||||
| β_surprise | (R_info − 1) / 2 × 1.0 | [0, 1.0] |
|
||||
| β_market | (M_regime − 1) / 0.45 × 0.5 | [0, 0.5] |
|
||||
|
||||
Maximum adaptive half-life: 6× base (when all factors at max).
|
||||
Minimum: τ_base (adaptive decay is never faster than fixed).
|
||||
|
||||
### 1B.5 Regime Multiplier
|
||||
|
||||
```
|
||||
z_r = (r_t − μ_20) / σ_20 (return z-score)
|
||||
z_v = (ln(V_t) − μ_V) / σ_V (log-volume z-score)
|
||||
M_regime = clamp(1 + 0.15·|z_r| + 0.10·|z_v|, 1.0, 2.5)
|
||||
```
|
||||
|
||||
Defaults to 1.0 when market data unavailable or σ = 0.
|
||||
|
||||
### 1B.6 Source Accuracy Factor
|
||||
|
||||
```
|
||||
F_accuracy = 0.5 + clamp(accuracy_ratio, 0, 1) if sample_count ≥ 10
|
||||
F_accuracy = 1.0 if sample_count < 10
|
||||
```
|
||||
|
||||
Stored in `source_accuracy` table, updated asynchronously from realized 7-day price outcomes.
|
||||
|
||||
---
|
||||
|
||||
## 2. Trend Summary Assembly
|
||||
|
||||
**Source:** `services/aggregation/worker.py`
|
||||
|
||||
### 2.1 Trend Direction
|
||||
|
||||
| Condition | Direction |
|
||||
|---|---|
|
||||
| S_avg ≥ 0.15 | Bullish |
|
||||
| S_avg ≤ −0.15 | Bearish |
|
||||
| contradiction > 0.10 AND |S_avg| < 0.30 | Mixed |
|
||||
| otherwise | Neutral |
|
||||
|
||||
### 2.2 Trend Strength
|
||||
|
||||
```
|
||||
strength = min(|S_avg|, 1.0)
|
||||
```
|
||||
|
||||
### 2.3 Contradiction Score
|
||||
|
||||
**Source:** `services/aggregation/contradiction.py`
|
||||
|
||||
```
|
||||
contradiction = W_minority / (W_positive + W_negative)
|
||||
```
|
||||
|
||||
where:
|
||||
```
|
||||
W_positive = Σ(W_combined_i × impact_i) for signals with sentiment > 0
|
||||
W_negative = Σ(W_combined_i × impact_i) for signals with sentiment < 0
|
||||
W_minority = min(W_positive, W_negative)
|
||||
```
|
||||
|
||||
Range: [0, 1]. 0 = full agreement, 0.5 = equal-weight disagreement.
|
||||
|
||||
### 2.4 Trend Confidence
|
||||
|
||||
```
|
||||
confidence = clamp(0.3 × F_count + 0.3 × C_avg + 0.4 × A_agreement − P_contradiction, 0, 1)
|
||||
```
|
||||
|
||||
| Component | Formula |
|
||||
|---|---|
|
||||
| F_count (source count) | min(N_unique / 15, 0.8) |
|
||||
| C_avg (extraction confidence) | mean of extraction confidences |
|
||||
| A_agreement (signal agreement) | fraction_same_direction × min(1, log₂(N_unique + 1) / log₂(8)) |
|
||||
| P_contradiction | contradiction_score × 0.4 |
|
||||
|
||||
---
|
||||
|
||||
## 2B. Probabilistic Trend Assembly (Feature-Flagged)
|
||||
|
||||
**Source:** `services/aggregation/worker.py`, `services/aggregation/bayesian.py`
|
||||
**Active when:** `probabilistic_scoring_enabled = true`
|
||||
|
||||
### 2B.1 Bayesian Posterior Accumulation
|
||||
|
||||
```
|
||||
L_t = Σ(W_combined_i × sentiment_i) (log-likelihood)
|
||||
P_bull = σ(L_t) = 1 / (1 + e^(−L_t)) (bullish probability)
|
||||
α = 1 + W_bull (W_bull = Σ W_combined for positive signals)
|
||||
β = 1 + W_bear (W_bear = Σ W_combined for negative signals)
|
||||
C_bayesian = 1 − 4αβ / (α + β)² (Bayesian confidence)
|
||||
H = −P_bull·log₂(P_bull) − (1−P_bull)·log₂(1−P_bull) (Shannon entropy)
|
||||
```
|
||||
|
||||
Uninformative prior (no signals): P_bull=0.5, α=1, β=1, C=0, H=1.0.
|
||||
|
||||
### 2B.2 Entropy-Based Direction
|
||||
|
||||
| Condition | Direction |
|
||||
|---|---|
|
||||
| H > 0.9 | Mixed |
|
||||
| P_bull > 0.65 | Bullish |
|
||||
| P_bull < 0.35 | Bearish |
|
||||
| otherwise | Neutral |
|
||||
|
||||
### 2B.3 Bayesian Trend Confidence
|
||||
|
||||
```
|
||||
confidence = clamp(0.5 × C_bayesian + 0.25 × F_count + 0.25 × C_avg_credibility − P_contradiction, 0, 1)
|
||||
```
|
||||
|
||||
| Component | Formula |
|
||||
|---|---|
|
||||
| C_bayesian | 1 − 4αβ/(α+β)² from Beta posterior |
|
||||
| F_count | min(N_unique_sources / 15, 0.8) |
|
||||
| C_avg_credibility | mean credibility weight across active signals |
|
||||
| P_contradiction | contradiction_entropy × regime.contradiction_penalty_multiplier |
|
||||
|
||||
### 2B.4 Weighted Disagreement Entropy (Contradiction)
|
||||
|
||||
**Source:** `services/aggregation/contradiction.py`
|
||||
|
||||
```
|
||||
f_pos = W_positive / (W_positive + W_negative)
|
||||
f_neg = 1 − f_pos
|
||||
H_contradiction = −f_pos·log₂(f_pos) − f_neg·log₂(f_neg)
|
||||
contradiction_score = H_contradiction × min(1.0, (W_pos + W_neg) / W_threshold)
|
||||
```
|
||||
|
||||
W_threshold default = 5.0. Returns 0.0 when only one direction exists.
|
||||
|
||||
### 2B.5 Regime Detection
|
||||
|
||||
**Source:** `services/aggregation/regime.py`
|
||||
|
||||
```
|
||||
R = sign(EMA_20 − EMA_100) (trend indicator)
|
||||
V_r = σ_20 / σ_100 (volatility ratio)
|
||||
```
|
||||
|
||||
| Condition | Regime | Threshold | Contradiction Mult |
|
||||
|---|---|---|---|
|
||||
| V_r > 1.5 | Panic | ±0.10 | 0.4 |
|
||||
| R ≠ 0 AND V_r < 1.2 | Trend-following | ±0.15 | 0.4 |
|
||||
| R = 0 AND V_r < 1.0 | Mean-reversion | ±0.20 | 0.4 |
|
||||
| otherwise | Uncertainty | ±0.15 | 0.6 |
|
||||
|
||||
Falls back to Uncertainty when data < 100 days or σ = 0.
|
||||
|
||||
---
|
||||
|
||||
## 3. Macro Impact Scoring (Layer 2)
|
||||
|
||||
**Source:** `services/aggregation/interpolation.py`
|
||||
|
||||
### 3.1 Overlap Components
|
||||
|
||||
**Geographic overlap:**
|
||||
```
|
||||
O_geo = Σ revenue_pct_r for each event region r in company's revenue mix
|
||||
```
|
||||
Range: [0, 1]
|
||||
|
||||
**Supply chain overlap:**
|
||||
```
|
||||
O_supply = |event_regions ∩ supply_regions| / |supply_regions|
|
||||
```
|
||||
|
||||
**Commodity overlap:**
|
||||
```
|
||||
O_commodity = |event_commodities ∩ company_commodities| / |company_commodities|
|
||||
```
|
||||
|
||||
**Sector overlap:**
|
||||
```
|
||||
O_sector = 1.0 if company_sector ∈ event_affected_sectors, else 0.0
|
||||
```
|
||||
|
||||
### 3.2 Raw Macro Impact Score
|
||||
|
||||
```
|
||||
S_raw = W_severity × (0.35 × O_geo + 0.25 × O_supply + 0.25 × O_commodity + 0.15 × O_sector)
|
||||
```
|
||||
|
||||
Severity weights:
|
||||
|
||||
| Severity | W_severity |
|
||||
|---|---|
|
||||
| critical | 1.0 |
|
||||
| high | 0.75 |
|
||||
| moderate | 0.5 |
|
||||
| low | 0.25 |
|
||||
|
||||
### 3.3 Resilience Modifier
|
||||
|
||||
For international events, the raw score is adjusted by market position:
|
||||
|
||||
```
|
||||
S_final = clamp(S_raw × R_tier, 0, 1)
|
||||
```
|
||||
|
||||
| Market Position Tier | R_tier |
|
||||
|---|---|
|
||||
| Global leader | 0.70 |
|
||||
| Multinational | 0.85 |
|
||||
| Regional | 1.00 |
|
||||
| Domestic | 1.20 |
|
||||
|
||||
For domestic-only events, R_tier = 1.0 regardless of tier.
|
||||
|
||||
### 3B. Multiplicative Macro Exposure (Probabilistic)
|
||||
|
||||
**Active when:** `probabilistic_scoring_enabled = true`
|
||||
|
||||
```
|
||||
S_raw = W_severity × (1 − Π_k(1 − w_k × O_k))
|
||||
= W_severity × (1 − (1−0.35·O_geo)(1−0.25·O_supply)(1−0.25·O_commodity)(1−0.15·O_sector))
|
||||
```
|
||||
|
||||
Zero overlap → 0.0. Max overlap (all 1.0) → severity × 0.689.
|
||||
|
||||
### 3B.1 Conditional Macro Integration
|
||||
|
||||
When both company and macro signals exist:
|
||||
```
|
||||
modifier = clamp(1 + M_macro × sign_alignment, 0.5, 1.5)
|
||||
S_adjusted = S_company × modifier
|
||||
```
|
||||
|
||||
sign_alignment = +1 (agree), −1 (disagree), 0 (neutral/mixed).
|
||||
|
||||
When only macro signals exist: additive fallback with weight 0.3.
|
||||
When only company signals exist: modifier = 1.0.
|
||||
|
||||
### 3.4 Macro Impact Confidence
|
||||
|
||||
```
|
||||
confidence = min(event_confidence × min(O_total + 0.3, 1.0), 1.0)
|
||||
```
|
||||
|
||||
where O_total = O_geo + O_supply + O_commodity + O_sector.
|
||||
|
||||
### 3.5 Accelerated Staleness Decay
|
||||
|
||||
For short-term events older than 48 hours:
|
||||
|
||||
```
|
||||
decay_standard = e^(−0.693 × t_age_hours / t_half_hours) (t_half default = 168h)
|
||||
decay_accelerated = decay_standard × 0.5
|
||||
```
|
||||
|
||||
### 3.6 Macro Signal as WeightedSignal
|
||||
|
||||
When merged into the aggregation engine:
|
||||
|
||||
```
|
||||
impact_score_macro = macro_impact_score × W_macro (W_macro = 0.3 default)
|
||||
sentiment_value = +1 if positive, −1 if negative
|
||||
```
|
||||
|
||||
Recency decay uses the global event's publication time.
|
||||
|
||||
---
|
||||
|
||||
## 4. Competitive Signals (Layer 3)
|
||||
|
||||
### 4.1 Pattern Confidence
|
||||
|
||||
**Source:** `services/aggregation/pattern_matcher.py`
|
||||
|
||||
```
|
||||
confidence = F_sample × 0.4 + F_consistency × 0.4 + F_recency × 0.2
|
||||
```
|
||||
|
||||
| Factor | Formula |
|
||||
|---|---|
|
||||
| F_sample | min(N_samples / 20, 1.0) |
|
||||
| F_consistency | max(pct_bullish, pct_bearish) |
|
||||
| F_recency | 1.0 if age ≤ 7d; 0.7 if age ≤ 90d; 0.4 otherwise |
|
||||
|
||||
**Modifiers:**
|
||||
- Major corporate decision (m&a, earnings, legal): confidence × 1.3
|
||||
- Insufficient data (N_samples < min_pattern_samples): cap at 0.25
|
||||
- Stale data (age > staleness_window_days): confidence × staleness_decay_penalty
|
||||
|
||||
**Lookback windows:**
|
||||
- Routine signals: 180 days
|
||||
- Major corporate decisions: 365 days
|
||||
|
||||
### 4.2 Cross-Company Signal Strength
|
||||
|
||||
**Source:** `services/aggregation/signal_propagation.py`
|
||||
|
||||
```
|
||||
S_competitive = clamp(S_pattern_avg × R_relationship × C_pattern × I_source, 0, 1)
|
||||
```
|
||||
|
||||
| Component | Description |
|
||||
|---|---|
|
||||
| S_pattern_avg | Average historical outcome strength [0, 1] |
|
||||
| R_relationship | Relationship strength from competitor_relationships [0, 1] |
|
||||
| C_pattern | Pattern confidence from §4.1 |
|
||||
| I_source | Source document's impact_score [0, 1] |
|
||||
|
||||
**Threshold gate:** Skipped if R_relationship < propagation_strength_threshold (default 0.2).
|
||||
|
||||
### 4B. Graph-Distance Attenuation (Probabilistic)
|
||||
|
||||
**Active when:** `probabilistic_scoring_enabled = true`
|
||||
|
||||
```
|
||||
S_transfer = S_source × ρ_historical × e^(−d_network)
|
||||
```
|
||||
|
||||
| Component | Description |
|
||||
|---|---|
|
||||
| S_source | Source signal strength |
|
||||
| ρ_historical | 90-day rolling Pearson correlation (default 0.3 same-sector, 0.1 cross-sector) |
|
||||
| d_network | Shortest path in competitor graph (capped at 3) |
|
||||
|
||||
No propagation when d_network > 3 (e^(−3) ≈ 0.05).
|
||||
|
||||
### 4.3 Competitive Signal as WeightedSignal
|
||||
|
||||
```
|
||||
impact_score_competitive = S_competitive × W_competitive (W_competitive = 0.2 default)
|
||||
direction = majority historical outcome (bullish or bearish)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 5. Trend Projection
|
||||
|
||||
**Source:** `services/aggregation/projection.py`
|
||||
|
||||
### 5.1 Trend Momentum
|
||||
|
||||
```
|
||||
momentum = S_current_signed − S_previous_signed
|
||||
```
|
||||
|
||||
where `S_signed = direction_sign × strength` (bullish = +1, bearish = −1, neutral = 0).
|
||||
|
||||
When no previous data exists:
|
||||
```
|
||||
momentum = direction_sign × strength × 0.5
|
||||
```
|
||||
|
||||
Range: [−1, 1]
|
||||
|
||||
### 5.2 Macro Decay Projection
|
||||
|
||||
For each active macro event projected forward by `H` days:
|
||||
|
||||
```
|
||||
F_future = 2^(−(t_current + H) / t_half)
|
||||
I_projected = macro_impact_score × F_future × W_severity
|
||||
```
|
||||
|
||||
Decay half-lives:
|
||||
|
||||
| Duration | t_half (days) |
|
||||
|---|---|
|
||||
| short_term | 1.0 |
|
||||
| medium_term | 7.0 |
|
||||
| long_term | 30.0 |
|
||||
|
||||
Aggregate direction: bullish if W_pos > 1.2 × W_neg; bearish if W_neg > 1.2 × W_pos; mixed if both > 0.
|
||||
|
||||
### 5.3 Projection Blending
|
||||
|
||||
```
|
||||
W_macro_blend = min(S_macro_projected × 0.4, 0.4)
|
||||
W_company = 1.0 − W_macro_blend
|
||||
|
||||
S_blended = W_company × S_momentum_projected + W_macro_blend × S_macro_signed
|
||||
```
|
||||
|
||||
**Catalyst boost:** `min(N_catalysts × 0.02, 0.1)` added to projected strength.
|
||||
|
||||
**Projected confidence:**
|
||||
```
|
||||
C_projected = C_base × 0.8 + min(S_macro × 0.15, 0.1)
|
||||
```
|
||||
|
||||
**Divergence detection:** Flagged when projected direction ≠ current trend direction.
|
||||
|
||||
### 5B. Exponentially Weighted Momentum (Probabilistic)
|
||||
|
||||
**Source:** `services/aggregation/projection.py`
|
||||
**Active when:** `probabilistic_scoring_enabled = true`
|
||||
|
||||
```
|
||||
M_t = Σ_{k=0}^{K-1} λ^k × ΔS_{t-k} (λ = 0.7, K ≤ 10)
|
||||
M_normalized = M_t / Σ_{k=0}^{K-1} λ^k (range: [−1, 1])
|
||||
M_adj = clamp(M_normalized / max(σ_20, 0.01), −2.0, 2.0)
|
||||
```
|
||||
|
||||
Falls back to heuristic momentum when < 2 historical cycles available.
|
||||
|
||||
---
|
||||
|
||||
## 6. Data Quality Suppression
|
||||
|
||||
**Source:** `services/recommendation/suppression.py`
|
||||
|
||||
### 6.1 Data Quality Score
|
||||
|
||||
```
|
||||
Q = 0.4 × Q_confidence + 0.3 × Q_freshness + 0.3 × Q_coverage
|
||||
```
|
||||
|
||||
| Component | Formula |
|
||||
|---|---|
|
||||
| Q_confidence | min(C_avg_extraction / 0.8, 1.0) |
|
||||
| Q_freshness | max(0, 1 − t_newest_hours / 168) |
|
||||
| Q_coverage | (N_valid / N_total) × min(N_valid / 10, 1.0) |
|
||||
|
||||
**Suppression triggers** (any one → informational only):
|
||||
|
||||
| Check | Threshold |
|
||||
|---|---|
|
||||
| Avg extraction confidence | < 0.40 |
|
||||
| Evidence staleness | > 168 hours (7 days) |
|
||||
| Source type diversity | < 1 distinct type |
|
||||
| Extraction failure rate | > 50% |
|
||||
| Valid document count | < 2 |
|
||||
| Data quality score | < 0.30 |
|
||||
|
||||
### 6.2 Safety Suppression
|
||||
|
||||
- **Macro-only:** If trend driven solely by macro signals with zero company evidence → forced informational
|
||||
- **Pattern-only:** If trend driven solely by pattern/competitive signals with no company or macro support → forced informational
|
||||
|
||||
---
|
||||
|
||||
## 7. Recommendation Eligibility
|
||||
|
||||
**Source:** `services/recommendation/eligibility.py`
|
||||
|
||||
### 7.1 Gate Checks (all must pass)
|
||||
|
||||
| Check | Threshold |
|
||||
|---|---|
|
||||
| Confidence | ≥ 0.35 |
|
||||
| Trend strength | ≥ 0.10 |
|
||||
| Contradiction score | ≤ 0.60 |
|
||||
| Evidence count | ≥ 2 |
|
||||
| Direction | ≠ neutral |
|
||||
|
||||
### 7.2 Action Mapping
|
||||
|
||||
| Condition | Action |
|
||||
|---|---|
|
||||
| Bullish AND strength ≥ 0.25 | BUY |
|
||||
| Bearish AND strength ≥ 0.25 | SELL |
|
||||
| Directional AND confidence ≥ 0.50 | HOLD |
|
||||
| Mixed or weak | WATCH |
|
||||
|
||||
### 7.3 Mode Escalation
|
||||
|
||||
| Mode | Requirements |
|
||||
|---|---|
|
||||
| live_eligible | confidence ≥ 0.70, contradiction ≤ 0.25, evidence ≥ 5 |
|
||||
| paper_eligible | confidence ≥ 0.50 |
|
||||
| informational | everything else (WATCH/HOLD always informational) |
|
||||
|
||||
### 7B. Expected Value Gate (Probabilistic)
|
||||
|
||||
**Active when:** `probabilistic_scoring_enabled = true`
|
||||
|
||||
```
|
||||
R_up = strength × σ_20 × √(horizon_days)
|
||||
R_down = (1 − strength) × σ_20 × √(horizon_days)
|
||||
EV = P_bull × R_up − (1 − P_bull) × R_down
|
||||
```
|
||||
|
||||
| Horizon window | horizon_days |
|
||||
|---|---|
|
||||
| intraday / 1d | 1 |
|
||||
| 7d | 7 |
|
||||
| 30d | 30 |
|
||||
| 90d | 90 |
|
||||
|
||||
- EV > 0.005 (0.5% expected return): recommendation proceeds through existing gates
|
||||
- EV ≤ 0.005: forced to informational mode regardless of confidence/strength
|
||||
- All existing eligibility gates (§7.1) remain as additional requirements
|
||||
|
||||
### 7.4 Position Sizing
|
||||
|
||||
```
|
||||
portfolio_pct = base + C_factor × S_factor × range × P_contradiction × P_evidence
|
||||
```
|
||||
|
||||
| Component | Formula | Default |
|
||||
|---|---|---|
|
||||
| base | base_portfolio_pct | 0.01 (1%) |
|
||||
| range | max_portfolio_pct − base_portfolio_pct | 0.09 (9%) |
|
||||
| C_factor | confidence_sizing_weight × confidence | 0.8 × confidence |
|
||||
| S_factor | 0.5 + 0.5 × trend_strength | [0.5, 1.0] |
|
||||
| P_contradiction | 1 − (contradiction_penalty × contradiction_score) | penalty = 0.5 |
|
||||
| P_evidence | 0.50 if evidence < 3; 0.75 if evidence < 5; 1.0 otherwise | |
|
||||
|
||||
Clamped to [base × 0.5, max_portfolio_pct].
|
||||
|
||||
**Max loss percentage** uses the same structure with base = 0.003 (0.3%) and max = 0.02 (2%).
|
||||
|
||||
---
|
||||
|
||||
## 8. Trading Engine — Position Sizing
|
||||
|
||||
**Source:** `services/trading/position_sizer.py`
|
||||
|
||||
### 8.1 Base Allocation
|
||||
|
||||
```
|
||||
raw_pct = (max_position_pct × 0.5) × (confidence / min_confidence) × multiplier
|
||||
clamped_pct = min(raw_pct, max_position_pct)
|
||||
dollar_amount = min(active_pool × clamped_pct, absolute_position_cap)
|
||||
```
|
||||
|
||||
### 8.2 Correlation Reduction
|
||||
|
||||
```
|
||||
ρ_avg = Σ(ρ_i × w_i) / Σ(w_i) for existing positions
|
||||
```
|
||||
|
||||
| ρ_avg | Action |
|
||||
|---|---|
|
||||
| > 0.8 | Reject order |
|
||||
| 0.5 < ρ_avg ≤ 0.8 | Reduce: factor = 1 − (ρ_avg − 0.5) / 0.3 |
|
||||
| ≤ 0.5 | No reduction |
|
||||
|
||||
### 8.3 Sector Exposure Reduction
|
||||
|
||||
```
|
||||
available = max(max_sector_pct × active_pool − current_sector_exposure, 0)
|
||||
dollar_amount = min(dollar_amount, available)
|
||||
```
|
||||
|
||||
### 8.4 Diversification Bonus
|
||||
|
||||
If < 3 sectors held AND entering a new sector: dollar_amount × 1.2 (capped at max_position_pct).
|
||||
|
||||
### 8.5 Earnings Proximity
|
||||
|
||||
| Days to earnings | Action |
|
||||
|---|---|
|
||||
| ≤ 1 | Reject |
|
||||
| 1–3 | 50% reduction |
|
||||
| > 3 | No adjustment |
|
||||
|
||||
### 8.6 Portfolio Heat Check
|
||||
|
||||
```
|
||||
heat_new = dollar_amount × atr_multiplier × 0.02
|
||||
heat_max = max_portfolio_heat × active_pool
|
||||
|
||||
Reject if: heat_current + heat_new > heat_max
|
||||
```
|
||||
|
||||
### 8.7 Share Rounding
|
||||
|
||||
```
|
||||
shares = floor(dollar_amount / current_price)
|
||||
final_dollar = shares × current_price
|
||||
```
|
||||
|
||||
Reject if shares = 0.
|
||||
|
||||
---
|
||||
|
||||
## 9. Stop-Loss and Take-Profit
|
||||
|
||||
**Source:** `services/trading/stop_loss_manager.py`
|
||||
|
||||
### 9.1 Initial Levels
|
||||
|
||||
```
|
||||
stop_distance = ATR × M_atr
|
||||
stop_loss = entry_price − stop_distance
|
||||
take_profit = entry_price + stop_distance × R_reward_risk
|
||||
```
|
||||
|
||||
| Trade type | M_atr | R_reward_risk |
|
||||
|---|---|---|
|
||||
| Standard | risk_tier.stop_loss_atr_multiplier | risk_tier.reward_risk_ratio |
|
||||
| Micro-trade | 1.0 | 1.5 |
|
||||
|
||||
### 9.2 Dynamic Tightening
|
||||
|
||||
| Condition | Effective multiplier |
|
||||
|---|---|
|
||||
| High-severity macro event | base × 0.5 |
|
||||
| Earnings within 3 days | base × 0.7 |
|
||||
| Portfolio heat > 80% of max | base × 0.7 |
|
||||
| Normal | base |
|
||||
|
||||
### 9.3 Trailing Stop Activation
|
||||
|
||||
Activates when:
|
||||
```
|
||||
favorable_move = current_price − entry_price > 0.5 × (take_profit − entry_price)
|
||||
```
|
||||
|
||||
Once active, stop-loss floor = entry_price (breakeven).
|
||||
|
||||
---
|
||||
|
||||
## 10. Risk Management
|
||||
|
||||
### 10.1 Position Limits
|
||||
|
||||
**Source:** `services/risk/engine.py`
|
||||
|
||||
| Limit | Default | Formula |
|
||||
|---|---|---|
|
||||
| Max position % | 5% | position_value / portfolio_value ≤ 0.05 |
|
||||
| Max position value | $10,000 | existing + new ≤ $10,000 |
|
||||
| Max shares/order | 1,000 | quantity ≤ 1,000 |
|
||||
| Max sector % | 25% | sector_value / portfolio_value ≤ 0.25 |
|
||||
| Max daily loss % | 2% | |daily_pnl| / portfolio_value ≤ 0.02 |
|
||||
| Max daily loss $ | $1,000 | |daily_pnl| ≤ $1,000 |
|
||||
| Max daily trades | 20 | trade_count < 20 |
|
||||
|
||||
### 10.2 Order Clamping
|
||||
|
||||
**Source:** `services/risk/engine.py` — `clamp_order_to_position_limits()`
|
||||
|
||||
When a buy order exceeds position limits, instead of rejecting:
|
||||
|
||||
```
|
||||
max_allowed_value = min(
|
||||
max_position_value − existing_value,
|
||||
max_position_pct × portfolio_value − existing_value
|
||||
)
|
||||
clamped_shares = min( floor(max_allowed_value / price_per_share), max_shares_per_order )
|
||||
```
|
||||
|
||||
### 10.3 News Shock Lockout
|
||||
|
||||
Trigger: impact_score ≥ 0.80 for catalyst ∈ {earnings, legal, m_and_a}
|
||||
Duration: 60 minutes (configurable)
|
||||
|
||||
### 10.4 Symbol Cooldown
|
||||
|
||||
Duration: 15 minutes between trades on same symbol.
|
||||
Max concurrent positions per symbol: 1.
|
||||
|
||||
---
|
||||
|
||||
## 11. Circuit Breaker
|
||||
|
||||
**Source:** `services/trading/circuit_breaker.py`
|
||||
|
||||
| Trigger | Condition | Cooldown |
|
||||
|---|---|---|
|
||||
| Daily loss | |daily_pnl| / portfolio_value > 0.05 | 2 hours |
|
||||
| Single position | position_loss_pct > 0.15 | 48 hours |
|
||||
| Volatility | ≥ 3 stop-losses within 30-minute window | 2 hours |
|
||||
|
||||
---
|
||||
|
||||
## 12. Risk Tier Auto-Adjustment
|
||||
|
||||
**Source:** `services/trading/risk_tier_controller.py`
|
||||
|
||||
Tiers: conservative → moderate → aggressive
|
||||
|
||||
**Downgrade** (any one triggers, drops one level):
|
||||
- 30-day win rate < 40%
|
||||
- Current drawdown > 15%
|
||||
|
||||
**Upgrade** (all must be true, raises one level):
|
||||
- 30-day win rate > 55%
|
||||
- Reserve pool > 20% of portfolio
|
||||
- Current drawdown < 5%
|
||||
|
||||
---
|
||||
|
||||
## 13. Portfolio Rebalancing
|
||||
|
||||
**Source:** `services/trading/rebalancer.py`
|
||||
|
||||
### 13.1 Single-Stock Rebalancing
|
||||
|
||||
```
|
||||
excess = market_value − max_position_pct × active_pool
|
||||
sell_qty = min( floor(excess / current_price), position_quantity )
|
||||
```
|
||||
|
||||
### 13.2 Sector Rebalancing
|
||||
|
||||
```
|
||||
sector_excess = Σ(market_value_i) − max_sector_pct × active_pool
|
||||
```
|
||||
|
||||
Sell from lowest-confidence positions first until excess is covered.
|
||||
|
||||
### 13.3 Max Positions Enforcement
|
||||
|
||||
```
|
||||
excess_count = N_positions − max_positions
|
||||
```
|
||||
|
||||
Sell entire lowest-confidence positions until count is within limit.
|
||||
|
||||
---
|
||||
|
||||
## Constants Summary
|
||||
|
||||
| Constant | Value | Location |
|
||||
|---|---|---|
|
||||
| Confidence gate floor | 0.20 | scoring.py |
|
||||
| Min recency weight | 0.01 | scoring.py |
|
||||
| Credibility floor/ceiling | 0.10 / 1.0 | scoring.py |
|
||||
| Novelty bonus max | 0.25 (25%) | scoring.py |
|
||||
| Volatility boost threshold | 1.0 price units | scoring.py |
|
||||
| Volatility boost max | 0.30 (30%) | scoring.py |
|
||||
| Volume surge threshold | 50% | scoring.py |
|
||||
| Volume surge boost | 0.15 (15%) | scoring.py |
|
||||
| Bullish/bearish threshold | ±0.15 | worker.py |
|
||||
| Mixed threshold | contradiction > 0.10, |S| < 0.30 | worker.py |
|
||||
| Macro signal weight | 0.30 | config.py |
|
||||
| Competitive signal weight | 0.20 | config.py |
|
||||
| Macro confidence threshold | 0.40 | interpolation.py |
|
||||
| Staleness accelerated decay | 0.50× | interpolation.py |
|
||||
| Short-term staleness hours | 48 | interpolation.py |
|
||||
| Pattern min samples | configurable | pattern_matcher.py |
|
||||
| Major decision weight multiplier | 1.3× | pattern_matcher.py |
|
||||
| Routine lookback | 180 days | pattern_matcher.py |
|
||||
| Major decision lookback | 365 days | pattern_matcher.py |
|
||||
| Propagation strength threshold | 0.20 | signal_propagation.py |
|
||||
| Data quality min score | 0.30 | suppression.py |
|
||||
| Evidence staleness max | 168 hours (7 days) | suppression.py |
|
||||
| Recommendation min confidence | 0.35 | eligibility.py |
|
||||
| Recommendation min strength | 0.10 | eligibility.py |
|
||||
| Action strength threshold | 0.25 | eligibility.py |
|
||||
| Live confidence threshold | 0.70 | eligibility.py |
|
||||
| Paper confidence threshold | 0.50 | eligibility.py |
|
||||
| Base portfolio allocation | 1% | eligibility.py |
|
||||
| Max portfolio allocation | 10% | eligibility.py |
|
||||
| Circuit breaker daily loss | 5% | circuit_breaker.py |
|
||||
| Circuit breaker single position | 15% | circuit_breaker.py |
|
||||
| Stop-loss cluster threshold | 3 hits / 30 min | circuit_breaker.py |
|
||||
| Tier downgrade win rate | < 40% | risk_tier_controller.py |
|
||||
| Tier upgrade win rate | > 55% | risk_tier_controller.py |
|
||||
| Tier upgrade max drawdown | < 5% | risk_tier_controller.py |
|
||||
| Tier upgrade min reserve | > 20% | risk_tier_controller.py |
|
||||
| **Probabilistic pipeline** | | |
|
||||
| Sigmoid steepness (k) | 5.0 | scoring.py |
|
||||
| Sigmoid midpoint (m) | 0.5 | scoring.py |
|
||||
| Info gain lambda (λ) | 0.3 | scoring.py |
|
||||
| Info gain max clamp | 3.0 | scoring.py |
|
||||
| Default base rate | 0.10 | scoring.py |
|
||||
| Adaptive decay impact scale | 1.0 | scoring.py |
|
||||
| Adaptive decay surprise scale | 1.0 | scoring.py |
|
||||
| Adaptive decay market scale | 0.5 | scoring.py |
|
||||
| Regime return weight | 0.15 | scoring.py |
|
||||
| Regime volume weight | 0.10 | scoring.py |
|
||||
| Regime multiplier max | 2.5 | scoring.py |
|
||||
| Source accuracy min samples | 10 | source_accuracy.py |
|
||||
| Contradiction W_threshold | 5.0 | contradiction.py |
|
||||
| EMA short period | 20 days | regime.py |
|
||||
| EMA long period | 100 days | regime.py |
|
||||
| Panic volatility ratio | > 1.5 | regime.py |
|
||||
| Trend-following vol ratio | < 1.2 | regime.py |
|
||||
| Mean-reversion vol ratio | < 1.0 | regime.py |
|
||||
| Panic threshold | ±0.10 | regime.py |
|
||||
| Mean-reversion threshold | ±0.20 | regime.py |
|
||||
| Uncertainty contradiction mult | 0.6 | regime.py |
|
||||
| EW momentum decay (λ) | 0.7 | projection.py |
|
||||
| EW momentum max lags (K) | 10 | projection.py |
|
||||
| Volatility floor (σ min) | 0.01 | projection.py |
|
||||
| Momentum clamp | ±2.0 | projection.py |
|
||||
| EV threshold | 0.005 (0.5%) | eligibility.py |
|
||||
| Graph distance max | 3 | signal_propagation.py |
|
||||
| Default correlation (same-sector) | 0.3 | signal_propagation.py |
|
||||
| Default correlation (cross-sector) | 0.1 | signal_propagation.py |
|
||||
@@ -0,0 +1,677 @@
|
||||
# Helm Chart Configuration Reference
|
||||
|
||||
Complete reference for the Stonks Oracle Helm chart at `infra/helm/stonks-oracle/`.
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| **Chart name** | `stonks-oracle` |
|
||||
| **Chart version** | `0.1.0` |
|
||||
| **App version** | `1.0.0` |
|
||||
| **Chart type** | `application` |
|
||||
|
||||
Install with:
|
||||
|
||||
```bash
|
||||
helm upgrade --install stonks-oracle infra/helm/stonks-oracle -n stonks-oracle
|
||||
```
|
||||
|
||||
Override values per stage:
|
||||
|
||||
```bash
|
||||
# Beta
|
||||
helm upgrade --install stonks-oracle infra/helm/stonks-oracle \
|
||||
-n stonks-oracle-beta -f infra/helm/stonks-oracle/values-beta.yaml
|
||||
|
||||
# Paper trading
|
||||
helm upgrade --install stonks-oracle infra/helm/stonks-oracle \
|
||||
-n stonks-oracle -f infra/helm/stonks-oracle/values-paper.yaml
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Table of Contents
|
||||
|
||||
- [image — Global Image Settings](#image--global-image-settings)
|
||||
- [pipelineEnabled — Pipeline Toggle](#pipelineenabled--pipeline-toggle)
|
||||
- [services — Service Deployments](#services--service-deployments)
|
||||
- [config — ConfigMap Environment Variables](#config--configmap-environment-variables)
|
||||
- [secrets — Kubernetes Secrets](#secrets--kubernetes-secrets)
|
||||
- [ingress — Ingress Configuration](#ingress--ingress-configuration)
|
||||
- [Analytics Stack — Trino, Hive Metastore, Superset](#analytics-stack--trino-hive-metastore-superset)
|
||||
- [networkPolicies — Network Policy Configuration](#networkpolicies--network-policy-configuration)
|
||||
- [Value Override Files](#value-override-files)
|
||||
|
||||
---
|
||||
|
||||
## `image` — Global Image Settings
|
||||
|
||||
Controls the container image registry, pull policy, and tag for all service deployments. Each service image is resolved as `{registry}/{service.image}:{tag}`.
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `image.registry` | string | `registry.celestium.life/stonks-oracle` | Container registry prefix. Each service appends its `image` name to this. |
|
||||
| `image.pullPolicy` | string | `Always` | Kubernetes `imagePullPolicy`. Use `Always` for latest-tag workflows. |
|
||||
| `image.tag` | string | `latest` | Image tag applied to all services. CI overrides this with the Git SHA via `--set image.tag=<sha>`. |
|
||||
|
||||
Example override:
|
||||
|
||||
```bash
|
||||
helm upgrade --install stonks-oracle infra/helm/stonks-oracle \
|
||||
--set image.tag=abc1234
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## `pipelineEnabled` — Pipeline Toggle
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `pipelineEnabled` | bool | `true` | Master toggle for the data pipeline. |
|
||||
|
||||
When `false`, all services with `pipeline: true` in their definition are scaled to **0 replicas**. API-tier and trading-tier services continue running normally.
|
||||
|
||||
**Affected services** (scaled to 0 when disabled): scheduler, ingestion, parser, extractor, aggregation, recommendation, broker-adapter, lake-publisher.
|
||||
|
||||
**Unaffected services** (always run): symbol-registry, query-api, trading-engine, risk-engine, dashboard.
|
||||
|
||||
The replica count logic in the deployment template:
|
||||
|
||||
```yaml
|
||||
replicas: {{ if and (hasKey $svc "pipeline") $svc.pipeline (not .Values.pipelineEnabled) }}0{{ else }}{{ $svc.replicas }}{{ end }}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## `services` — Service Deployments
|
||||
|
||||
Each key under `services` defines a Kubernetes Deployment. The deployments template iterates over all entries and creates a Deployment + optional Service for each.
|
||||
|
||||
### Per-Service Structure
|
||||
|
||||
| Field | Type | Required | Description |
|
||||
|-------|------|----------|-------------|
|
||||
| `replicas` | int | yes | Number of pod replicas. Set to 0 by `pipelineEnabled: false` for pipeline services. |
|
||||
| `image` | string | yes | Image name appended to `image.registry`. Also used as the Deployment name and pod label (`app: <image>`). |
|
||||
| `command` | string | no | Shell command passed as `["sh", "-c", "<command>"]`. Omit for images with a built-in entrypoint (e.g., dashboard/nginx). |
|
||||
| `tier` | string | yes | Service tier label (`stonks-oracle/tier`). One of: `api`, `frontend`, `processing`, `trading`, `orchestration`, `analytics`, `ingestion`. |
|
||||
| `port` | int | no | Container port. When set, a Kubernetes Service is created mapping `port -> port`. |
|
||||
| `pipeline` | bool | no | If `true`, replicas are set to 0 when `pipelineEnabled` is `false`. |
|
||||
| `secrets` | list(string) | no | List of Secret names to mount via `envFrom.secretRef`. |
|
||||
| `resources` | object | yes | Kubernetes resource requests and limits (`cpu`, `memory`). |
|
||||
| `probes.readiness` | object | no | HTTP readiness probe: `path`, `port`, `initialDelay`, `period`. |
|
||||
| `probes.liveness` | object | no | HTTP liveness probe: `path`, `port`, `initialDelay`, `period`. |
|
||||
|
||||
### Service Definitions
|
||||
|
||||
#### scheduler
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| `replicas` | `1` |
|
||||
| `pipeline` | `true` |
|
||||
| `image` | `scheduler` |
|
||||
| `command` | `python -m services.scheduler.app` |
|
||||
| `tier` | `orchestration` |
|
||||
| `port` | — |
|
||||
| `secrets` | `stonks-core-secrets` |
|
||||
| `resources.requests` | cpu: 50m, memory: 64Mi |
|
||||
| `resources.limits` | cpu: 200m, memory: 128Mi |
|
||||
| `probes` | — |
|
||||
|
||||
The scheduler deployment has three init containers (not configurable via values):
|
||||
1. **run-migrations** — applies all SQL files from `infra/migrations/*.sql` in sorted order.
|
||||
2. **seed-if-empty** — runs `python -m services.symbol_registry.seed` if the `companies` table is empty.
|
||||
3. **backfill-market-data** — runs `scripts/backfill_market_data.py` if available (skips gracefully if not).
|
||||
|
||||
#### symbolRegistry
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| `replicas` | `1` |
|
||||
| `image` | `symbol-registry` |
|
||||
| `command` | `uvicorn services.symbol_registry.app:app --host 0.0.0.0 --port 8000` |
|
||||
| `tier` | `api` |
|
||||
| `port` | `8000` |
|
||||
| `secrets` | `stonks-core-secrets` |
|
||||
| `resources.requests` | cpu: 100m, memory: 128Mi |
|
||||
| `resources.limits` | cpu: 500m, memory: 256Mi |
|
||||
| `probes.readiness` | path: `/docs`, port: 8000, initialDelay: 5s, period: 10s |
|
||||
| `probes.liveness` | path: `/docs`, port: 8000, initialDelay: 10s, period: 30s |
|
||||
|
||||
#### ingestion
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| `replicas` | `1` |
|
||||
| `pipeline` | `true` |
|
||||
| `image` | `ingestion` |
|
||||
| `command` | `python -m services.ingestion.worker` |
|
||||
| `tier` | `ingestion` |
|
||||
| `port` | — |
|
||||
| `secrets` | `stonks-core-secrets`, `stonks-market-secrets`, `stonks-broker-secrets` |
|
||||
| `resources.requests` | cpu: 100m, memory: 128Mi |
|
||||
| `resources.limits` | cpu: 500m, memory: 256Mi |
|
||||
|
||||
#### parser
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| `replicas` | `2` |
|
||||
| `pipeline` | `true` |
|
||||
| `image` | `parser` |
|
||||
| `command` | `python -m services.parser.worker` |
|
||||
| `tier` | `processing` |
|
||||
| `port` | — |
|
||||
| `secrets` | `stonks-core-secrets` |
|
||||
| `resources.requests` | cpu: 100m, memory: 128Mi |
|
||||
| `resources.limits` | cpu: 500m, memory: 256Mi |
|
||||
|
||||
#### extractor
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| `replicas` | `1` |
|
||||
| `pipeline` | `true` |
|
||||
| `image` | `extractor` |
|
||||
| `command` | `python -m services.extractor.main` |
|
||||
| `tier` | `processing` |
|
||||
| `port` | — |
|
||||
| `secrets` | `stonks-core-secrets` |
|
||||
| `resources.requests` | cpu: 200m, memory: 256Mi |
|
||||
| `resources.limits` | cpu: 1, memory: 512Mi |
|
||||
|
||||
Single replica is recommended — the extractor is bottlenecked by the shared Ollama GPU.
|
||||
|
||||
#### aggregation
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| `replicas` | `4` |
|
||||
| `pipeline` | `true` |
|
||||
| `image` | `aggregation` |
|
||||
| `command` | `python -m services.aggregation.main` |
|
||||
| `tier` | `processing` |
|
||||
| `port` | — |
|
||||
| `secrets` | `stonks-core-secrets` |
|
||||
| `resources.requests` | cpu: 100m, memory: 128Mi |
|
||||
| `resources.limits` | cpu: 500m, memory: 256Mi |
|
||||
|
||||
#### recommendation
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| `replicas` | `1` |
|
||||
| `pipeline` | `true` |
|
||||
| `image` | `recommendation` |
|
||||
| `command` | `python -m services.recommendation.main` |
|
||||
| `tier` | `processing` |
|
||||
| `port` | — |
|
||||
| `secrets` | `stonks-core-secrets` |
|
||||
| `resources.requests` | cpu: 100m, memory: 128Mi |
|
||||
| `resources.limits` | cpu: 500m, memory: 256Mi |
|
||||
|
||||
#### tradingEngine
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| `replicas` | `1` |
|
||||
| `image` | `trading-engine` |
|
||||
| `command` | `uvicorn services.trading.app:app --host 0.0.0.0 --port 8000` |
|
||||
| `tier` | `trading` |
|
||||
| `port` | `8000` |
|
||||
| `secrets` | `stonks-core-secrets`, `stonks-broker-secrets`, `stonks-gmail-secrets` |
|
||||
| `resources.requests` | cpu: 100m, memory: 256Mi |
|
||||
| `resources.limits` | cpu: 500m, memory: 512Mi |
|
||||
| `probes.readiness` | path: `/ready`, port: 8000, initialDelay: 5s, period: 10s |
|
||||
| `probes.liveness` | path: `/health`, port: 8000, initialDelay: 10s, period: 30s |
|
||||
|
||||
#### riskEngine
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| `replicas` | `1` |
|
||||
| `image` | `risk` |
|
||||
| `command` | `uvicorn services.risk.app:app --host 0.0.0.0 --port 8000` |
|
||||
| `tier` | `trading` |
|
||||
| `port` | `8000` |
|
||||
| `secrets` | `stonks-core-secrets`, `stonks-broker-secrets` |
|
||||
| `resources.requests` | cpu: 100m, memory: 128Mi |
|
||||
| `resources.limits` | cpu: 500m, memory: 256Mi |
|
||||
|
||||
#### brokerAdapter
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| `replicas` | `1` |
|
||||
| `pipeline` | `true` |
|
||||
| `image` | `broker-adapter` |
|
||||
| `command` | `python -m services.adapters.broker_service` |
|
||||
| `tier` | `trading` |
|
||||
| `port` | — |
|
||||
| `secrets` | `stonks-core-secrets`, `stonks-broker-secrets` |
|
||||
| `resources.requests` | cpu: 50m, memory: 64Mi |
|
||||
| `resources.limits` | cpu: 200m, memory: 128Mi |
|
||||
|
||||
#### lakePublisher
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| `replicas` | `1` |
|
||||
| `pipeline` | `true` |
|
||||
| `image` | `lake-publisher` |
|
||||
| `command` | `python -m services.lake_publisher.jobs` |
|
||||
| `tier` | `analytics` |
|
||||
| `port` | — |
|
||||
| `secrets` | `stonks-core-secrets` |
|
||||
| `resources.requests` | cpu: 100m, memory: 128Mi |
|
||||
| `resources.limits` | cpu: 500m, memory: 256Mi |
|
||||
|
||||
#### queryApi
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| `replicas` | `1` |
|
||||
| `image` | `query-api` |
|
||||
| `command` | `uvicorn services.api.app:app --host 0.0.0.0 --port 8000` |
|
||||
| `tier` | `api` |
|
||||
| `port` | `8000` |
|
||||
| `secrets` | `stonks-core-secrets`, `stonks-market-secrets` |
|
||||
| `resources.requests` | cpu: 100m, memory: 128Mi |
|
||||
| `resources.limits` | cpu: 500m, memory: 256Mi |
|
||||
| `probes.readiness` | path: `/docs`, port: 8000, initialDelay: 5s, period: 10s |
|
||||
|
||||
#### dashboard
|
||||
|
||||
| Field | Value |
|
||||
|-------|-------|
|
||||
| `replicas` | `1` |
|
||||
| `image` | `dashboard` |
|
||||
| `command` | — (nginx built-in entrypoint) |
|
||||
| `tier` | `frontend` |
|
||||
| `port` | `8080` |
|
||||
| `secrets` | — |
|
||||
| `resources.requests` | cpu: 50m, memory: 64Mi |
|
||||
| `resources.limits` | cpu: 200m, memory: 128Mi |
|
||||
| `probes.readiness` | path: `/`, port: 8080, initialDelay: 3s, period: 10s |
|
||||
| `probes.liveness` | path: `/`, port: 8080, initialDelay: 5s, period: 30s |
|
||||
|
||||
---
|
||||
|
||||
## `config` — ConfigMap Environment Variables
|
||||
|
||||
All keys under `config` are rendered into a Kubernetes ConfigMap named `stonks-config` and injected into every service pod via `envFrom.configMapRef`. Values are strings.
|
||||
|
||||
### Database
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `config.POSTGRES_HOST` | string | `postgresql-rw.postgresql-service.svc.cluster.local` | PostgreSQL hostname. Points to the CloudNativePG read-write service. |
|
||||
| `config.POSTGRES_PORT` | string | `5432` | PostgreSQL port. |
|
||||
| `config.POSTGRES_DB` | string | `stonks` | Database name. Override per stage (e.g., `stonks_beta`, `stonks_paper`). |
|
||||
| `config.POSTGRES_USER` | string | `stonks` | Database user. Override per stage. |
|
||||
| `config.REDIS_HOST` | string | `redis-master.redis-service.svc.cluster.local` | Redis hostname. |
|
||||
| `config.REDIS_PORT` | string | `6379` | Redis port. |
|
||||
| `config.REDIS_DB` | string | `0` | Redis database index. Use different indices per stage to isolate keys (beta: `1`, paper: `2`). |
|
||||
|
||||
### Object Storage
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `config.MINIO_ENDPOINT` | string | `minio.minio-service.svc.cluster.local:80` | MinIO API endpoint (host:port). |
|
||||
| `config.MINIO_SECURE` | string | `false` | Use HTTPS for MinIO connections. Set to `true` if MinIO has TLS. |
|
||||
|
||||
### LLM / Ollama
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `config.OLLAMA_BASE_URL` | string | `http://10.1.1.12:2701` | Ollama API base URL. Points to the external Ollama endpoint by default. |
|
||||
| `config.OLLAMA_MODEL` | string | `qwen3.5:9b-fast` | Default LLM model for extraction and classification agents. |
|
||||
| `config.OLLAMA_TIMEOUT` | string | `240` | Request timeout in seconds for Ollama API calls. |
|
||||
| `config.OLLAMA_MAX_RETRIES` | string | `2` | Maximum retry attempts for failed Ollama requests. |
|
||||
| `config.OLLAMA_RETRY_BASE_DELAY` | string | `1.0` | Base delay in seconds for exponential backoff on Ollama retries. |
|
||||
| `config.OLLAMA_RETRY_MAX_DELAY` | string | `10.0` | Maximum delay cap in seconds for Ollama retry backoff. |
|
||||
| `config.OLLAMA_RETRY_BACKOFF_MULTIPLIER` | string | `2.0` | Multiplier for exponential backoff between Ollama retries. |
|
||||
|
||||
### vLLM
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `config.VLLM_BASE_URL` | string | `http://10.1.1.12:2701` | vLLM API base URL. Alternative LLM backend using OpenAI-compatible API. |
|
||||
| `config.VLLM_MODEL` | string | `qwen3.5:9b-fast` | vLLM model identifier. |
|
||||
| `config.VLLM_TIMEOUT` | string | `120` | Request timeout in seconds for vLLM API calls. |
|
||||
| `config.VLLM_MAX_RETRIES` | string | `2` | Maximum retry attempts for failed vLLM requests. |
|
||||
| `config.VLLM_TEMPERATURE` | string | `0.7` | Sampling temperature for vLLM generation (0.0-1.0). |
|
||||
| `config.VLLM_API_KEY` | string | `""` (empty) | API key for vLLM authentication. Leave empty if not required. |
|
||||
|
||||
### Analytics / Trino
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `config.TRINO_HOST` | string | `trino.stonks-oracle.svc.cluster.local` | Trino coordinator hostname. |
|
||||
| `config.TRINO_PORT` | string | `8080` | Trino coordinator port. |
|
||||
| `config.TRINO_CATALOG` | string | `lakehouse` | Default Trino catalog for Hive-based queries. |
|
||||
| `config.TRINO_SCHEMA` | string | `stonks` | Default Trino schema. |
|
||||
| `config.TRINO_ICEBERG_CATALOG` | string | `iceberg` | Trino catalog for Iceberg table queries. |
|
||||
|
||||
### Broker / Trading
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `config.BROKER_MODE` | string | `paper` | Broker execution mode. `paper` for simulated trading, `live` for real orders. |
|
||||
| `config.BROKER_PROVIDER` | string | `""` (empty) | Broker provider name (e.g., `alpaca`). |
|
||||
| `config.MARKET_DATA_BASE_URL` | string | `https://api.polygon.io` | Market data API base URL. |
|
||||
| `config.MARKET_DATA_PROVIDER` | string | `polygon` | Market data provider identifier. |
|
||||
| `config.TRADING_ENABLED` | string | `true` | Master toggle for the trading engine. Set to `false` to disable order submission. |
|
||||
| `config.TRADING_RISK_TIER` | string | `moderate` | Default risk tier for position sizing. Options: `conservative`, `moderate`, `aggressive`. |
|
||||
| `config.TRADING_ABSOLUTE_POSITION_CAP` | string | `10000.0` | Maximum dollar value per position. |
|
||||
| `config.TRADING_MAX_OPEN_POSITIONS` | string | `10` | Maximum number of concurrent open positions. |
|
||||
|
||||
### Data Retention
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `config.RETENTION_RAW_MARKET_DAYS` | string | `90` | Days to retain raw market data before cleanup. |
|
||||
| `config.RETENTION_RAW_NEWS_DAYS` | string | `180` | Days to retain raw news articles. |
|
||||
| `config.RETENTION_RAW_FILINGS_DAYS` | string | `365` | Days to retain raw SEC filings. |
|
||||
| `config.RETENTION_NORMALIZED_DAYS` | string | `180` | Days to retain normalized/parsed documents. |
|
||||
| `config.RETENTION_LLM_PROMPTS_DAYS` | string | `365` | Days to retain LLM prompt logs. |
|
||||
| `config.RETENTION_LLM_RESULTS_DAYS` | string | `365` | Days to retain LLM extraction results. |
|
||||
| `config.RETENTION_LAKEHOUSE_DAYS` | string | `730` | Days to retain lakehouse fact tables. |
|
||||
| `config.RETENTION_AUDIT_DAYS` | string | `730` | Days to retain audit trail events. |
|
||||
| `config.RETENTION_CLEANUP_INTERVAL_HOURS` | string | `24` | Hours between retention cleanup runs. |
|
||||
| `config.RETENTION_BATCH_SIZE` | string | `1000` | Number of rows deleted per cleanup batch. |
|
||||
|
||||
### Logging and Deployment
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `config.LOG_LEVEL` | string | `INFO` | Python logging level. Options: `DEBUG`, `INFO`, `WARNING`, `ERROR`. |
|
||||
| `config.JSON_LOGS` | string | `true` | Emit structured JSON logs when `true`. |
|
||||
| `config.DEPLOY_STAGE` | string | `""` (empty) | Deployment stage identifier. Used to isolate Redis keys and MinIO buckets per stage (e.g., `beta`, `paper`). |
|
||||
| `config.TZ` | string | `America/Los_Angeles` | Container timezone. Affects log timestamps and any time-aware formatting. The frontend uses the browser's local timezone for display. |
|
||||
|
||||
### Alerting
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `config.ALERT_SOURCE_FAILURE_THRESHOLD` | string | `3` | Number of consecutive source failures before firing an alert. |
|
||||
| `config.ALERT_SOURCE_FAILURE_WINDOW_HOURS` | string | `6` | Time window (hours) for evaluating source failure count. |
|
||||
| `config.ALERT_SCHEMA_FAILURE_RATE_THRESHOLD` | string | `0.3` | Schema validation failure rate (0.0-1.0) that triggers an alert. |
|
||||
| `config.ALERT_SCHEMA_FAILURE_WINDOW_HOURS` | string | `1` | Time window (hours) for evaluating schema failure rate. |
|
||||
| `config.ALERT_LAKE_LAG_THRESHOLD_MINUTES` | string | `60` | Minutes of lakehouse publish lag before alerting. |
|
||||
| `config.ALERT_BROKER_ERROR_THRESHOLD` | string | `3` | Number of broker errors before firing an alert. |
|
||||
| `config.ALERT_BROKER_ERROR_WINDOW_HOURS` | string | `1` | Time window (hours) for evaluating broker error count. |
|
||||
| `config.ALERT_CHECK_INTERVAL_SECONDS` | string | `120` | Seconds between alert evaluation cycles. |
|
||||
|
||||
---
|
||||
|
||||
## `secrets` — Kubernetes Secrets
|
||||
|
||||
Secrets are rendered into five Kubernetes Secret objects. Inject real values at deploy time using `--set` flags or a values override file. The base `values.yaml` contains placeholder values — override them for each environment.
|
||||
|
||||
### Secret Objects
|
||||
|
||||
| Secret Name | Values Key | Consumed By |
|
||||
|-------------|-----------|-------------|
|
||||
| `stonks-core-secrets` | `secrets.core` | All services |
|
||||
| `stonks-broker-secrets` | `secrets.broker` | ingestion, trading-engine, risk-engine, broker-adapter |
|
||||
| `stonks-market-secrets` | `secrets.market` | ingestion, query-api |
|
||||
| `stonks-gmail-secrets` | `secrets.gmail` | trading-engine |
|
||||
| `stonks-dashboard-secrets` | `secrets.dashboard` | superset |
|
||||
|
||||
### `secrets.core`
|
||||
|
||||
| Key | Type | Description |
|
||||
|-----|------|-------------|
|
||||
| `POSTGRES_PASSWORD` | string | PostgreSQL password. |
|
||||
| `MINIO_ACCESS_KEY` | string | MinIO access key (AWS-style). |
|
||||
| `MINIO_SECRET_KEY` | string | MinIO secret key. |
|
||||
| `REDIS_PASSWORD` | string | Redis authentication password. |
|
||||
|
||||
### `secrets.broker`
|
||||
|
||||
| Key | Type | Description |
|
||||
|-----|------|-------------|
|
||||
| `BROKER_API_KEY` | string | Broker API key (e.g., Alpaca paper trading key). |
|
||||
| `BROKER_API_SECRET` | string | Broker API secret. |
|
||||
| `BROKER_BASE_URL` | string | Broker API base URL (e.g., `https://paper-api.alpaca.markets`). |
|
||||
|
||||
### `secrets.market`
|
||||
|
||||
| Key | Type | Description |
|
||||
|-----|------|-------------|
|
||||
| `MARKET_DATA_API_KEY` | string | Market data provider API key (e.g., Polygon.io). |
|
||||
|
||||
### `secrets.gmail`
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `GMAIL_SENDER` | string | `celes@celestium.life` | Gmail sender address for trading notifications. |
|
||||
| `GMAIL_RECIPIENT` | string | `celes@celestium.life` | Gmail recipient address for trading notifications. |
|
||||
| `GMAIL_APP_PASSWORD` | string | `""` | Gmail app password for SMTP authentication. |
|
||||
|
||||
### `secrets.dashboard`
|
||||
|
||||
| Key | Type | Description |
|
||||
|-----|------|-------------|
|
||||
| `SUPERSET_SECRET_KEY` | string | Flask secret key for Superset session encryption. |
|
||||
| `SUPERSET_ADMIN_PASSWORD` | string | Superset admin user password. |
|
||||
|
||||
### Injecting Secrets at Deploy Time
|
||||
|
||||
```bash
|
||||
helm upgrade --install stonks-oracle infra/helm/stonks-oracle \
|
||||
-n stonks-oracle \
|
||||
--set secrets.core.POSTGRES_PASSWORD="<password>" \
|
||||
--set secrets.core.MINIO_ACCESS_KEY="<key>" \
|
||||
--set secrets.core.MINIO_SECRET_KEY="<secret>" \
|
||||
--set secrets.core.REDIS_PASSWORD="<password>" \
|
||||
--set secrets.broker.BROKER_API_KEY="<key>" \
|
||||
--set secrets.broker.BROKER_API_SECRET="<secret>" \
|
||||
--set secrets.broker.BROKER_BASE_URL="https://paper-api.alpaca.markets" \
|
||||
--set secrets.market.MARKET_DATA_API_KEY="<key>" \
|
||||
--set secrets.gmail.GMAIL_APP_PASSWORD="<password>" \
|
||||
--set secrets.dashboard.SUPERSET_SECRET_KEY="<key>" \
|
||||
--set secrets.dashboard.SUPERSET_ADMIN_PASSWORD="<password>"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## `ingress` — Ingress Configuration
|
||||
|
||||
Controls Traefik Ingress resources with TLS via cert-manager.
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `ingress.enabled` | bool | `true` | Create Ingress resources. Set to `false` for port-forward-only access. |
|
||||
| `ingress.className` | string | `traefik` | Kubernetes IngressClass name. |
|
||||
| `ingress.clusterIssuer` | string | `ca-issuer` | cert-manager ClusterIssuer for TLS certificates. |
|
||||
|
||||
### Host Mappings
|
||||
|
||||
| Key | Default | Routes To | Port |
|
||||
|-----|---------|-----------|------|
|
||||
| `ingress.hosts.queryApi` | `stonks-api.celestium.life` | query-api Service | 8000 |
|
||||
| `ingress.hosts.symbolRegistry` | `stonks-registry.celestium.life` | symbol-registry Service | 8000 |
|
||||
| `ingress.hosts.dashboard` | `stonks.celestium.life` | dashboard Service | 8080 |
|
||||
| `ingress.hosts.superset` | `stonks-dash.celestium.life` | superset Service | 8088 |
|
||||
| `ingress.hosts.trino` | `stonks-trino.celestium.life` | trino Service | 8080 |
|
||||
| `ingress.hosts.tradingEngine` | `stonks-trading.celestium.life` | trading-engine Service | 8000 |
|
||||
|
||||
Setting `superset` or `trino` host to an empty string (`""`) disables that Ingress resource (the template uses a conditional check).
|
||||
|
||||
Each Ingress resource gets a dedicated TLS secret (e.g., `stonks-api-tls`, `stonks-registry-tls`) automatically provisioned by cert-manager.
|
||||
|
||||
---
|
||||
|
||||
## Analytics Stack — Trino, Hive Metastore, Superset
|
||||
|
||||
The analytics stack provides SQL-based querying over the lakehouse data stored in MinIO. Each component can be independently enabled or disabled.
|
||||
|
||||
### `trino`
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `trino.enabled` | bool | `true` | Deploy the Trino coordinator. |
|
||||
| `trino.resources.requests.cpu` | string | `500m` | CPU request. |
|
||||
| `trino.resources.requests.memory` | string | `1Gi` | Memory request. |
|
||||
| `trino.resources.limits.cpu` | string | `2` | CPU limit. |
|
||||
| `trino.resources.limits.memory` | string | `4Gi` | Memory limit. |
|
||||
|
||||
When enabled, Trino deploys with two auto-configured catalogs:
|
||||
- **`lakehouse`** — Hive connector for Parquet fact tables in MinIO.
|
||||
- **`iceberg`** — Iceberg connector for Iceberg-format tables.
|
||||
|
||||
Both catalogs connect to the Hive Metastore for schema metadata and to MinIO for data via S3A. MinIO credentials are read from `stonks-core-secrets`.
|
||||
|
||||
### `hiveMetastore`
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `hiveMetastore.enabled` | bool | `true` | Deploy the Hive Metastore. |
|
||||
| `hiveMetastore.storageSize` | string | `1Gi` | PersistentVolumeClaim size for the embedded Derby metastore database. |
|
||||
| `hiveMetastore.resources.requests.cpu` | string | `200m` | CPU request. |
|
||||
| `hiveMetastore.resources.requests.memory` | string | `512Mi` | Memory request. |
|
||||
| `hiveMetastore.resources.limits.cpu` | string | `1` | CPU limit. |
|
||||
| `hiveMetastore.resources.limits.memory` | string | `1Gi` | Memory limit. |
|
||||
|
||||
Uses `apache/hive:4.0.0` with an embedded Derby database. The Thrift metastore listens on port 9083. MinIO credentials are injected from `stonks-core-secrets` via an init container that generates `core-site.xml` and `metastore-site.xml`.
|
||||
|
||||
### `superset`
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `superset.enabled` | bool | `true` | Deploy Apache Superset. |
|
||||
| `superset.storageSize` | string | `2Gi` | PersistentVolumeClaim size for Superset home directory. |
|
||||
| `superset.resources.requests.cpu` | string | `200m` | CPU request. |
|
||||
| `superset.resources.requests.memory` | string | `512Mi` | Memory request. |
|
||||
| `superset.resources.limits.cpu` | string | `1` | CPU limit. |
|
||||
| `superset.resources.limits.memory` | string | `2Gi` | Memory limit. |
|
||||
|
||||
Uses a custom image (`registry.celestium.life/stonks-oracle/superset`) with Trino and psycopg2 drivers pre-installed. Superset's metadata database is PostgreSQL (same cluster instance). Redis is used for caching. Credentials come from `stonks-core-secrets` and `stonks-dashboard-secrets`.
|
||||
|
||||
Superset listens on port 8088 with a readiness probe at `/health`.
|
||||
|
||||
### Disabling the Analytics Stack
|
||||
|
||||
To disable the entire analytics stack (e.g., in beta environments):
|
||||
|
||||
```yaml
|
||||
trino:
|
||||
enabled: false
|
||||
hiveMetastore:
|
||||
enabled: false
|
||||
superset:
|
||||
enabled: false
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## `networkPolicies` — Network Policy Configuration
|
||||
|
||||
| Key | Type | Default | Description |
|
||||
|-----|------|---------|-------------|
|
||||
| `networkPolicies.enabled` | bool | `true` | Deploy NetworkPolicy resources. |
|
||||
|
||||
When enabled, the chart creates a **default-deny-ingress** policy that blocks all inbound traffic to every pod in the namespace. Individual allow policies are then created for services that need ingress:
|
||||
|
||||
| Policy | Target Pod | Allowed Sources | Port |
|
||||
|--------|-----------|-----------------|------|
|
||||
| `allow-query-api-ingress` | `query-api` | kube-system (Traefik), dashboard | 8000 |
|
||||
| `allow-symbol-registry-ingress` | `symbol-registry` | kube-system (Traefik), dashboard | 8000 |
|
||||
| `allow-risk-engine-ingress` | `risk` | broker-adapter, query-api, dashboard | 8000 |
|
||||
| `allow-trading-engine-ingress` | `trading-engine` | query-api, dashboard, kube-system (Traefik) | 8000 |
|
||||
| `allow-superset-ingress` | `superset` | kube-system (Traefik) | 8088 |
|
||||
| `allow-trino-ingress` | `trino` | superset, query-api, kube-system (Traefik) | 8080 |
|
||||
| `allow-hive-metastore-ingress` | `hive-metastore` | trino, lake-publisher | 9083 |
|
||||
| `allow-dashboard-ingress` | `dashboard` | kube-system (Traefik) | 8080 |
|
||||
| `deny-broker-adapter-ingress` | `broker-adapter` | (none — explicit deny) | — |
|
||||
|
||||
The trading-engine also has egress rules allowing outbound connections to PostgreSQL (5432), Redis (6379), HTTPS (443), SMTP (587), and DNS (53).
|
||||
|
||||
Pipeline workers (scheduler, ingestion, parser, extractor, aggregation, recommendation, lake-publisher) have no explicit ingress allow policies — they rely on the default-deny and communicate only via outbound connections to Redis queues and PostgreSQL.
|
||||
|
||||
---
|
||||
|
||||
## Value Override Files
|
||||
|
||||
The chart ships with two override files for staged deployments. ArgoCD or Kargo applies these during promotion.
|
||||
|
||||
### `values-beta.yaml` — Beta / Integration Testing
|
||||
|
||||
**Purpose**: Integration testing environment deployed to `stonks-oracle-beta` namespace. Shares infrastructure with paper but uses isolated database (`stonks_beta`), Redis DB index (`1`), and separate ingress hostnames.
|
||||
|
||||
Key overrides:
|
||||
|
||||
| Key | Beta Value | Reason |
|
||||
|-----|-----------|--------|
|
||||
| `pipelineEnabled` | `true` | Services deployed (ArgoCD health checks), but pipeline defaults to OFF via `PIPELINE_DEFAULT_OFF`. |
|
||||
| `config.DEPLOY_STAGE` | `beta` | Isolates Redis keys (`stonks:beta:*`) and MinIO buckets (`beta-stonks-*`). |
|
||||
| `config.POSTGRES_DB` | `stonks_beta` | Separate database for beta data. |
|
||||
| `config.POSTGRES_USER` | `stonks_beta` | Separate database user for beta. |
|
||||
| `config.REDIS_DB` | `1` | Separate Redis DB index. |
|
||||
| `config.LOG_LEVEL` | `DEBUG` | Verbose logging for debugging. |
|
||||
| `config.TRADING_ENABLED` | `true` | Trading engine active but constrained by paper broker mode. |
|
||||
| `config.PIPELINE_DEFAULT_OFF` | `true` | Scheduler won't enqueue jobs unless explicitly enabled via the UI. |
|
||||
| `config.BROKER_MODE` | `paper` | Simulated order execution. |
|
||||
| `config.BROKER_PROVIDER` | `alpaca` | Alpaca paper trading API. |
|
||||
| `config.OLLAMA_MODEL` | `qwen3.6` | May use a different model version for testing. |
|
||||
| `trino.enabled` | `false` | Analytics stack disabled in beta. |
|
||||
| `hiveMetastore.enabled` | `false` | Analytics stack disabled in beta. |
|
||||
| `superset.enabled` | `false` | Analytics stack disabled in beta. |
|
||||
|
||||
Beta also configures vLLM settings (`VLLM_BASE_URL`, `VLLM_MODEL`, etc.) for testing alternative LLM backends.
|
||||
|
||||
Beta ingress hostnames:
|
||||
|
||||
| Service | Hostname |
|
||||
|---------|----------|
|
||||
| Query API | `stonks-api-beta.celestium.life` |
|
||||
| Symbol Registry | `stonks-registry-beta.celestium.life` |
|
||||
| Dashboard | `stonks-beta.celestium.life` |
|
||||
| Trading Engine | `stonks-trading-beta.celestium.life` |
|
||||
| Superset | (disabled) |
|
||||
| Trino | (disabled) |
|
||||
|
||||
### `values-paper.yaml` — Paper Trading
|
||||
|
||||
**Purpose**: Paper trading environment with real market data but simulated order execution via Alpaca's paper trading API. Deployed to the main `stonks-oracle` namespace.
|
||||
|
||||
Key overrides:
|
||||
|
||||
| Key | Paper Value | Reason |
|
||||
|-----|-----------|--------|
|
||||
| `config.BROKER_MODE` | `paper` | Simulated order execution. |
|
||||
| `config.BROKER_PROVIDER` | `alpaca` | Alpaca paper trading API. |
|
||||
| `config.TRADING_ENABLED` | `true` | Trading engine active. |
|
||||
| `config.POSTGRES_DB` | `stonks_paper` | Separate database for paper trading data. |
|
||||
| `config.POSTGRES_USER` | `stonks_paper` | Separate database user. |
|
||||
| `config.REDIS_DB` | `2` | Separate Redis DB index. |
|
||||
| `config.DEPLOY_STAGE` | `paper` | Stage identifier. |
|
||||
| `config.LOG_LEVEL` | `INFO` | Standard logging. |
|
||||
| `services.extractor.replicas` | `1` | Single replica (GPU bottleneck). |
|
||||
|
||||
Paper ingress hostnames:
|
||||
|
||||
| Service | Hostname |
|
||||
|---------|----------|
|
||||
| Query API | `stonks-paper-api.celestium.life` |
|
||||
| Symbol Registry | `stonks-paper-registry.celestium.life` |
|
||||
| Dashboard | `stonks-paper.celestium.life` |
|
||||
| Superset | `stonks-paper-dash.celestium.life` |
|
||||
| Trino | `stonks-paper-trino.celestium.life` |
|
||||
| Trading Engine | `stonks-paper-trading.celestium.life` |
|
||||
|
||||
### Deployment Stage Progression
|
||||
|
||||
```
|
||||
values-beta.yaml values-paper.yaml values.yaml (base)
|
||||
Beta -> Paper Trading -> Production
|
||||
Integration Simulated orders Live trading
|
||||
testing Real market data Real orders
|
||||
Pipeline OFF Pipeline ON Pipeline ON
|
||||
Trading ON Trading ON Trading ON
|
||||
Analytics OFF Analytics ON Analytics ON
|
||||
```
|
||||
|
||||
Promotion between stages is managed by Kargo/ArgoCD. CI sets the image tag, and the promotion pipeline applies the appropriate values file.
|
||||
@@ -0,0 +1,130 @@
|
||||
# Page 1 — Data Ingestion and Preparation
|
||||
|
||||
Every signal that Stonks Oracle eventually acts on begins its life as raw data pulled from an external source. Before any AI agent can extract structured intelligence, before any trend can accumulate, and before any trade can be placed, the platform must first discover new content, fetch it reliably, eliminate duplicates, store the raw artifacts for audit, and normalize the text into a form suitable for downstream processing. This page traces that journey from external API to parser output, covering the Scheduler, Ingestion Worker, deduplication layer, raw storage, and Parser in detail.
|
||||
|
||||
For a visual overview of the full flow described here, see the [Ingestion to Extraction Flow diagram](diagrams/ingestion-to-extraction-flow.md).
|
||||
|
||||
---
|
||||
|
||||
## Four Categories of Input Data
|
||||
|
||||
Stonks Oracle tracks 50 companies across 10 sectors, and it draws intelligence from four distinct categories of external data. Each category has its own adapter, its own API conventions, and its own scheduling cadence, but all of them feed into the same ingestion pipeline.
|
||||
|
||||
The first category is **company news**, sourced from the Polygon.io ticker news endpoint (`/v2/reference/news`). The `PolygonNewsAdapter` in `services/adapters/news_adapter.py` fetches articles linked to a specific ticker, returning structured results that include title, publisher, article URL, description, keywords, and publication timestamp. Each request can return up to 1,000 articles, though the default limit is 20 per fetch. The adapter tracks the most recent `published_utc` value and uses it on subsequent fetches to avoid re-retrieving articles the system has already seen.
|
||||
|
||||
The second category is **SEC filings**, sourced from the SEC EDGAR full-text search system (EFTS). The `SECEdgarAdapter` in `services/adapters/filings_adapter.py` queries the `/LATEST/search-index` endpoint for 8-K, 10-Q, 10-K, and other form types associated with a company's ticker or CIK number. Unlike the Polygon endpoints, EDGAR is a public API that requires no key — only a descriptive `User-Agent` header per the SEC's fair-access policy. The adapter deduplicates results by accession number (`adsh`), filters out non-primary documents like XML fragments and graphics, and constructs the SEC EDGAR filing index URL for each hit so downstream services can fetch the full document.
|
||||
|
||||
The third category is **market data**, also sourced from Polygon.io. The `PolygonMarketAdapter` in `services/adapters/market_adapter.py` supports multiple endpoints: previous-day aggregate bars (`/v2/aggs/ticker/{ticker}/prev`), range bars for custom date windows, intraday hourly bars, grouped daily bars that return data for all tickers in a single call (`/v2/aggs/grouped/locale/us/market/stocks/{date}`), and ticker detail lookups. Market data follows a different path than textual content — it does not pass through the Parser or Extractor, since the structured numeric data is already in a usable form.
|
||||
|
||||
The fourth category is **macro and geopolitical news**, fetched by the `MacroNewsAdapter` in `services/adapters/macro_news_adapter.py`. Unlike the other three categories, macro news is not company-specific. These sources have `source_type='macro_news'` in the `sources` database table and may have a `NULL` `company_id`. The adapter fetches from a configurable HTTP endpoint (typically the Polygon news API filtered for broad market topics) and returns articles that describe global events — trade policy shifts, central bank decisions, geopolitical conflicts — rather than company-specific developments. Macro news articles are eventually classified by the Global Event Classifier agent and routed through a separate queue, as described in [Page 2](02-ai-agent-processing-and-extraction.md).
|
||||
|
||||
All four adapter classes inherit from `BaseAdapter` defined in `services/adapters/base.py` and return an `AdapterResult` dataclass containing the raw payload bytes, a SHA-256 content hash, a list of parsed item dicts, HTTP metadata (status code, response time), and an error field that is `None` on success. This uniform interface allows the Ingestion Worker to handle all source types through a single dispatch mechanism.
|
||||
|
||||
---
|
||||
|
||||
## The Scheduler: Orchestrating Ingestion Cycles
|
||||
|
||||
The Scheduler (`services/scheduler/app.py`) is the heartbeat of the ingestion pipeline. It runs a continuous loop that ticks every 15 seconds (`SCHEDULER_TICK = 15`), and on each tick it evaluates which sources are due for their next fetch. The Scheduler does not fetch data itself — it enqueues jobs onto the `stonks:queue:ingestion` Redis list for the Ingestion Worker to process.
|
||||
|
||||
Each source type has a default polling cadence defined in the `DEFAULT_CADENCES` dictionary:
|
||||
|
||||
| Source Type | Default Cadence |
|
||||
|---------------|-----------------|
|
||||
| `market_api` | 300 seconds |
|
||||
| `news_api` | 300 seconds |
|
||||
| `filings_api` | 3,600 seconds |
|
||||
| `macro_news` | 600 seconds |
|
||||
| `web_scrape` | 1,800 seconds |
|
||||
| `broker` | 30 seconds |
|
||||
|
||||
Individual sources can override their cadence via the `polling_interval_seconds` field in their `config` JSONB column in the `sources` table. The `get_cadence_for_source()` function checks for this override first, falling back to the default if none is set, and enforces a minimum interval of 10 seconds.
|
||||
|
||||
The Scheduler determines whether a source is due by calling `is_source_due()`, which considers several conditions. If a source has never run before (no entry in the `ingestion_runs` table), it is immediately due. If the last run failed, the Scheduler respects an exponential backoff computed by `compute_backoff()`: the delay starts at 60 seconds (`DEFAULT_BACKOFF_BASE`) and doubles with each retry up to a maximum of 3,600 seconds (`MAX_BACKOFF`). If a source has failed 10 consecutive times (`MAX_RETRY_COUNT`), the Scheduler stops scheduling it entirely until an operator manually resets the retry state. If the last run is still marked as `running`, the source is skipped to prevent double-scheduling. Otherwise, the Scheduler checks whether enough time has elapsed since the last completed run based on the source's cadence.
|
||||
|
||||
Rate limiting adds another layer of protection. The `check_rate_limit()` function enforces two constraints. First, each source type has a per-type limit defined in `DEFAULT_RATE_LIMITS` — for example, `market_api` and `news_api` are each capped at 20 requests per minute, while `filings_api` and `macro_news` are capped at 10. Second, because `market_api` and `news_api` both use the same Polygon.io API key, a global Polygon rate limit of 45 requests per minute (`POLYGON_GLOBAL_RATE_LIMIT`) is enforced across both types combined. Rate limit state is tracked in Redis using keys of the form `stonks:ratelimit:{source_type}:{window}`, where the window is a minute-granularity timestamp. If a source type exceeds its limit, the Scheduler logs a warning and skips that source for the current tick.
|
||||
|
||||
The Scheduler handles three categories of sources in each cycle. First, it fetches all active company-specific sources (excluding `macro_news`) by joining the `sources` and `companies` tables. Second, it fetches active macro news sources separately, since these may not have a `company_id`. Third, it fetches global market sources — those with `source_type='market_api'` and `company_id IS NULL` — which represent endpoints like the grouped daily bars that return data for all tickers in a single API call. For intraday bar sources, the Scheduler expands a single global source into per-ticker jobs for every active company.
|
||||
|
||||
Each enqueued job payload includes the `source_id`, `company_id`, `ticker`, `legal_name`, `source_type`, `source_name`, `config`, `credibility_score`, a list of company `aliases` (fetched from the `company_aliases` table), and a `scheduled_at` timestamp. The job is pushed onto `stonks:queue:ingestion` via Redis `RPUSH`.
|
||||
|
||||
Beyond scheduling, the Scheduler also performs periodic maintenance. Every ~20 cycles (~5 minutes), it runs `recover_stale_documents()` to re-enqueue documents that have been stuck in `parsed` status for longer than 240 minutes — a safety net for cases where Redis loses queue entries due to pod restarts or OOM events. Every ~40 cycles (~10 minutes), it runs `retry_failed_extractions()` to give documents in `extraction_failed` status another chance, resetting them to `parsed` and deleting the failed `document_intelligence` row so the Extractor treats them as fresh. Every ~100 cycles (~25 minutes), it runs `cleanup_all_tables()` to enforce retention policies across tables like `competitive_signal_records` (30 days), `ingestion_runs` (14 days), and `trading_decisions` (90 days).
|
||||
|
||||
For more detail on the Scheduler's configuration and operational behavior, see the [Services Reference](../services.md).
|
||||
|
||||
---
|
||||
|
||||
## The Ingestion Worker: Adapter Dispatch and Persistence
|
||||
|
||||
The Ingestion Worker (`services/ingestion/worker.py`) is a long-running process that continuously pops jobs from the `stonks:queue:ingestion` Redis list and processes them. On startup, it initializes one instance of each adapter class and stores them in a dispatch dictionary keyed by `source_type`:
|
||||
|
||||
```
|
||||
adapters = {
|
||||
"market_api": PolygonMarketAdapter(...),
|
||||
"news_api": PolygonNewsAdapter(...),
|
||||
"filings_api": SECEdgarAdapter(),
|
||||
"web_scrape": WebScrapeAdapter(),
|
||||
"broker": AlpacaBrokerAdapter(...),
|
||||
"macro_news": MacroNewsAdapter(...),
|
||||
}
|
||||
```
|
||||
|
||||
When a job arrives, the `process_job()` function looks up the appropriate adapter by `source_type` and calls its `fetch()` method with the ticker and source config. Before fetching, it records a new row in the `ingestion_runs` table with status `running`. If the adapter returns an error, the worker calls `record_retrieval_failure()` to update the run status and increment the source's retry counter with exponential backoff timing.
|
||||
|
||||
On a successful fetch, the worker performs several steps in sequence. First, it uploads the raw payload to MinIO via `upload_raw_artifact()` in `services/shared/storage.py`. The target bucket is determined by the source type through the `SOURCE_BUCKET_MAP`: `market_api` payloads go to `stonks-raw-market`, `news_api` and `macro_news` payloads go to `stonks-raw-news`, and `filings_api` payloads go to `stonks-raw-filings`. Objects are stored under a path that encodes the source type, ticker, date hierarchy, and document ID — for example, `news_api/AAPL/2025/01/15/{run_id}/raw.json`.
|
||||
|
||||
---
|
||||
|
||||
## Content Deduplication via Redis
|
||||
|
||||
After storing the raw artifact, the Ingestion Worker checks for duplicate content. Deduplication operates at two levels.
|
||||
|
||||
At the payload level, the worker checks the overall `content_hash` (a SHA-256 digest of the raw API response) against Redis. The key pattern is `stonks:dedupe:{content_hash}` with a 24-hour TTL (86,400 seconds). If the hash is already present, the entire payload is skipped — the `ingestion_runs` row is marked as completed with `items_new=0`, and no downstream jobs are enqueued. If the hash is new, the worker sets the marker in Redis so future fetches of identical content are caught.
|
||||
|
||||
At the individual item level, for source types other than `market_api` and `broker`, the worker calls `dedupe_items()` from `services/shared/dedupe.py`. This function checks each item against a layered deduplication strategy. The fast path checks Redis for both content-hash markers (`stonks:dedupe:{hash}`) and canonical-URL markers (`stonks:dedupe:url:{url_hash}`), both with 24-hour TTLs. If the Redis check misses, the function falls back to PostgreSQL, querying the `documents` table by `content_hash` or `canonical_url` for durable cross-source matching. When a duplicate is found through the PostgreSQL fallback, the function warms the Redis cache so subsequent checks are fast.
|
||||
|
||||
Items identified as duplicates are not discarded entirely. If the duplicate document was originally ingested for a different company, the worker creates a cross-source mention link in the `document_company_mentions` table via `persist_document_company_mention()`. This ensures that a news article mentioning both Apple and Microsoft is linked to both companies even if it was first ingested through Apple's news source.
|
||||
|
||||
New (non-duplicate) items are persisted to PostgreSQL through `persist_ingestion_items()` in `services/shared/metadata.py`, which inserts rows into the `documents` table and records company mentions in `document_company_mentions`. Each new document ID is then pushed onto `stonks:queue:parsing` for the Parser to process. After persistence, the worker calls `mark_as_seen()` to set Redis dedupe markers for both the content hash and canonical URL of each new item, ensuring that the next fetch cycle's deduplication checks are fast.
|
||||
|
||||
On successful completion, the worker updates the `ingestion_runs` row with the final counts (`items_fetched`, `items_new`) and calls `reset_source_retry_state()` to clear any accumulated backoff from previous failures. For news-type sources (`news_api` and `macro_news`), the worker also updates the source's `config` JSONB column with the latest `published_utc` value, so the next fetch only retrieves newer articles.
|
||||
|
||||
---
|
||||
|
||||
## The Parser: Normalization, Quality Scoring, and Routing
|
||||
|
||||
Documents that pass through ingestion arrive on the `stonks:queue:parsing` Redis list as JSON payloads containing a `document_id`, `ticker`, and `source_type`. The Parser Worker (`services/parser/worker.py`) pops these jobs and transforms raw HTML or text into normalized, quality-scored documents ready for AI extraction.
|
||||
|
||||
The parsing pipeline begins with HTML fetching. If the document has a URL (looked up from the `documents` table if not present in the job payload), the worker calls `fetch_html()` to retrieve the page content. SEC EDGAR URLs receive a specialized `User-Agent` header to comply with the SEC's fair-access policy. The raw HTML is then passed to `parse_html()` in `services/parser/html_parser.py`, which runs a multi-stage extraction pipeline.
|
||||
|
||||
The HTML parser first strips non-content tags — `script`, `style`, `nav`, `footer`, `header`, `aside`, `iframe`, and others — and removes boilerplate containers identified by CSS class or ID patterns (sidebars, ad slots, newsletter signups, social share bars, and similar UI elements). It then searches for the article body using a priority list of semantic selectors (`article`, `[role='main']`, `.article-body`, `.post-content`, and others). If no semantic match is found, it falls back to text-density scoring across candidate `div`, `section`, and `td` elements, selecting the block with the highest composite score based on text density, link density, paragraph count, and word count. The extracted text undergoes further cleaning: regex-based removal of residual boilerplate phrases (copyright notices, "subscribe to our newsletter" prompts, "share this article" fragments), removal of short orphan lines that are likely UI fragments, detection and collapse of repeated template blocks, and whitespace normalization.
|
||||
|
||||
Metadata extraction pulls the document title (from `og:title` or `<title>`), author, publisher (from `og:site_name` or hostname), publication date (from `article:published_time` or JSON-LD `datePublished`), canonical URL, language, description, and keywords from the HTML head elements.
|
||||
|
||||
If the parsed body text is shorter than 500 characters, the worker attempts to enrich it by reading the raw API payload from MinIO and extracting the Polygon article description, keywords, and author fields for the matching article. This enrichment step ensures that even articles with minimal scrapeable HTML still have enough textual content for meaningful AI extraction.
|
||||
|
||||
Quality scoring is performed by `score_parse_quality()` in `services/parser/html_parser.py`, which evaluates six weighted signals to produce a composite score between 0 and 0.95:
|
||||
|
||||
| Signal | Weight | What It Measures |
|
||||
|--------------------|--------|-----------------------------------------------------------------|
|
||||
| `word_count` | 0.30 | Length of extracted text (thresholds at 20, 50, 150, 300 words) |
|
||||
| `body_found` | 0.20 | Whether a semantic article body element was located |
|
||||
| `diversity` | 0.15 | Vocabulary richness (unique words / total words) |
|
||||
| `sentence` | 0.15 | Presence of proper sentence structure (terminal punctuation) |
|
||||
| `paragraph` | 0.10 | Multi-paragraph structure (blocks separated by blank lines) |
|
||||
| `metadata` | 0.10 | Presence of title, author, publisher, and publication date |
|
||||
|
||||
The composite score maps to a confidence label: scores below 0.35 are labeled `low`, scores between 0.35 and 0.65 are `medium`, and scores 0.65 and above are `high`. Documents with `low` confidence are marked with status `low_quality` in the `documents` table and are not enqueued for extraction — they are effectively filtered out of the pipeline at this stage.
|
||||
|
||||
Company mention detection runs next. The worker fetches all known aliases from the `company_aliases` table (plus tickers and legal names from the `companies` table) and calls `detect_company_mentions()` in `services/parser/html_parser.py`. The matching strategy varies by alias length: one-to-two character aliases use case-sensitive word-boundary matching to avoid false positives (the letter "A" should not match every occurrence of the word "a"), three-to-four character aliases use case-insensitive word-boundary matching (standard ticker format), and aliases of five or more characters use case-insensitive substring matching (company names and brands). Confidence scores vary by alias type: ticker matches receive 0.9, legal name matches 0.85, general aliases 0.7, and brand matches 0.6. Multiple alias hits for the same company are deduplicated, keeping the highest-confidence match and summing match counts. Detected mentions are persisted to the `document_company_mentions` table.
|
||||
|
||||
The normalized text and a structured parser output JSON (containing all metadata, quality signals, warnings, outbound links, tags, and mentions) are uploaded to the `stonks-normalized` MinIO bucket. The `documents` row is updated with the normalized storage reference, parser output reference, quality score, and confidence level.
|
||||
|
||||
Finally, the Parser makes a routing decision. If the document's `document_type` is `macro_event`, it is pushed onto `stonks:queue:macro_classification` for the Global Event Classifier agent. All other documents are pushed onto `stonks:queue:extraction` for the Document Intelligence Extractor agent. Both queues feed into the Extractor service described in [Page 2](02-ai-agent-processing-and-extraction.md). The job payload includes the `document_id`, `ticker`, and the first 32,000 characters of the normalized text, giving the downstream agent immediate access to the content without needing to fetch it from MinIO.
|
||||
|
||||
For additional detail on queue topology and data store layout, see the [Data Pipeline Architecture](../architecture-data-pipeline.md) documentation.
|
||||
|
||||
---
|
||||
|
||||
## What Comes Next
|
||||
|
||||
At this point, raw data has been fetched from four external sources, deduplicated, stored in MinIO, parsed into normalized text, scored for quality, tagged with company mentions, and routed to the appropriate extraction queue. The documents sitting on `stonks:queue:extraction` and `stonks:queue:macro_classification` are clean, quality-filtered, and ready for AI processing. [Page 2 — AI Agent Processing and Structured Extraction](02-ai-agent-processing-and-extraction.md) picks up the story from here, explaining how the Document Intelligence Extractor and Global Event Classifier agents use LLM inference to transform these normalized documents into the structured JSON intelligence that feeds the rest of the pipeline.
|
||||
@@ -0,0 +1,164 @@
|
||||
# Page 2 — AI Agent Processing and Structured Extraction
|
||||
|
||||
Documents that arrive on the `stonks:queue:extraction` and `stonks:queue:macro_classification` Redis queues are clean, quality-filtered, and normalized — but they are still unstructured text. The job of the Extractor service is to transform that text into structured JSON intelligence that the rest of the pipeline can reason about quantitatively. Two AI agents share this responsibility: the Document Intelligence Extractor handles company-specific news, filings, and transcripts, while the Global Event Classifier handles macro-level geopolitical and economic events. Both agents run through the same Ollama-based inference infrastructure, share a common JSON repair pipeline, and persist their results to PostgreSQL and MinIO for downstream consumption and audit.
|
||||
|
||||
This page explains how each agent works, what schemas they produce, how the system validates and repairs LLM output, how runtime configuration is resolved from the database, and how the final structured records are persisted. For a visual overview of the full flow from ingestion through extraction, see the [Ingestion to Extraction Flow diagram](diagrams/ingestion-to-extraction-flow.md). For reference-level detail on agent configuration and the variant management API, see the [AI Agents Guide](../ai-agents.md).
|
||||
|
||||
---
|
||||
|
||||
## The Document Intelligence Extractor
|
||||
|
||||
The Document Intelligence Extractor is the primary AI agent in the pipeline. Registered under the slug `document-extractor` in the `ai_agents` database table, it processes every non-macro document that passes through the Parser — news articles, SEC filings, earnings transcripts, and press releases. Its purpose is to read a normalized document and produce a structured JSON object that captures the document's summary, the companies it affects, the sentiment and impact for each company, the catalysts driving that impact, and the evidence supporting the analysis.
|
||||
|
||||
The entry point is `services/extractor/main.py`, which runs a continuous worker loop polling the `stonks:queue:extraction` Redis list. When a job arrives, the worker extracts the `document_id`, `ticker`, and `text` fields from the JSON payload. If the job payload does not include the document text directly, the worker fetches it from MinIO using the `normalized_storage_ref` stored in the `documents` table — the Parser uploaded the normalized text to the `stonks-normalized` bucket during the previous pipeline stage (see [Page 1](01-data-ingestion-and-preparation.md)).
|
||||
|
||||
The actual LLM inference is handled by `OllamaClient` in `services/extractor/client.py`. The client sends the document to a local Ollama instance via the `/api/chat` HTTP endpoint with `stream=False` and `think=False`. The `think=False` flag is a deliberate performance choice — it disables the model's chain-of-thought reasoning phase, which would otherwise add two to four minutes of latency per document. The client does not use Ollama's `format` parameter for structured output because of a known Ollama bug (#14645) where the format constraint is silently ignored when `think=False` on qwen3.5 models. Instead, the system relies on prompt engineering to produce JSON and repairs any syntax issues after the fact.
|
||||
|
||||
The prompt sent to the model has two parts. The system prompt, defined in `services/extractor/prompts.py`, establishes the model's role as a financial document analyst and sets strict output rules: return only a single JSON object, no markdown fences, no explanation text, every schema field is required, use `"other"` for `catalyst_type` when unsure, keep evidence spans under 20 words, and limit key facts to three to five items. The user prompt, built by `build_extraction_prompt()` in the same module, provides the document text along with document-type-specific guidance. Four guidance variants exist — one each for articles, filings, transcripts, and press releases — each calibrated to the conventions and biases of that document type. For example, the filing guidance instructs the model to preserve the precise legal language of SEC documents, while the press release guidance warns that sentiment may be biased positive and directs the model to focus on concrete metrics rather than marketing language.
|
||||
|
||||
The user prompt also includes a list of all tracked tickers from the `companies` table, along with rules for how the model should use them. If a tracked ticker appears verbatim in the text, the model must include it in the output with at least one evidence span. If the article discusses a sector or theme that clearly affects a tracked company (oil prices affecting XOM, AI chip demand affecting NVDA), the model should include that company as well. The model is explicitly told not to invent tickers that are not in the provided list. Documents longer than 8,000 characters are truncated before being included in the prompt, with a `[... truncated for extraction ...]` marker appended.
|
||||
|
||||
The `OllamaClient` also supports a `context_window` override via the Ollama `num_ctx` option, which can be configured per agent variant through the `AgentConfigResolver` mechanism described later in this page.
|
||||
|
||||
---
|
||||
|
||||
## The ExtractionResult Schema
|
||||
|
||||
The structured output that the Document Intelligence Extractor produces is defined by the `ExtractionResult` Pydantic model in `services/extractor/schemas.py`. Every field is required — the model has no defaults — so the generated JSON schema forces the LLM to produce every field explicitly. The top-level fields are:
|
||||
|
||||
**`summary`** — a concise one-to-three sentence summary of the document's main point. This becomes the human-readable description stored in the `document_intelligence` table.
|
||||
|
||||
**`companies`** — an array of `CompanyExtractionItem` objects, one per affected company. Each company entry contains:
|
||||
|
||||
- `ticker` — the stock ticker symbol (validated against a regex pattern of one to five uppercase letters).
|
||||
- `company_name` — the full company name as referenced in the document.
|
||||
- `relevance` — a float between 0.0 and 1.0 indicating how relevant the document is to this company, where 0 means tangential and 1 means the company is the primary subject.
|
||||
- `sentiment` — one of `positive`, `negative`, `neutral`, or `mixed`, representing the overall sentiment toward this company in the document.
|
||||
- `impact_score` — a float between 0.0 and 1.0 estimating the magnitude of impact, where 0 is negligible and 1 is highly material.
|
||||
- `impact_horizon` — one of `intraday`, `1d`, `1d_7d`, `1d_30d`, `30d_90d`, or `90d_plus`, indicating the expected timeframe over which the impact will play out.
|
||||
- `catalyst_type` — exactly one of `earnings`, `product`, `legal`, `macro`, `supply_chain`, `m_and_a`, `rating_change`, or `other`. The prompt instructs the model to use `other` when none of the specific categories fit.
|
||||
- `key_facts` — a list of facts explicitly stated in the document. The prompt emphasizes that the model must not infer or fabricate facts.
|
||||
- `risks` — a list of risks explicitly mentioned in the document.
|
||||
- `evidence_spans` — short verbatim quotes from the document supporting the analysis. The prompt requests these be kept under 20 words each.
|
||||
|
||||
**`macro_themes`** — a list of broad economic or market themes mentioned in the document, such as `rates`, `inflation`, or `ai_capex`.
|
||||
|
||||
**`novelty_score`** — a float between 0.0 and 1.0 indicating how novel or surprising the information is. Routine earnings reports score low; unexpected regulatory actions score high. This value feeds into the novelty bonus component of the signal weighting formula described in [Page 3](03-signal-scoring-and-weighted-signals.md).
|
||||
|
||||
**`confidence`** — a float between 0.0 and 1.0 representing the model's confidence in the accuracy of its extraction. Lower values indicate ambiguous or incomplete source text. This value becomes the confidence gate input for signal scoring.
|
||||
|
||||
**`extraction_warnings`** — a list of issues encountered during extraction, such as `ambiguous_ticker`, `incomplete_text`, or `low_confidence`. These warnings are persisted alongside the intelligence record for operational monitoring.
|
||||
|
||||
The JSON schema is generated programmatically from the Pydantic models via `generate_json_schema()` in `services/extractor/schemas.py`, which calls Pydantic's `model_json_schema()` and then inlines all `$defs` references so the schema is self-contained and Ollama-friendly.
|
||||
|
||||
---
|
||||
|
||||
## The Global Event Classifier
|
||||
|
||||
Not all documents describe company-specific developments. Macro news articles — those tagged with `document_type='macro_event'` by the Parser — describe events that affect entire markets, sectors, or economies: trade wars, central bank rate decisions, commodity supply disruptions, geopolitical conflicts. These documents are routed to the `stonks:queue:macro_classification` Redis queue and processed by the Global Event Classifier agent, registered under the slug `event-classifier` in the `ai_agents` table.
|
||||
|
||||
The classifier is implemented in `services/extractor/event_classifier.py`. When the extractor worker in `services/extractor/main.py` pops a job and determines that the document type is `macro_event` (either because the job came from the macro queue or because the `documents` table records it as such), it routes the document to `_process_macro_classification()` instead of the standard extraction pipeline. This function calls `classify_global_event()`, which builds a dedicated prompt, sends it to Ollama through the same `OllamaClient` infrastructure, parses the response, and persists the result.
|
||||
|
||||
The classifier's system prompt is distinct from the extractor's. It establishes the model's role as a macro-level news classifier and includes explicit anti-hallucination rules that are critical to preventing the classifier from overreaching. The prompt states that the model should only classify articles about macro events that affect entire markets, sectors, or economies — trade wars, interest rate changes, commodity supply disruptions, regulatory changes, geopolitical conflicts, natural disasters. It explicitly lists what should not be classified as macro events: individual company earnings, lawsuits against a single company, single-company management changes, individual stock analysis, company-specific debt or bankruptcy, and product launches by one company. For these company-specific articles that were incorrectly routed, the model is instructed to set severity to `"low"`, confidence below 0.3, and leave the `affected_regions`, `affected_sectors`, and `affected_commodities` arrays empty.
|
||||
|
||||
The user prompt, built by `build_event_classification_prompt()`, reinforces these anti-hallucination rules and provides additional guidance. It instructs the model to only extract facts explicitly stated in the text, to set confidence below 0.4 for vague or speculative content, to distinguish announced policy from rumored policy, and to reserve `"critical"` severity for events affecting multiple countries or entire global markets. Articles longer than 6,000 characters are truncated before inclusion in the prompt.
|
||||
|
||||
The output schema is the `GlobalEvent` dataclass, which contains:
|
||||
|
||||
- `event_types` — a list of impact type strings, drawn from a fixed set: `supply_disruption`, `demand_shift`, `cost_increase`, `regulatory_pressure`, `currency_impact`, `commodity_shock`, `trade_barrier`, and `geopolitical_risk`. The model is instructed to include all applicable types rather than collapsing to a single category.
|
||||
- `severity` — one of `low`, `moderate`, `high`, or `critical`.
|
||||
- `affected_regions` — ISO 3166-1 alpha-2 country codes or region names (e.g., `US`, `CN`, `EU`, `GB`, `JP`). Only regions explicitly mentioned or clearly implied should be included.
|
||||
- `affected_sectors` — GICS sector identifiers such as `Energy`, `Financials`, `Information Technology`, or `Industrials`.
|
||||
- `affected_commodities` — commodity identifiers like `crude_oil`, `natural_gas`, `gold`, `copper`, `wheat`, `lithium`, or `semiconductors`. An empty list if no commodities are directly affected.
|
||||
- `summary` — a one-to-three sentence summary of the event and its market implications.
|
||||
- `key_facts` — facts explicitly stated in the article, limited to three to five items.
|
||||
- `estimated_duration` — one of `short_term` (days to weeks), `medium_term` (weeks to months), or `long_term` (months to years).
|
||||
- `confidence` — a float between 0.0 and 1.0, clamped during parsing.
|
||||
|
||||
Each `GlobalEvent` also carries a `model_metadata` object recording the provider (`ollama`), model name, prompt version (`event-classification-v1`), and schema version (`1.0.0`), plus a `source_document_id` linking back to the originating document.
|
||||
|
||||
After a successful classification, the system computes macro impact records for all tracked companies using the exposure-based interpolation engine in `services/aggregation/interpolation.py`. Each company's exposure profile — geographic revenue mix, supply chain regions, key input commodities, regulatory jurisdictions, and market position tier — determines how much a given macro event affects that company. Companies with non-zero macro impact scores get `macro_impact_records` rows persisted to PostgreSQL, and aggregation jobs are enqueued to `stonks:queue:aggregation` for each affected ticker. The extractor worker tracks consecutive macro classification failures and emits a critical-level alert after three consecutive failures, continuing with company-only signals in the meantime.
|
||||
|
||||
---
|
||||
|
||||
## The JSON Repair Pipeline
|
||||
|
||||
LLM output is inherently unreliable at the syntactic level. Models sometimes wrap JSON in markdown fences, produce trailing commas, leave strings unterminated, or truncate output mid-object when they hit token limits. The extractor addresses this with a three-stage JSON repair pipeline implemented across `services/extractor/client.py` and `services/extractor/schemas.py`.
|
||||
|
||||
The first stage is a direct `json.loads()` call. If the raw model output is already valid JSON, no repair is needed and the pipeline moves straight to validation. This is the fast path for well-behaved model responses.
|
||||
|
||||
The second stage strips markdown fences. Models frequently wrap their output in `` ```json ... ``` `` blocks despite being told not to. The `_strip_markdown_fences()` function in `services/extractor/client.py` uses a regex to detect and remove these wrappers before attempting another parse.
|
||||
|
||||
The third stage invokes the `json-repair` library as a fallback. The `_repair_json()` function in `services/extractor/client.py` calls `repair_json()` with `return_objects=False` to get a repaired JSON string. This library handles a wide range of common LLM JSON errors — trailing commas, missing quotes, unescaped characters — that would otherwise require custom repair logic.
|
||||
|
||||
The `services/extractor/schemas.py` module contains an additional layer of repair logic in its own `_repair_json()` function, which handles cases that the library might miss. It strips non-JSON prefixes (models sometimes prepend explanatory text before the opening brace), removes control characters that break parsing, fixes trailing commas before closing brackets, and as a last resort calls `_repair_truncated_json()` — a state-machine parser that walks the string tracking bracket depth and string state, then appends the necessary closing tokens to complete a truncated JSON object.
|
||||
|
||||
For the Global Event Classifier, the `_parse_classification_response()` function in `services/extractor/event_classifier.py` reuses the same `_strip_markdown_fences()` and `_repair_json()` functions from the client module, and additionally handles the case where the model wraps the output object in a single-element list — a quirk observed with some model configurations.
|
||||
|
||||
---
|
||||
|
||||
## Structural and Semantic Validation
|
||||
|
||||
Repairing JSON syntax is only the first step. The `validate_extraction()` function in `services/extractor/schemas.py` performs both structural and semantic validation on the parsed output, and the distinction between the two is important for understanding the retry logic.
|
||||
|
||||
Structural validation begins with normalization. The `_normalize_extraction_data()` function fills in missing top-level fields with sensible defaults (empty summary, empty companies array, 0.5 novelty score, 0.3 confidence), clamps numeric fields to the [0.0, 1.0] range, and normalizes per-company fields. Catalyst types that the model produces as free-text alternatives — `"strategic pivot"`, `"acquisition"`, `"lawsuit"`, `"inflation"`, `"launch"` — are mapped to their canonical enum values through a comprehensive alias dictionary. Impact horizons like `"long-term"`, `"short"`, `"immediate"`, or `"near-term"` are similarly mapped to the valid set (`intraday`, `1d`, `1d_7d`, `1d_30d`, `30d_90d`, `90d_plus`). After normalization, the data is validated against the `ExtractionResult` Pydantic model, which enforces type constraints, enum membership, and range bounds.
|
||||
|
||||
Semantic validation catches issues that are structurally valid but logically suspect. The `_semantic_checks()` function runs a series of cross-field consistency checks that produce either errors (which trigger a retry) or warnings (which are logged but do not block acceptance). Semantic errors include duplicate tickers across company entries, missing ticker fields, and invalid impact horizon values. Semantic warnings include empty summaries, low confidence with companies present, invalid ticker formats (not matching the one-to-five uppercase letter pattern), missing evidence spans, evidence spans that are too short (under 8 characters) or too long (over 500 characters), high impact scores with no supporting key facts, very low relevance scores, and strong sentiment paired with negligible impact scores.
|
||||
|
||||
When the original document text is available, the validator also performs an evidence grounding check: each evidence span is searched for in the source text (case-insensitive), and spans not found in the document are flagged with a warning. This helps detect hallucinated evidence — quotes the model fabricated rather than extracted from the actual text.
|
||||
|
||||
If validation produces any semantic errors, the `ValidationReport` is marked as invalid and the `OllamaClient` retry loop treats it as a failed attempt. The retry logic uses exponential backoff with configurable parameters: a base delay (default from `OllamaConfig`), a multiplier applied on each retry, and a maximum delay cap. The number of retries is configurable per agent through the `max_retries` field in the `ai_agents` or `agent_variants` table. Non-retryable errors — HTTP 400, 401, 403, 404, and 422 responses from Ollama — short-circuit the retry loop immediately, since these indicate a problem with the request itself rather than a transient model failure.
|
||||
|
||||
Every attempt, whether successful or not, is recorded in an `ExtractionAttempt` dataclass that captures the raw output, validation report, error description, duration in milliseconds, model name, and whether the error was retryable. The full list of attempts is preserved in the `ExtractionResponse` for audit purposes and uploaded to MinIO by the persistence layer.
|
||||
|
||||
---
|
||||
|
||||
## The AgentConfigResolver: Hot-Swapping Models and Prompts
|
||||
|
||||
Both the Document Intelligence Extractor and the Global Event Classifier resolve their runtime configuration through the `AgentConfigResolver` in `services/shared/agent_config.py`. This mechanism allows operators to change models, prompts, timeouts, retry counts, and token budgets without restarting any service — changes take effect within 60 seconds.
|
||||
|
||||
The resolver works by querying the `ai_agents` and `agent_variants` PostgreSQL tables with a single SQL statement that uses `COALESCE` to prefer variant values over base agent values. When the extractor worker starts, it creates an `AgentConfigResolver` instance with a 60-second TTL cache and calls `resolver.resolve("document-extractor")` to get the active configuration. If an active variant exists for the agent (enforced by a unique partial index on `agent_variants` that allows at most one active variant per agent), the variant's `model_name`, `system_prompt`, `temperature`, `max_tokens`, `context_window`, `timeout_seconds`, and `max_retries` override the base agent's values wherever the variant provides a non-NULL value. If no active variant exists, the base agent's configuration is used. If the database query fails entirely, the resolver returns `None` and the worker falls back to environment-variable-based `OllamaConfig` defaults.
|
||||
|
||||
The resolved configuration is captured in a `ResolvedAgentConfig` frozen dataclass that includes the `agent_id`, `variant_id` (if any), `model_provider`, `model_name`, `system_prompt`, `user_prompt_template`, `prompt_version`, `temperature`, `max_tokens`, `context_window`, `input_token_limit`, `token_budget`, `timeout_seconds`, and `max_retries`. The extractor worker uses this to build an `OllamaConfig` that is passed to the `OllamaClient`.
|
||||
|
||||
The 60-second TTL cache means the resolver only hits the database once per minute per agent slug. Cache entries are keyed by slug and timestamped with `time.monotonic()`. When a cached entry expires, the next `resolve()` call re-queries the database and refreshes the cache. The `invalidate()` method can clear a single slug or the entire cache, though in practice the TTL-based expiry is sufficient for normal operations.
|
||||
|
||||
The extractor worker re-resolves its configuration every 100 jobs. If the resolved model name has changed (for example, because an operator activated a variant that uses a different model), the worker closes the old `OllamaClient` and creates a new one with the updated configuration. The event classifier is resolved separately and can use a different model than the document extractor — the worker maintains two independent `OllamaClient` instances when the models differ.
|
||||
|
||||
Token budget enforcement adds another layer of control. If a variant specifies a `token_budget` (total tokens per hour), the worker checks the `agent_performance_log` table before each invocation to see whether the budget has been exceeded. If so, the invocation is skipped entirely. Input token limits work similarly: if a variant sets an `input_token_limit`, the worker truncates the document text to approximately that many tokens (estimated at four characters per token) before sending it to the model.
|
||||
|
||||
For a complete guide to creating variants, activating them, and comparing their performance, see the [AI Agents Guide](../ai-agents.md).
|
||||
|
||||
---
|
||||
|
||||
## Persistence: From Extraction to Database
|
||||
|
||||
Once the LLM produces a valid extraction and it passes validation, the `persist_extraction()` function in `services/extractor/worker.py` orchestrates the full persistence pipeline. This function writes to both MinIO (for audit) and PostgreSQL (for downstream consumption), ensuring that every extraction attempt is fully traceable.
|
||||
|
||||
The MinIO persistence layer uploads four artifacts per extraction, all stored under date-partitioned paths in dedicated buckets. The prompt metadata (prompt version, schema version, model name) goes to `stonks-llm-prompts`. The raw model output for every attempt — including failed ones — goes to `stonks-llm-results`, preserving the full retry history. A validation report summarizing the final attempt's status, errors, and warnings is uploaded alongside the raw output. On success, the final parsed intelligence object (the `ExtractionResult` serialized as JSON) is uploaded to a separate path for easy retrieval.
|
||||
|
||||
The PostgreSQL persistence writes to two tables. The `document_intelligence` table receives one row per document, containing the summary, macro themes, novelty score, source credibility, extraction warnings, confidence, model metadata (provider, model name, prompt version, schema version), references to the MinIO artifacts (raw output ref, prompt ref), validation status (`valid` or `failed`), validation errors, and retry count. This row is the authoritative record of what the AI extracted from the document.
|
||||
|
||||
The `document_impact_records` table receives one row per company mention within the extraction. Each impact record is linked to the parent `document_intelligence` row via `intelligence_id` and to the `companies` table via `company_id`. The record captures the ticker, relevance, sentiment, impact score, impact horizon, catalyst type, key facts, risks, and evidence spans for that specific company. The `company_id` is resolved from a ticker-to-UUID mapping that the worker maintains by querying the `companies` table (refreshed every 100 jobs). If a ticker in the extraction output does not match any tracked company, the impact record is skipped with a warning — the system only persists impact records for companies in its tracked universe.
|
||||
|
||||
After persisting the intelligence and impact records, the worker updates the document's status in the `documents` table to `extracted` (or `extraction_failed` if all retry attempts were exhausted). Even failed extractions get a `document_intelligence` row with `validation_status='failed'`, empty summary, zero confidence, and the accumulated error messages — this ensures the failure is visible in the database rather than silently lost.
|
||||
|
||||
Performance metrics are collected for every extraction via `collect_metrics()` in `services/extractor/metrics.py` and persisted to a metrics table. Prometheus counters and histograms track extraction attempts, duration, retries, confidence distribution, validation errors, and estimated token usage (input and output, estimated at four characters per token). When a resolved agent config is available, the worker also logs to the `agent_performance_log` table with variant attribution, enabling the A/B comparison queries described in the [AI Agents Guide](../ai-agents.md).
|
||||
|
||||
For the Global Event Classifier, persistence follows a parallel path. The prompt and raw output are uploaded to MinIO under an `event_classification/macro/` path prefix. The parsed `GlobalEvent` is persisted to the `global_events` PostgreSQL table, which stores the event types, severity, affected regions, affected sectors, affected commodities, summary, key facts, estimated duration, confidence, source document ID, and model metadata. Downstream, the macro interpolation engine computes `macro_impact_records` for each affected company and persists those as well.
|
||||
|
||||
---
|
||||
|
||||
## Enqueuing Aggregation Jobs
|
||||
|
||||
The final step in the extraction pipeline is to notify the downstream aggregation engine that new intelligence is available. After a successful document extraction, the worker pushes a job onto the `stonks:queue:aggregation` Redis list containing the ticker of the affected company. The aggregation engine (described in [Page 3](03-signal-scoring-and-weighted-signals.md)) will pick up this job and recompute the weighted signals and trend summaries for that ticker, incorporating the freshly extracted intelligence.
|
||||
|
||||
For macro events, the enqueue logic is more expansive. After the Global Event Classifier produces a `GlobalEvent` and the interpolation engine computes macro impact records, the worker enqueues an aggregation job for every ticker that received a non-zero macro impact score. A single macro event — say, a new tariff announcement affecting the Energy and Industrials sectors — can trigger aggregation recomputation for dozens of tickers simultaneously. The aggregation job payload includes both the `ticker` and the `macro_event_id`, so the aggregation engine knows to incorporate the new macro signals.
|
||||
|
||||
The worker alternates between the extraction and macro classification queues to prevent starvation: every third job is pulled from `stonks:queue:macro_classification`, with the remaining two-thirds from `stonks:queue:extraction`. If the preferred queue is empty, the worker falls back to the other queue, ensuring that neither pipeline stalls while the other has work available.
|
||||
|
||||
---
|
||||
|
||||
## What Comes Next
|
||||
|
||||
At this point, documents have been transformed from unstructured text into structured JSON intelligence — `ExtractionResult` objects for company-specific documents and `GlobalEvent` objects for macro news. These structured records are persisted in PostgreSQL and their tickers have been enqueued for aggregation. But raw extraction output is not yet actionable for trading decisions. The extraction tells us that a document is bearish for AAPL with an impact score of 0.7 and a confidence of 0.8, but it does not tell us how much weight that signal should carry relative to other signals about AAPL, or how it compares to signals from different sources, time periods, or market conditions. [Page 3 — Signal Scoring and the WeightedSignal Abstraction](03-signal-scoring-and-weighted-signals.md) picks up the story from here, explaining how the aggregation engine transforms these raw extraction outputs into weighted signals through confidence gating, recency decay, source credibility scoring, novelty bonuses, and market context multipliers.
|
||||
@@ -0,0 +1,210 @@
|
||||
# Page 3 — Signal Scoring and the WeightedSignal Abstraction
|
||||
|
||||
The extraction pipeline described in [Page 2](02-ai-agent-processing-and-extraction.md) produces structured intelligence records — `document_impact_records` for company-specific documents, `macro_impact_records` for global events, and `competitive_signal_records` for cross-company pattern propagation. Each record carries a sentiment, an impact score, a confidence value, and a publication timestamp. But these raw values are not directly comparable. A high-confidence extraction from a reputable source published ten minutes ago should carry far more weight than a low-confidence extraction from an unknown source published three weeks ago. A document that breaks genuinely novel information should matter more than one that rehashes yesterday's earnings call. And when the market is moving fast — high volatility, surging volume — fresh signals become even more critical.
|
||||
|
||||
The signal scoring layer in `services/aggregation/scoring.py` solves this problem by transforming each raw intelligence record into a `WeightedSignal` object: a document reference paired with a composite aggregation weight that encodes recency, credibility, novelty, confidence, and market conditions into a single number. This page explains how that weight is computed, how sentiment labels become numeric values, and how three independent signal layers — Company, Macro, and Competitive — each produce `WeightedSignal` objects that are concatenated into a unified list before the aggregation engine computes trend summaries. For a visual breakdown of the composite weight formula, see the [Weighted Signal Computation diagram](diagrams/weighted-signal-computation.md). For the full picture of how the three layers merge, see the [Three-Layer Signal Merging diagram](diagrams/three-layer-signal-merging.md).
|
||||
|
||||
---
|
||||
|
||||
## The WeightedSignal and SignalWeight Dataclasses
|
||||
|
||||
The core abstraction is the `WeightedSignal` dataclass, defined in `services/aggregation/scoring.py`. It pairs a document reference with the computed weight and the signal's sentiment and impact values:
|
||||
|
||||
- **`document_id`** — the UUID of the source document (for company and macro signals) or a synthetic identifier for pattern-derived signals (e.g., `pattern:AAPL:earnings:7d`).
|
||||
- **`weight`** — a `SignalWeight` object containing the component breakdown and the final combined score.
|
||||
- **`sentiment_value`** — a numeric sentiment value: `+1.0` for positive, `-1.0` for negative, `0.0` for neutral or mixed.
|
||||
- **`impact_score`** — the magnitude of impact, drawn from the extraction's per-company impact score for company signals, or scaled by a layer-specific weight multiplier for macro and competitive signals.
|
||||
|
||||
The `SignalWeight` dataclass captures the individual components that feed into the combined weight, making the scoring decision fully transparent and auditable:
|
||||
|
||||
- **`recency`** — the exponential decay weight based on document age.
|
||||
- **`credibility`** — the source credibility weight after clamping and exponentiation.
|
||||
- **`novelty_bonus`** — the additive bonus derived from the document's novelty score.
|
||||
- **`confidence_gate`** — either `1.0` (signal passes) or `0.0` (signal is gated out).
|
||||
- **`market_ctx_multiplier`** — a multiplicative boost from market conditions, always `>= 1.0`.
|
||||
- **`combined`** — the final composite weight used by the aggregation engine.
|
||||
|
||||
The `ScoringConfig` frozen dataclass holds all tunable parameters for the scoring functions — half-life hours per window, credibility bounds, novelty bonus cap, confidence floor, and market context thresholds. A module-level `DEFAULT_CONFIG` singleton provides the production defaults, but every scoring function accepts an optional `config` parameter so that tests and alternative configurations can override any parameter without modifying global state.
|
||||
|
||||
---
|
||||
|
||||
## The Composite Weight Formula
|
||||
|
||||
The `compute_signal_weight()` function in `services/aggregation/scoring.py` computes the combined weight for a single document signal. The formula is:
|
||||
|
||||
```
|
||||
combined = gate × recency × credibility × (1 + novelty_bonus) × market_context_multiplier
|
||||
```
|
||||
|
||||
Each factor is computed independently and then multiplied together. This multiplicative structure means that any single factor can zero out the entire weight (the confidence gate) or amplify it (the market context multiplier), and the interaction between factors is naturally captured — a highly credible, very recent document with novel information in a volatile market receives the maximum possible weight, while a stale, low-credibility document with routine information receives a weight close to zero.
|
||||
|
||||
The following sections describe each component in detail.
|
||||
|
||||
---
|
||||
|
||||
## Confidence Gate
|
||||
|
||||
The confidence gate is the first and most decisive filter. If the extraction confidence for a document falls below the `confidence_floor` threshold — set to `0.2` in the default `ScoringConfig` — the gate evaluates to `0.0` and the entire combined weight becomes zero. The document is effectively excluded from aggregation. If the confidence meets or exceeds the threshold, the gate evaluates to `1.0` and has no further effect on the weight.
|
||||
|
||||
This binary gate exists because documents with very low extraction confidence are too unreliable to aggregate. A confidence of 0.15 typically means the LLM struggled to parse the document — perhaps the text was truncated, the language was ambiguous, or the document type was unusual. Including such signals would add noise rather than information. The threshold of 0.2 is deliberately low; it filters only the most unreliable extractions while allowing moderately confident signals to participate (their lower confidence is reflected through the credibility component instead).
|
||||
|
||||
---
|
||||
|
||||
## Recency Decay
|
||||
|
||||
The `recency_weight()` function computes an exponential decay based on how old a document is relative to the aggregation anchor time. The formula is:
|
||||
|
||||
```
|
||||
w = 2^(−age_hours / half_life)
|
||||
```
|
||||
|
||||
A document published exactly one half-life ago receives a recency weight of `0.5`. A document published two half-lives ago receives `0.25`, and so on. A document published at or after the reference time receives the maximum weight of `1.0`.
|
||||
|
||||
The half-life varies by trend window, reflecting the intuition that shorter windows need faster decay to stay responsive, while longer windows should give older documents more influence. The default half-lives, configured in `ScoringConfig.half_life_hours`, are:
|
||||
|
||||
| Window | Half-Life |
|
||||
|--------|-----------|
|
||||
| `intraday` | 2 hours |
|
||||
| `1d` | 12 hours |
|
||||
| `7d` | 72 hours (3 days) |
|
||||
| `30d` | 240 hours (10 days) |
|
||||
| `90d` | 720 hours (30 days) |
|
||||
|
||||
For the intraday window, a document published four hours ago already has a recency weight of `0.25` — it is rapidly losing influence as newer information arrives. For the 90-day window, that same four-hour-old document still has a recency weight of essentially `1.0`, because the 30-day half-life means age only becomes significant over weeks.
|
||||
|
||||
A floor value of `min_recency_weight = 0.01` prevents very old documents from being completely zeroed out. Even a document from months ago retains a trace-level weight of 1%, ensuring it can still contribute to trend computation if no newer signals exist. Both timestamps are normalized to UTC; naive datetimes are treated as UTC to avoid timezone-related scoring errors.
|
||||
|
||||
---
|
||||
|
||||
## Source Credibility
|
||||
|
||||
The `credibility_weight()` function transforms a source's credibility score into a weight component. The raw credibility value — a float between 0.0 and 1.0 stored in the `document_intelligence` table — is first clamped to the range `[0.1, 1.0]` using the `credibility_floor` and `credibility_ceiling` parameters from `ScoringConfig`. This clamping ensures that even the least credible sources retain a minimum weight of 0.1 rather than being completely silenced, while preventing any source from exceeding a weight of 1.0.
|
||||
|
||||
After clamping, the value is raised to the `credibility_exponent` power. The default exponent is `1.0`, which means the clamped credibility passes through unchanged. Setting the exponent above 1.0 would penalize low-credibility sources more aggressively — for example, an exponent of 2.0 would reduce a credibility of 0.5 to a weight of 0.25. Setting it below 1.0 would flatten the curve, making the system more tolerant of lower-credibility sources. The exponent is configurable through `ScoringConfig` to allow operators to tune the credibility sensitivity without changing the scoring code.
|
||||
|
||||
---
|
||||
|
||||
## Novelty Bonus
|
||||
|
||||
The novelty bonus rewards documents that contain genuinely new information. The bonus is computed as:
|
||||
|
||||
```
|
||||
novelty_bonus = novelty_score × novelty_bonus_max
|
||||
```
|
||||
|
||||
where `novelty_score` is the 0.0-to-1.0 value produced by the extraction model (see the `ExtractionResult` schema in [Page 2](02-ai-agent-processing-and-extraction.md)) and `novelty_bonus_max` is `0.25` by default. This means the bonus ranges from `0.0` (completely routine information) to `0.25` (maximally novel information), providing up to a 25% boost to the signal weight.
|
||||
|
||||
The bonus enters the composite formula as `(1 + novelty_bonus)`, so it acts as a multiplicative amplifier on the base weight. A document with a novelty score of 1.0 gets its weight multiplied by 1.25; a document with a novelty score of 0.0 gets multiplied by 1.0 (no change). This design ensures that novelty can only increase a signal's weight, never decrease it — routine information is not penalized, it simply does not receive the bonus.
|
||||
|
||||
---
|
||||
|
||||
## Market Context Multiplier
|
||||
|
||||
The `market_context_multiplier()` function computes a boost factor based on real-time market conditions for the ticker being aggregated. The multiplier is always `>= 1.0`, meaning market context can only amplify signal weights, never reduce them. When no market context data is available (the `MarketContext` object from `services/shared/schemas.py` has `has_data == False`), the multiplier defaults to `1.0`.
|
||||
|
||||
Two market features contribute to the boost:
|
||||
|
||||
**Volatility boost.** When the ticker's price volatility exceeds the `volatility_recency_boost_threshold` (default `1.0` in price units), the excess volatility is transformed through a logarithmic scaling function: `log₁₊(excess) × 0.15`. The logarithmic scaling prevents extreme volatility from producing runaway weight amplification. The boost is capped at `volatility_recency_boost_max = 0.30`, so the maximum volatility contribution is a 30% weight increase. The rationale is that in highly volatile markets, fresh intelligence is disproportionately valuable — a signal about NVDA matters more when NVDA is swinging 5% intraday than when it is trading in a tight range.
|
||||
|
||||
**Volume surge boost.** When the ticker's volume change percentage exceeds `volume_surge_threshold_pct = 50.0%` (meaning trading volume is at least 50% above the prior period's average), a flat `volume_surge_boost = 0.15` is added. Unlike the volatility boost, this is binary — either the volume threshold is met and the full 15% boost applies, or it is not and no boost is added. High-volume moves carry more conviction because they represent broader market participation rather than thin-market noise.
|
||||
|
||||
The two boosts are additive within the multiplier: `multiplier = 1.0 + volatility_boost + volume_surge_boost`. In the most extreme case — high volatility and a volume surge — the combined multiplier reaches `1.0 + 0.30 + 0.15 = 1.45`, amplifying the signal weight by 45%. The `MarketContext` data is fetched by `services/aggregation/market_context.py` from the market data tables in PostgreSQL, using the same ticker and window parameters as the impact record query.
|
||||
|
||||
---
|
||||
|
||||
## Sentiment Mapping
|
||||
|
||||
Before signals can be aggregated into trend summaries, the categorical sentiment labels from the extraction output must be converted to numeric values. The `sentiment_to_numeric()` function in `services/aggregation/scoring.py` performs this mapping:
|
||||
|
||||
| Sentiment Label | Numeric Value |
|
||||
|----------------|---------------|
|
||||
| `positive` | `+1.0` |
|
||||
| `negative` | `-1.0` |
|
||||
| `neutral` | `0.0` |
|
||||
| `mixed` | `0.0` |
|
||||
|
||||
The mapping is case-insensitive. Any unrecognized label defaults to `0.0`. The choice to map both `neutral` and `mixed` to `0.0` is deliberate — a mixed-sentiment document (one that contains both positive and negative signals for the same company) should not push the trend in either direction. The contradiction between the positive and negative aspects is captured separately by the contradiction detection system described in [Page 4](04-trend-aggregation-and-accumulating-signals.md), rather than being baked into the sentiment value itself.
|
||||
|
||||
For macro signals, the direction-to-sentiment mapping in `services/aggregation/worker.py` follows the same pattern: `positive` maps to `+1.0`, `negative` to `-1.0`, and both `mixed` and `neutral` to `0.0`. For competitive signals built by `build_pattern_weighted_signals()` in `services/aggregation/signal_propagation.py`, the sentiment is derived from the pattern's directional bias: `+1.0` if `bullish_pct > bearish_pct`, `-1.0` otherwise.
|
||||
|
||||
---
|
||||
|
||||
## Weighted Sentiment Average
|
||||
|
||||
The `weighted_sentiment_average()` function computes the central metric that drives trend direction: a weight-adjusted average sentiment across all signals for a ticker in a given window. The formula is:
|
||||
|
||||
```
|
||||
weighted_avg = Σ(combined_weight × impact_score × sentiment_value) / Σ(combined_weight × impact_score)
|
||||
```
|
||||
|
||||
Each signal contributes its sentiment value scaled by both its composite weight and its impact score. The denominator normalizes by the total effective weight, producing a value in the range `[-1.0, +1.0]`. A result near `+1.0` means the weighted evidence is overwhelmingly positive; near `-1.0` means overwhelmingly negative; near `0.0` means either neutral or evenly split.
|
||||
|
||||
The use of `combined_weight × impact_score` as the effective weight means that high-impact, high-weight signals dominate the average. A single high-confidence, recent, credible document with a strong impact score can outweigh several older, lower-impact documents — which is the intended behavior. The aggregation engine in `services/aggregation/worker.py` passes this weighted average to `derive_trend_direction()`, which maps it to a `TrendDirection` enum value (bullish, bearish, mixed, or neutral) using the thresholds described in [Page 4](04-trend-aggregation-and-accumulating-signals.md).
|
||||
|
||||
If the total effective weight is zero — either because no signals exist or all signals were gated out by the confidence floor — the function returns `0.0`, which maps to a neutral trend direction.
|
||||
|
||||
---
|
||||
|
||||
## The Three Signal Layers
|
||||
|
||||
The aggregation engine in `services/aggregation/worker.py` does not treat all intelligence sources equally. Signals flow through three independent layers, each with a different relative weight, before being concatenated into a single `WeightedSignal` list for trend computation. This layered architecture allows the system to incorporate diverse intelligence sources while controlling how much influence each source type has on the final trend.
|
||||
|
||||
### Layer 1 — Company Signals (Weight: 1.0)
|
||||
|
||||
Company signals are the primary layer. They are built by `build_weighted_signals()` in `services/aggregation/worker.py` from `document_impact_records` — the per-company extraction output produced by the Document Intelligence Extractor (see [Page 2](02-ai-agent-processing-and-extraction.md)). Each impact record's sentiment is converted via `sentiment_to_numeric()`, and its impact score is used directly without any layer-level scaling. The `compute_signal_weight()` function produces the composite weight using the document's publication time, source credibility, novelty score, extraction confidence, and the ticker's current market context.
|
||||
|
||||
Company signals carry a relative weight of `1.0` — they are the baseline against which other layers are measured. This reflects the design principle that direct, company-specific intelligence (an earnings report about AAPL, a product launch by TSLA, a lawsuit against META) is the most relevant and reliable signal for that company's trend.
|
||||
|
||||
### Layer 2 — Macro Signals (Weight: 0.3)
|
||||
|
||||
Macro signals capture the indirect impact of global events on individual companies. They are built by `build_macro_weighted_signals()` in `services/aggregation/worker.py` from `macro_impact_records` — the per-company impact scores computed by the exposure-based interpolation engine after the Global Event Classifier processes a macro news article. The sentiment is mapped from the `impact_direction` field (`positive` → `+1.0`, `negative` → `-1.0`, `mixed`/`neutral` → `0.0`), and the impact score is scaled by `MACRO_SIGNAL_WEIGHT`, which defaults to `0.3` in `AggregationConfig`.
|
||||
|
||||
The 0.3 weight means that a macro signal's impact score is reduced to 30% of its raw value before entering the aggregation. This attenuation reflects the inherent uncertainty in macro-to-company impact estimation — a tariff announcement might affect XOM's revenue, but the magnitude depends on exposure profiles, supply chain flexibility, and competitive dynamics that the interpolation engine can only approximate. By weighting macro signals at 0.3 relative to company signals at 1.0, the system ensures that macro intelligence informs the trend without overwhelming direct company-specific evidence.
|
||||
|
||||
The recency decay, credibility, and confidence gating for macro signals use the same `compute_signal_weight()` function as company signals. The `published_at` timestamp comes from the global event's source document (the macro news article), and the `source_credibility` and `extraction_confidence` both use the macro impact record's `confidence` field.
|
||||
|
||||
### Layer 3 — Competitive Signals (Weight: 0.2)
|
||||
|
||||
Competitive signals capture cross-company effects: when a catalyst hits one company, historical patterns suggest how competitors might be affected. They are built by `build_pattern_weighted_signals()` in `services/aggregation/signal_propagation.py` from two sources: `HistoricalPattern` objects (self-company patterns mined by `services/aggregation/pattern_matcher.py`) and `CompetitiveSignalRecord` objects (cross-company propagation signals stored in `competitive_signal_records`).
|
||||
|
||||
For historical patterns, the sentiment is derived from the pattern's directional bias (`+1.0` if `bullish_pct > bearish_pct`, `-1.0` otherwise), and the impact score is the pattern's `avg_strength` multiplied by `competitive_signal_weight` (default `0.2` from `CompetitiveConfig`). The `published_at` for recency decay uses the pattern's `data_end` — the most recent data point in the pattern's sample — and the `extraction_confidence` uses the pattern's `pattern_confidence`. Source credibility is set to `1.0` because patterns are derived from validated historical data, and novelty is fixed at `0.5`.
|
||||
|
||||
For competitive signal records, the same structure applies: sentiment from `signal_direction`, impact from `signal_strength × competitive_signal_weight`, recency from `computed_at`, and confidence from `pattern_confidence`.
|
||||
|
||||
The 0.2 weight makes competitive signals the lightest layer. This is appropriate because competitive signal propagation involves the most inference — the system is predicting how Company B will react based on what happened to Company A in historically similar situations. The signal is valuable as supplementary evidence but should not drive trend direction on its own.
|
||||
|
||||
---
|
||||
|
||||
## Signal Merging in the Aggregation Engine
|
||||
|
||||
The `aggregate_company_window()` function in `services/aggregation/worker.py` orchestrates the merging of all three layers for a single ticker and window. The process follows a clear sequence:
|
||||
|
||||
1. **Fetch company impact records** from `document_impact_records` for the ticker within the window's time range.
|
||||
2. **Fetch market context** for the ticker from market data tables.
|
||||
3. **Build company weighted signals** via `build_weighted_signals()`.
|
||||
4. **Check the macro toggle** — query `risk_configs` for the `macro_enabled` flag, then fetch and merge macro signals if enabled.
|
||||
5. **Check the competitive toggle** — query `risk_configs` for the `competitive_enabled` flag, then fetch patterns, fetch competitive signals, and merge if enabled.
|
||||
6. **Concatenate** all `WeightedSignal` lists into a single list.
|
||||
7. **Assemble the `TrendSummary`** from the merged signals.
|
||||
|
||||
The concatenation in step 6 is a simple list append — `signals = signals + macro_signals` followed by `signals = signals + pattern_weighted`. There is no re-weighting or normalization at the merge point. The relative influence of each layer is already encoded in the impact scores (scaled by 0.3 for macro, 0.2 for competitive, 1.0 for company) and in the composite weights computed by `compute_signal_weight()`. The `weighted_sentiment_average()` function then naturally produces a sentiment average that reflects these relative weights.
|
||||
|
||||
---
|
||||
|
||||
## Runtime Toggles and Graceful Degradation
|
||||
|
||||
Both the macro and competitive signal layers can be enabled or disabled at runtime through the `risk_configs` PostgreSQL table, without restarting any service. The toggle state is read fresh from the database at the start of every aggregation cycle — there is no caching — so changes take effect on the very next cycle.
|
||||
|
||||
The `fetch_macro_enabled()` function in `services/aggregation/worker.py` queries the most recent active `risk_configs` row and reads the `config->>'macro_enabled'` JSON field. If the field is explicitly set to `"true"` or `"false"`, that value overrides the `AggregationConfig` default. If no config row exists or the field is absent, the function returns `None` and the engine falls back to the `AggregationConfig.macro_enabled` default (which is `True`). The `fetch_competitive_enabled()` function follows the identical pattern for the `competitive_enabled` field.
|
||||
|
||||
When a layer is disabled, the aggregation engine simply skips the fetch-and-merge step for that layer. Company signals are always computed — they cannot be toggled off. This means the system degrades gracefully: disabling the macro layer produces trends based on company signals alone (plus competitive signals if enabled), and disabling the competitive layer produces trends based on company and macro signals. Disabling both layers reduces the engine to its original single-layer behavior, using only direct document intelligence.
|
||||
|
||||
Crucially, disabling a layer does not stop upstream processing. When the macro layer is disabled, the Global Event Classifier continues to classify macro events and the interpolation engine continues to compute `macro_impact_records`. The data accumulates in PostgreSQL. When the layer is re-enabled, the aggregation engine immediately picks up all the macro impact records that were computed while the layer was disabled — there is no data loss or gap in coverage. The same applies to competitive signals: pattern mining and signal propagation continue regardless of the toggle state.
|
||||
|
||||
If the competitive signal fetch fails at runtime (for example, due to a database timeout), the aggregation engine catches the exception, logs it, and continues with company and macro signals only. This exception-based graceful degradation ensures that a transient failure in one layer does not block trend computation entirely.
|
||||
|
||||
---
|
||||
|
||||
## What Comes Next
|
||||
|
||||
At this point, every document intelligence record, macro impact record, and competitive signal record has been transformed into a `WeightedSignal` with a composite weight that encodes recency, credibility, novelty, confidence, and market conditions. The three signal layers have been merged into a single list, and the weighted sentiment average has been computed. But a single aggregation cycle produces only a snapshot — a point-in-time view of the evidence. The real power of the system emerges when these snapshots accumulate across multiple documents and time windows, building a case for action. [Page 4 — Trend Aggregation and Accumulating Signals](04-trend-aggregation-and-accumulating-signals.md) explains how the aggregation engine computes `TrendSummary` objects across five time windows, how consecutive same-direction signals strengthen trend confidence and escalate the system's response from neutral observation to actionable trading recommendations, and how contradiction detection and evidence ranking ensure that the trend reflects genuine consensus rather than noise.
|
||||
+267
@@ -0,0 +1,267 @@
|
||||
# Page 4 — Trend Aggregation and Accumulating Signals
|
||||
|
||||
The scoring layer described in [Page 3](03-signal-scoring-and-weighted-signals.md) transforms every intelligence record into a `WeightedSignal` — a document reference paired with a composite weight that encodes recency, credibility, novelty, confidence, and market conditions. Three independent signal layers (Company at weight 1.0, Macro at 0.3, Competitive at 0.2) each produce `WeightedSignal` objects that are concatenated into a single list. But a single list of weighted signals is still just raw material. The aggregation engine in `services/aggregation/worker.py` is where that raw material becomes a decision-grade assessment: a `TrendSummary` object that captures the direction, strength, confidence, contradiction level, and supporting evidence for a ticker across a specific time window. This page explains how that transformation works — from weighted sentiment averages through trend direction derivation, contradiction detection, evidence ranking, and confidence computation — and, critically, how consecutive signals pointing in the same direction accumulate across documents and time windows to escalate the system's response from passive observation to actionable trading recommendations.
|
||||
|
||||
For a visual overview of the accumulation and escalation process, see the [Trend Accumulation and Escalation diagram](diagrams/trend-accumulation-escalation.md). For how the three signal layers merge into the aggregation engine, see the [Three-Layer Signal Merging diagram](diagrams/three-layer-signal-merging.md).
|
||||
|
||||
---
|
||||
|
||||
## Five Time Windows
|
||||
|
||||
The aggregation engine does not compute a single trend for each ticker. It computes five, one for each time window defined in `services/aggregation/worker.py`:
|
||||
|
||||
| Window | Lookback Duration |
|
||||
|--------|-------------------|
|
||||
| `intraday` | 12 hours |
|
||||
| `1d` | 1 day |
|
||||
| `7d` | 7 days |
|
||||
| `30d` | 30 days |
|
||||
| `90d` | 90 days |
|
||||
|
||||
Each window produces an independent `TrendSummary` by fetching all impact records, macro impacts, and competitive signals for the ticker within that window's time range. The `aggregate_company_window()` function in `services/aggregation/worker.py` orchestrates this per-window computation: it determines the time range from the window's lookback duration, fetches `document_impact_records` from PostgreSQL, retrieves market context, builds company weighted signals, checks the macro and competitive runtime toggles (see [Page 3](03-signal-scoring-and-weighted-signals.md) for toggle details), merges any enabled layer signals, and then assembles the `TrendSummary`.
|
||||
|
||||
The five-window design serves a specific purpose. Short windows (intraday, 1d) capture fast-moving sentiment shifts — a breaking earnings miss, a sudden regulatory action — while long windows (30d, 90d) reveal sustained trends that persist across many documents and news cycles. A ticker might show a bearish intraday trend after a single negative article, but a neutral 30-day trend because the broader evidence base is balanced. The recommendation engine downstream (described in [Page 5](05-recommendation-generation.md)) evaluates each window's `TrendSummary` independently, so the system can respond to both short-term catalysts and long-term directional shifts.
|
||||
|
||||
The `aggregate_company()` function iterates over all effective windows (configurable via `AggregationConfig.windows`, defaulting to all five) and calls `aggregate_company_window()` for each one. This means a single aggregation cycle for one ticker produces up to five `TrendSummary` objects, each reflecting a different temporal perspective on the same underlying evidence.
|
||||
|
||||
---
|
||||
|
||||
## Trend Direction Derivation
|
||||
|
||||
Once the weighted sentiment average has been computed from the merged signal list (see the `weighted_sentiment_average()` function described in [Page 3](03-signal-scoring-and-weighted-signals.md)), the `derive_trend_direction()` function in `services/aggregation/worker.py` maps that numeric value to a `TrendDirection` enum. The rules are evaluated in a specific order, and the first matching rule wins:
|
||||
|
||||
1. **Mixed** — If the contradiction score exceeds `0.10` (the `MIXED_THRESHOLD` constant) *and* the absolute value of the average sentiment is below `0.30`, the direction is `MIXED`. This rule fires first because high contradiction with a weak directional signal indicates genuine disagreement in the evidence — the trend is not simply neutral, it is actively contested.
|
||||
|
||||
2. **Bullish** — If the average sentiment is `≥ 0.15` (the `BULLISH_THRESHOLD` constant), the direction is `BULLISH`. This means the weight-adjusted evidence leans positive with enough conviction to cross the threshold.
|
||||
|
||||
3. **Bearish** — If the average sentiment is `≤ -0.15` (the `BEARISH_THRESHOLD` constant), the direction is `BEARISH`. The symmetric threshold ensures that bullish and bearish classifications require the same magnitude of evidence.
|
||||
|
||||
4. **Neutral** — If none of the above conditions are met, the direction is `NEUTRAL`. This covers the range where the average sentiment falls between -0.15 and +0.15 without high contradiction — the evidence is either balanced or insufficient to establish a directional lean.
|
||||
|
||||
The mixed-first evaluation order is important. Consider a scenario where five documents are bullish and four are bearish, all with similar weights. The weighted sentiment average might be slightly positive (say, +0.08), which would normally map to neutral. But the contradiction score — computed from the minority/majority weight split — would be high (close to 0.44). The mixed rule catches this case: the evidence is not neutral, it is conflicted. This distinction matters downstream because mixed trends receive different treatment in the recommendation engine than neutral trends.
|
||||
|
||||
---
|
||||
|
||||
## Contradiction Detection
|
||||
|
||||
The contradiction detection module in `services/aggregation/contradiction.py` provides a structured analysis of disagreement within the signal set. Rather than collapsing contradictory evidence into a single number, it produces a `ContradictionResult` containing both an overall score and a list of `DisagreementDetail` objects that explain *where* the disagreement lies.
|
||||
|
||||
The `detect_contradictions()` function runs two analyses:
|
||||
|
||||
### Sentiment Disagreement
|
||||
|
||||
The `_detect_sentiment_disagreement()` function examines whether both positive and negative sentiment signals exist in the signal set. For each signal with a non-zero effective weight (`combined_weight × impact_score > 0`), it classifies the signal as positive or negative based on its `sentiment_value` and accumulates the effective weight for each side. If both sides have at least one signal, it produces a `DisagreementDetail` with dimension `"sentiment"`, listing the document IDs and weights for each side, along with a human-readable description like "Sentiment split: 3 positive vs 2 negative signals (minority weight ratio 38%)".
|
||||
|
||||
### Catalyst-Level Disagreement
|
||||
|
||||
The `_detect_catalyst_disagreement()` function goes deeper. It groups signals by their `catalyst_type` (earnings, product_launch, regulatory, etc.) using `CatalystEntry` objects built from the `document_impact_records`. Within each catalyst group, it checks whether both positive and negative signals exist. If they do, it produces a `DisagreementDetail` with dimension `"catalyst:<type>"` — for example, `"catalyst:earnings"` when some documents interpret an earnings report positively and others negatively. This catalyst-level analysis is valuable because it pinpoints the specific topic of disagreement rather than just flagging that disagreement exists somewhere in the evidence.
|
||||
|
||||
### The Overall Contradiction Score
|
||||
|
||||
The `_compute_overall_score()` function computes the backward-compatible scalar contradiction score using the minority/majority weight ratio formula:
|
||||
|
||||
```
|
||||
contradiction_score = minority_weight / total_weight
|
||||
```
|
||||
|
||||
where `minority_weight` is the smaller of the positive and negative effective weights, and `total_weight` is their sum. Signals with zero effective weight or neutral sentiment are excluded. The score ranges from `0.0` (complete agreement — all signals point the same direction) to `0.5` (perfect split — positive and negative weights are exactly equal). A score of `0.0` means no contradiction at all. A score above `0.10` combined with a weak average sentiment triggers the mixed direction classification in `derive_trend_direction()`.
|
||||
|
||||
The contradiction score also feeds directly into the confidence computation as a penalty, described in the next section. High contradiction reduces the system's confidence in the trend, which in turn affects whether the trend can escalate to actionable recommendations.
|
||||
|
||||
---
|
||||
|
||||
## Evidence Ranking
|
||||
|
||||
Not all documents contributing to a trend are equally important. The `rank_evidence()` function in `services/aggregation/worker.py` delegates to the evidence ranking module (`services/aggregation/evidence.py`) to produce ordered lists of the most influential supporting and opposing documents. The ranking uses a composite scoring approach configured by `EvidenceRankConfig`, considering multiple factors:
|
||||
|
||||
- **Weight** — the signal's composite weight from the scoring layer, reflecting recency, credibility, novelty, confidence, and market context.
|
||||
- **Impact** — the extraction's impact score for the company, reflecting how significant the document's content is.
|
||||
- **Recency** — how recently the document was published, with more recent documents ranked higher.
|
||||
- **Confidence** — the extraction confidence, reflecting how reliably the LLM parsed the document.
|
||||
|
||||
Signals are split into supporting (positive sentiment) and opposing (negative sentiment) groups. Neutral and mixed sentiment signals are excluded from evidence lists — they do not argue for or against the trend direction. Within each group, signals are sorted by their composite rank score in descending order, and the top entries (up to `MAX_EVIDENCE_REFS = 10` per side) are returned as document ID lists.
|
||||
|
||||
The `assemble_trend_with_evidence()` function in `services/aggregation/worker.py` uses the detailed variant `rank_evidence_detailed()` to get `RankedEvidence` objects that include the individual scoring components (weight, impact, recency, confidence, sentiment value). These detailed rankings are persisted to the `trend_evidence` table for auditability, while the document ID lists are stored directly in the `TrendSummary` as `top_supporting_evidence` and `top_opposing_evidence`.
|
||||
|
||||
The evidence ranking serves two purposes. First, it provides the recommendation engine with the most relevant documents to cite in its thesis generation (see [Page 5](05-recommendation-generation.md)). Second, it gives human reviewers a quick way to understand *why* the system reached a particular trend assessment — the top-ranked documents are the ones that most influenced the direction and strength.
|
||||
|
||||
---
|
||||
|
||||
## Confidence Computation
|
||||
|
||||
The `compute_trend_confidence()` function in `services/aggregation/worker.py` produces the confidence score for a `TrendSummary`. This score is critical because it directly gates whether a trend can produce actionable recommendations — the eligibility evaluation in `services/recommendation/eligibility.py` requires a minimum confidence of `0.35` to generate any recommendation at all, and higher confidence thresholds control escalation to paper and live trading modes.
|
||||
|
||||
Confidence is computed from four components:
|
||||
|
||||
### Unique Source Count
|
||||
|
||||
The function counts the number of unique document IDs across all active signals (those with `combined_weight > 0`). This count is divided by 15 and capped at `0.8`:
|
||||
|
||||
```
|
||||
count_factor = min(unique_sources / 15.0, 0.8)
|
||||
```
|
||||
|
||||
A trend backed by 15 or more unique source documents reaches the maximum count contribution of `0.8`. A trend backed by a single document gets only `0.067`. This component rewards breadth of evidence — a trend confirmed by many independent sources is more trustworthy than one driven by a single article, regardless of how high that article's individual weight might be.
|
||||
|
||||
### Average Extraction Credibility
|
||||
|
||||
The average credibility weight across all active signals provides a baseline quality measure. If most contributing documents come from high-credibility sources, this component is high. If the evidence is dominated by low-credibility sources, confidence is penalized accordingly.
|
||||
|
||||
### Signal Agreement with Sample-Size Dampening
|
||||
|
||||
The agreement ratio measures what fraction of directional signals (bullish + bearish, excluding neutral) agree on the majority direction. If 8 out of 10 directional signals are bullish, the raw agreement is `0.8`. But raw agreement is misleading with small sample sizes — 1 out of 1 signals agreeing gives a perfect `1.0` agreement, which is not meaningful.
|
||||
|
||||
To address this, the agreement is dampened by a logarithmic sample-size factor:
|
||||
|
||||
```
|
||||
agreement_dampener = min(1.0, log₂(unique_sources + 1) / log₂(8))
|
||||
```
|
||||
|
||||
This dampener saturates at `1.0` when `unique_sources` reaches approximately 7 (since `log₂(8) = 3.0` and `log₂(8) = 3.0`). With fewer sources, the dampener reduces the agreement contribution: 1 source gives a dampener of `0.33`, 3 sources give `0.67`, and 7 sources give the full `1.0`. The log₂ scaling means that each additional source provides diminishing marginal improvement to the dampener, which matches the intuition that the jump from 1 to 3 sources is far more meaningful than the jump from 15 to 17.
|
||||
|
||||
### Contradiction Penalty
|
||||
|
||||
The contradiction score computed by `services/aggregation/contradiction.py` is applied as a direct penalty:
|
||||
|
||||
```
|
||||
contradiction_penalty = contradiction_score × 0.4
|
||||
```
|
||||
|
||||
A contradiction score of `0.5` (perfect split) produces a penalty of `0.2`, which is substantial enough to push a moderately confident trend below the eligibility threshold.
|
||||
|
||||
### The Combined Formula
|
||||
|
||||
The four components are combined as:
|
||||
|
||||
```
|
||||
confidence = 0.3 × count_factor + 0.3 × avg_credibility + 0.4 × agreement − contradiction_penalty
|
||||
```
|
||||
|
||||
The result is clamped to `[0.0, 1.0]`. The weighting gives signal agreement the largest share (40%), reflecting the principle that consensus among diverse sources is the strongest indicator of a reliable trend. Source count and credibility each contribute 30%, providing a balanced assessment of evidence breadth and quality. The contradiction penalty can reduce confidence significantly — a highly contradicted trend with a score of 0.4 loses 0.16 points of confidence, which can easily drop it below the 0.35 eligibility gate.
|
||||
|
||||
---
|
||||
|
||||
## How Accumulating Signals Escalate Decisions
|
||||
|
||||
The trend direction, strength, and confidence computed by the aggregation engine are not just descriptive — they directly determine what action the system takes. The escalation path from passive observation to active trading is governed by the eligibility thresholds defined in `services/recommendation/eligibility.py`, and the key insight is that consecutive signals pointing in the same direction naturally strengthen the trend metrics that control this escalation.
|
||||
|
||||
### The Escalation Ladder
|
||||
|
||||
The `EligibilityConfig` dataclass in `services/recommendation/eligibility.py` defines the thresholds that map trend metrics to actions:
|
||||
|
||||
**Neutral (no recommendation).** A trend fails the eligibility gates entirely when confidence is below `0.35`, trend strength is below `0.10`, contradiction exceeds `0.60`, evidence count is below `2`, or the direction is neutral. The `_check_gates()` function evaluates these hard gates — if any gate fails, no recommendation is generated for that window.
|
||||
|
||||
**Watch.** A trend that passes the gates but has a direction of mixed, or has strength below `0.25` with confidence below `0.50`, maps to a `WATCH` action via `_determine_action()`. This is the system's way of saying "something is happening, but the evidence is not strong enough to act on." Watch recommendations are always `informational` mode — they are logged for human review but never trigger trades.
|
||||
|
||||
**Hold.** When the trend has a clear direction (bullish or bearish) but strength remains below `0.25` while confidence reaches `0.50` or above, the action maps to `HOLD`. This indicates that the directional signal is real but not yet strong enough for a position change. Like watch, hold recommendations are `informational` mode.
|
||||
|
||||
**Buy / Sell.** When trend strength reaches `0.25` or above with a bullish direction, the action is `BUY`. With a bearish direction at the same strength threshold, the action is `SELL`. These are the only actions that can escalate beyond informational mode — `_determine_mode()` evaluates whether the recommendation qualifies for `paper_eligible` (confidence ≥ `0.50`) or `live_eligible` (confidence ≥ `0.70`, contradiction ≤ `0.25`, evidence ≥ `5`).
|
||||
|
||||
### How Accumulation Drives Escalation
|
||||
|
||||
Consider a ticker that starts with no recent intelligence. The first bearish article arrives — a single document with negative sentiment. In the intraday window, this produces:
|
||||
|
||||
- **Trend strength** = `|avg_sentiment|` ≈ the absolute weighted sentiment from one signal, likely close to the impact score.
|
||||
- **Confidence** = low, because `count_factor = min(1/15, 0.8) = 0.067` and the agreement dampener is only `log₂(2)/log₂(8) = 0.33`.
|
||||
- **Direction** = bearish (if the weighted sentiment is ≤ -0.15).
|
||||
|
||||
With confidence well below `0.35`, this trend fails the eligibility gate entirely. No recommendation is generated. The system is in the neutral state.
|
||||
|
||||
A second bearish article arrives hours later. Now the intraday window has two signals:
|
||||
|
||||
- **Unique sources** = 2, so `count_factor = 0.133` and `agreement_dampener = log₂(3)/log₂(8) ≈ 0.53`.
|
||||
- **Agreement** = `1.0 × 0.53 = 0.53` (both signals agree on bearish).
|
||||
- **Confidence** ≈ `0.3 × 0.133 + 0.3 × avg_cred + 0.4 × 0.53` — likely around `0.35-0.45` depending on credibility.
|
||||
|
||||
If confidence crosses `0.35` and strength exceeds `0.10`, the trend passes the eligibility gates. But with strength below `0.25`, the action is `WATCH` or `HOLD` depending on confidence.
|
||||
|
||||
A third and fourth bearish article arrive over the next day. The 1-day window now has four agreeing signals:
|
||||
|
||||
- **Unique sources** = 4, so `count_factor = 0.267` and `agreement_dampener = log₂(5)/log₂(8) ≈ 0.77`.
|
||||
- **Agreement** = `1.0 × 0.77 = 0.77`.
|
||||
- **Confidence** ≈ `0.3 × 0.267 + 0.3 × avg_cred + 0.4 × 0.77` — likely `0.50-0.60`.
|
||||
- **Strength** = `|avg_sentiment|` — with four bearish signals and no contradicting evidence, this could easily exceed `0.25`.
|
||||
|
||||
Now the trend maps to `SELL` with `paper_eligible` mode (confidence ≥ `0.50`). The system has escalated from no recommendation to a paper-eligible sell recommendation purely through the accumulation of consistent bearish evidence.
|
||||
|
||||
If the bearish evidence continues — more documents, more sources, higher credibility — confidence climbs further. At confidence ≥ `0.70` with contradiction ≤ `0.25` and evidence ≥ `5`, the recommendation reaches `live_eligible` mode, the highest escalation level.
|
||||
|
||||
The same process works in reverse for bullish accumulation: consecutive positive signals strengthen the bullish trend, increase confidence through source diversity and agreement, and escalate from watch through hold to buy.
|
||||
|
||||
### The Role of Contradiction in Preventing False Escalation
|
||||
|
||||
Accumulation only works when signals agree. If the fifth article about a ticker is bullish while the previous four were bearish, the contradiction score jumps — `minority_weight / total_weight` increases because the minority (bullish) side now has non-zero weight. This has two effects: the contradiction penalty reduces confidence (potentially dropping it below an eligibility threshold), and if the contradiction exceeds `0.10` with `|avg_sentiment| < 0.30`, the direction flips to mixed, which maps to `WATCH` regardless of strength. The system effectively de-escalates when the evidence becomes contested, requiring a clearer consensus before re-escalating.
|
||||
|
||||
---
|
||||
|
||||
## Trend Projections
|
||||
|
||||
After the `TrendSummary` is assembled and persisted, the aggregation engine computes a forward-looking `TrendProjection` via `compute_projection()` in `services/aggregation/projection.py`. Projections estimate where the trend is heading based on current momentum, macro signal decay, and upcoming catalysts. They are advisory — they do not directly trigger recommendations — but they provide valuable context for human reviewers and can inform future automated decision-making.
|
||||
|
||||
### Momentum
|
||||
|
||||
The `compute_trend_momentum()` function computes the rate of change in signed trend strength between the current and previous aggregation cycles. If the current window shows a bearish trend at strength `0.40` and the previous cycle showed bearish at `0.30`, the momentum is `-0.10` (strengthening bearish). If no previous data is available, the function uses a heuristic: momentum is estimated as half the current signed strength, providing a reasonable baseline for new trends.
|
||||
|
||||
Momentum enters the projection as a half-weighted adjustment to the current signed strength:
|
||||
|
||||
```
|
||||
momentum_projected_signed = direction_sign × current_strength + momentum × 0.5
|
||||
```
|
||||
|
||||
This means momentum influences the projection but does not dominate it — a strong current trend with weakening momentum still projects as directional, just with reduced strength.
|
||||
|
||||
### Macro Decay
|
||||
|
||||
The `project_macro_decay()` function estimates how active macro events will evolve over the projection horizon. Each macro event has an `estimated_duration` that maps to a decay half-life:
|
||||
|
||||
| Duration | Half-Life |
|
||||
|----------|-----------|
|
||||
| `short_term` | 1 day |
|
||||
| `medium_term` | 7 days |
|
||||
| `long_term` | 30 days |
|
||||
|
||||
For each event, the function computes the projected remaining impact at the end of the horizon using exponential decay: `future_factor = 2^(−future_age_days / half_life)`. The impact is further scaled by a severity weight (`critical`: 1.0, `high`: 0.75, `moderate`: 0.5, `low`: 0.25). Positive and negative macro impacts are accumulated separately, and the projected macro direction is determined by comparing the two sides — bullish if positive exceeds negative by 20%, bearish if the reverse, mixed if both are present without a clear majority.
|
||||
|
||||
When the macro layer is enabled and macro events exist, the projection blends the company-specific momentum projection with the macro trajectory. The macro weight is capped at `0.4` (40% of the blended projection), ensuring that macro signals inform but do not overwhelm the company-specific trend. The blending formula combines the signed company projection with the signed macro projection:
|
||||
|
||||
```
|
||||
blended = company_weight × momentum_projected + macro_weight × macro_signed
|
||||
```
|
||||
|
||||
### Driving Factors
|
||||
|
||||
The projection records a list of human-readable driving factors that explain what is influencing the projected direction. These include momentum descriptions ("Positive momentum (+0.150) in recent trend strength"), macro impact projections ("Macro signals project bearish impact (strength 0.350) over 7d"), and upcoming catalysts drawn from the trend's `dominant_catalysts` list (limited to the top 3). If no specific factors are identified, a baseline continuation factor is recorded.
|
||||
|
||||
### Divergence Detection
|
||||
|
||||
After computing the projected direction, the function compares it to the current trend direction. If they differ — for example, the current trend is bearish but the projection is bullish due to decaying negative macro events and positive momentum — the projection is flagged with `diverges_from_current = True` and a divergence driving factor is appended. Divergence signals are particularly valuable because they indicate that the trend may be about to reverse, giving the recommendation engine and human reviewers an early warning.
|
||||
|
||||
The projection also flags low confidence when `projected_confidence` falls below the default threshold of `0.3`. Projection confidence starts at 80% of the current trend confidence (reflecting the inherent uncertainty of forward-looking estimates), with a small boost if macro data is available and a further reduction if the macro layer is disabled entirely.
|
||||
|
||||
---
|
||||
|
||||
## Persistence
|
||||
|
||||
Each aggregation cycle persists its results to four PostgreSQL tables, creating a durable record of the trend assessment and its supporting evidence.
|
||||
|
||||
### `trend_windows` — Current State
|
||||
|
||||
The `persist_trend_summary()` function in `services/aggregation/worker.py` upserts the `TrendSummary` into the `trend_windows` table, keyed by `(entity_type, entity_id, window)`. Each cycle overwrites the previous row for that ticker and window, so `trend_windows` always reflects the most recent assessment. The row includes the trend direction, strength, confidence, contradiction score, disagreement details (as JSON), supporting and opposing evidence document IDs (as JSON arrays), dominant catalysts, material risks, market context, and the generation timestamp.
|
||||
|
||||
### `trend_history` — Time-Series Snapshots
|
||||
|
||||
Immediately after the upsert, `persist_trend_summary()` also inserts a snapshot row into the `trend_history` table. Unlike `trend_windows`, this table is append-only — every aggregation cycle adds a new row, creating a time-series of how the trend evolved over time. The history table stores the direction, strength, confidence, contradiction score, catalysts, risks, and timestamp. This time-series data powers the trend charts in the dashboard and enables the momentum computation in `services/aggregation/projection.py` by providing the previous cycle's strength and direction. If the history insert fails (for example, if the table does not yet exist in a development environment), the failure is logged at debug level and does not block the main upsert.
|
||||
|
||||
### `trend_evidence` — Per-Document Rankings
|
||||
|
||||
The `persist_trend_evidence()` function writes detailed evidence ranking rows to the `trend_evidence` table, linked to the `trend_windows` row by its UUID. Each row records a document ID, its role (supporting or opposing), and the individual scoring components: rank score, weight component, impact component, recency component, confidence component, and sentiment value. Non-UUID document IDs (such as synthetic pattern signal IDs like `pattern:AAPL:earnings:7d`) are filtered out before insertion, since the `trend_evidence` table enforces a foreign key to the `documents` table.
|
||||
|
||||
### `trend_projections` — Forward-Looking Estimates
|
||||
|
||||
The `persist_trend_projection()` function in `services/aggregation/projection.py` inserts the `TrendProjection` into the `trend_projections` table, linked to the `trend_windows` row. The row stores the projected direction, strength, confidence, projection horizon, driving factors (as JSON), macro contribution percentage, divergence flag, and computation timestamp. Like trend history, projections accumulate over time, allowing analysis of how well the system's forward-looking estimates matched subsequent reality.
|
||||
|
||||
---
|
||||
|
||||
## What Comes Next
|
||||
|
||||
At this point, the aggregation engine has transformed weighted signals into `TrendSummary` objects across five time windows, detected contradictions, ranked evidence, computed confidence, and persisted everything to PostgreSQL. The trend metrics — direction, strength, confidence, contradiction score — encode the accumulated weight of evidence for each ticker. But a `TrendSummary` is still an assessment, not an action. The next stage translates these assessments into concrete recommendations: should the system buy, sell, hold, or simply watch? And with what conviction? [Page 5 — Recommendation Generation](05-recommendation-generation.md) explains how the recommendation engine applies data quality suppression, eligibility evaluation, position sizing, thesis generation, and risk classification to convert trend summaries into actionable `Recommendation` objects that the trading engine can execute.
|
||||
@@ -0,0 +1,226 @@
|
||||
# Page 5 — Recommendation Generation and Signal-to-Action Translation
|
||||
|
||||
The aggregation engine described in [Page 4](04-trend-aggregation-and-accumulating-signals.md) produces `TrendSummary` objects across five time windows for each ticker, encoding the direction, strength, confidence, contradiction level, and supporting evidence accumulated from all three signal layers. But a `TrendSummary` is an assessment — it describes what the evidence says, not what the system should do about it. The recommendation engine is where assessment becomes action. It takes each `TrendSummary`, subjects it to a series of deterministic evaluations, and produces a `Recommendation` object that specifies a concrete action (buy, sell, hold, or watch), an execution mode (informational, paper-eligible, or live-eligible), a position sizing guideline, a human-readable thesis, and a risk classification. Every decision in this pipeline is rule-based and fully traceable — the LLM is only involved in an optional downstream step that rewrites the thesis wording.
|
||||
|
||||
The recommendation worker in `services/recommendation/main.py` polls the `stonks:queue:recommendation` Redis queue for jobs, each specifying a ticker and time window. For each job, it delegates to `generate_recommendation()` in `services/recommendation/worker.py`, which orchestrates the full pipeline: fetch the latest trend summary, check for duplicate recommendations, fetch any available trend projection, evaluate data quality suppression, evaluate eligibility, optionally rewrite the thesis via LLM, build the `Recommendation` object, and persist everything to PostgreSQL. For a visual overview of this flow, see the [Recommendation Generation Flow diagram](diagrams/recommendation-generation-flow.md).
|
||||
|
||||
---
|
||||
|
||||
## Data Quality Suppression
|
||||
|
||||
Before the eligibility engine evaluates whether a trend is strong enough to act on, the suppression layer in `services/recommendation/suppression.py` asks a more fundamental question: is the underlying data reliable enough to act on at all? A trend might show high confidence and strong directionality, but if the documents feeding it are stale, poorly extracted, or drawn from a single source type, the apparent signal quality is illusory. The suppression layer acts as a pre-filter on data quality, running before the eligibility engine and forcing any recommendation built on unreliable data to `informational` mode regardless of how strong the trend metrics look.
|
||||
|
||||
The `evaluate_suppression()` function accepts a `TrendSummary` and a `DataQualityContext` — a set of metrics about the documents underlying the trend, populated by querying `documents` and `document_intelligence` tables for the evidence document IDs stored in the trend summary. When full document-level metrics are not available (for example, in a development environment without the full document pipeline), the function falls back to `build_quality_context_from_summary()`, which estimates quality metrics from the trend summary's own evidence counts and confidence.
|
||||
|
||||
### The Six Data Quality Checks
|
||||
|
||||
The suppression evaluation runs six independent checks, each comparing a data quality metric against a configurable threshold defined in `SuppressionConfig`. If any single check fails, the recommendation is suppressed:
|
||||
|
||||
1. **Low extraction confidence** — If the average extraction confidence across the evidence documents falls below `0.40` (`min_avg_extraction_confidence`), the underlying LLM extractions are too unreliable. This catches cases where the extractor struggled with document formatting, ambiguous content, or low-quality source material, as described in [Page 2](02-ai-agent-processing-and-extraction.md).
|
||||
|
||||
2. **Evidence staleness** — If the most recent evidence document is older than `168` hours (7 days, `max_evidence_staleness_hours`), the trend is based on outdated information. Markets move fast, and a week-old evidence base may no longer reflect current conditions. When documents exist but no timestamp is available, the evidence is conservatively treated as stale.
|
||||
|
||||
3. **Low source diversity** — If fewer than `1` distinct source type (`min_source_types`) contributed to the evidence, the signal may be driven by a single unreliable source class. In practice, this check fires when the quality context has documents but all come from the same source type (for example, all news articles with no filings or market data to corroborate).
|
||||
|
||||
4. **High extraction failure rate** — If more than `50%` (`max_extraction_failure_rate`) of the documents that should have contributed to the trend failed extraction entirely, the data pipeline is unreliable for this ticker. A high failure rate means the trend summary is built from a biased subset of the available evidence — the failed documents might have told a different story.
|
||||
|
||||
5. **Insufficient valid documents** — If fewer than `2` valid (non-failed) documents (`min_valid_documents`) contributed to the trend, there simply is not enough data to act on. A single document, no matter how high-quality, does not provide the corroboration needed for automated trading decisions.
|
||||
|
||||
6. **Low data quality score** — The `_compute_data_quality_score()` function computes an overall quality score from three weighted components: extraction confidence (40% weight, normalized against a 0.8 baseline), evidence freshness (30% weight, linear decay over the staleness window), and document coverage (30% weight, combining the valid/total ratio with a count factor that saturates at 10 documents). If this composite score falls below `0.30` (`min_data_quality_score`) and the low-confidence check has not already fired, a general suppression reason is added.
|
||||
|
||||
When any check triggers, the `SuppressionResult` records the specific reasons (as `SuppressionReason` enum values) and the computed data quality score. The worker in `services/recommendation/worker.py` uses this result to force the recommendation's mode to `informational` and append a suppression note to the thesis text, ensuring the suppression decision is visible in the audit trail.
|
||||
|
||||
### Safety Suppressions: Macro-Only and Pattern-Only Signals
|
||||
|
||||
Beyond the six data quality checks, two additional safety suppressions protect against acting on signals that lack company-specific corroboration:
|
||||
|
||||
**Macro-only suppression** (`evaluate_macro_only_suppression()`) fires when macro signals are the sole basis for a trend direction — no company-specific signals contributed at all. As described in [Page 3](03-signal-scoring-and-weighted-signals.md), macro signals enter the aggregation engine at a reduced weight of `0.3` relative to company signals. But even at reduced weight, macro signals alone can shift a trend direction if no company-specific evidence exists. When this happens, the recommendation is forced to `informational` mode with a caveat noting that the signal is macro-only and should not be used for automated trading.
|
||||
|
||||
**Pattern-only suppression** (`evaluate_pattern_only_suppression()`) applies the same logic to competitive/pattern signals. When pattern-based signals from `services/aggregation/pattern_matcher.py` and `services/aggregation/signal_propagation.py` are the sole contributors — no company-specific or macro signals — the recommendation is suppressed. Historical patterns are valuable context, but acting on them without any current evidence is too speculative for automated trading.
|
||||
|
||||
Both safety suppressions are evaluated in the worker after the main suppression check, and both force the mode to `informational` when triggered.
|
||||
|
||||
---
|
||||
|
||||
## Eligibility Evaluation
|
||||
|
||||
Recommendations that survive the suppression layer enter the eligibility evaluation in `services/recommendation/eligibility.py`. This is the core decision logic — a set of deterministic rules that map trend metrics to actions, execution modes, and position sizing. The `evaluate_eligibility()` function is the single entry point, accepting a `TrendSummary` and an `EligibilityConfig` of tunable thresholds.
|
||||
|
||||
### Gate Checks
|
||||
|
||||
The `_check_gates()` function applies five hard gates. If any gate fails, the trend is ineligible for a recommendation (though the action and mode are still computed for the audit trace):
|
||||
|
||||
| Gate | Threshold | Rejection Reason |
|
||||
|------|-----------|-----------------|
|
||||
| Confidence | ≥ `0.35` | `low_confidence` |
|
||||
| Trend strength | ≥ `0.10` | `low_trend_strength` |
|
||||
| Contradiction score | ≤ `0.60` | `high_contradiction` |
|
||||
| Evidence count | ≥ `2` (supporting + opposing) | `insufficient_evidence` |
|
||||
| Direction | ≠ `neutral` | `neutral_direction` |
|
||||
|
||||
These gates are intentionally conservative. A confidence threshold of `0.35` means the system needs meaningful evidence breadth and agreement before generating any recommendation at all (see the confidence computation in [Page 4](04-trend-aggregation-and-accumulating-signals.md)). The contradiction ceiling of `0.60` allows moderately contested trends through — only when the evidence is deeply split does the gate reject. The evidence minimum of `2` ensures that no recommendation is ever based on a single document.
|
||||
|
||||
When a trend fails any gate, the resulting `EligibilityResult` has `eligible = False` and the mode is forced to `informational`, regardless of what the mode escalation logic would otherwise compute.
|
||||
|
||||
### Action Mapping
|
||||
|
||||
The `_determine_action()` function maps the trend's direction and strength to one of four action types. The logic evaluates in a specific order:
|
||||
|
||||
**Mixed or neutral direction → WATCH.** If the trend direction is `mixed` (high contradiction with weak directional signal) or `neutral`, the action is always `WATCH`. There is no directional conviction to act on.
|
||||
|
||||
**Strong directional signal → BUY or SELL.** If the trend strength reaches `0.25` or above (`action_strength_threshold`), the action follows the direction: `BUY` for bullish, `SELL` for bearish. This threshold ensures that only trends with meaningful magnitude trigger position-changing actions.
|
||||
|
||||
**Weak directional signal with decent confidence → HOLD.** If the trend has a clear direction (bullish or bearish) but strength remains below `0.25`, the action depends on confidence. If confidence reaches `0.50` or above (`hold_confidence_threshold`), the action is `HOLD` — the system recognizes the directional lean but does not have enough conviction to recommend a position change. Below `0.50` confidence, the action falls to `WATCH`.
|
||||
|
||||
This mapping creates the escalation ladder described in [Page 4](04-trend-aggregation-and-accumulating-signals.md): as consecutive signals accumulate and strengthen the trend metrics, the action naturally progresses from WATCH → HOLD → BUY/SELL.
|
||||
|
||||
### Mode Escalation
|
||||
|
||||
The `_determine_mode()` function determines the highest execution mode allowed for the recommendation. Mode controls whether the recommendation is purely informational, eligible for paper trading, or eligible for live trading:
|
||||
|
||||
**WATCH and HOLD → always informational.** These actions do not trigger trades, so they are always `informational` mode. They are logged for human review and dashboard display but never enter the trading engine.
|
||||
|
||||
**BUY and SELL → escalation based on signal quality.** For actionable recommendations, mode escalates through three tiers:
|
||||
|
||||
- **`informational`** — The default when confidence is below `0.50`. The recommendation is recorded but not eligible for any trading.
|
||||
- **`paper_eligible`** — When confidence reaches `0.50` or above (`paper_confidence_threshold`). The recommendation can be picked up by the paper trading engine described in [Page 6](06-trading-decisions-and-execution.md).
|
||||
- **`live_eligible`** — The strictest tier, requiring confidence ≥ `0.70` (`live_confidence_threshold`), contradiction ≤ `0.25` (`live_max_contradiction`), and evidence count ≥ `5` (`live_min_evidence`). This triple gate ensures that only high-conviction, well-corroborated, low-contradiction recommendations can trigger live trades.
|
||||
|
||||
The evidence count for mode escalation is computed as the sum of supporting and opposing evidence documents, matching the same count used in the gate checks.
|
||||
|
||||
---
|
||||
|
||||
## Position Sizing
|
||||
|
||||
The `_compute_position_sizing()` function in `services/recommendation/eligibility.py` translates signal quality into a portfolio allocation guideline. Position sizing is not a fixed value — it scales dynamically with the confidence and strength of the underlying trend, penalized by contradiction and thin evidence.
|
||||
|
||||
### Base and Scaling
|
||||
|
||||
The computation starts with a base portfolio allocation of `1%` (`base_portfolio_pct = 0.01`) and scales upward based on two factors:
|
||||
|
||||
- **Confidence factor** — `0.8 × confidence` (`confidence_sizing_weight`), reflecting how much the system trusts the trend assessment.
|
||||
- **Strength factor** — `0.5 + 0.5 × trend_strength`, ranging from `0.5` (weakest trend) to `1.0` (strongest trend).
|
||||
|
||||
The raw portfolio percentage is computed as:
|
||||
|
||||
```
|
||||
raw_portfolio = base + confidence_factor × strength_factor × (max - base)
|
||||
```
|
||||
|
||||
where `max` is `10%` (`max_portfolio_pct = 0.10`). At maximum confidence (1.0) and maximum strength (1.0), the raw allocation reaches the full 10%. At typical values (confidence 0.6, strength 0.3), the raw allocation is considerably lower.
|
||||
|
||||
### Contradiction Penalty
|
||||
|
||||
The contradiction score applies a multiplicative penalty:
|
||||
|
||||
```
|
||||
portfolio_pct = raw_portfolio × (1.0 − 0.5 × contradiction_score)
|
||||
```
|
||||
|
||||
A contradiction score of `0.40` reduces the allocation by 20%. A score of `0.0` (no contradiction) applies no penalty. This ensures that contested trends receive smaller position sizes even when they pass the eligibility gates.
|
||||
|
||||
### Evidence Count Penalty
|
||||
|
||||
Thin evidence further reduces the allocation:
|
||||
|
||||
- Fewer than `3` evidence documents → multiply by `0.5` (halved).
|
||||
- Fewer than `5` evidence documents → multiply by `0.75`.
|
||||
- `5` or more documents → no penalty.
|
||||
|
||||
This penalty stacks with the contradiction penalty, so a trend with high contradiction and thin evidence receives a substantially reduced position size.
|
||||
|
||||
### Max Loss Scaling
|
||||
|
||||
The same scaling logic applies to the maximum loss percentage, which starts at a base of `0.3%` (`base_max_loss_pct = 0.003`) and scales up to `2%` (`max_max_loss_pct = 0.02`). Higher-conviction positions are allowed larger loss tolerances, while low-conviction or contested positions are constrained to tighter stops.
|
||||
|
||||
The final `PositionSizing` object (defined in `services/shared/schemas.py`) contains `portfolio_pct` and `max_loss_pct`, both clamped to their respective bounds. This object is embedded in the `Recommendation` and later consumed by the trading engine's own position sizer (described in [Page 6](06-trading-decisions-and-execution.md)), which applies additional portfolio-level constraints.
|
||||
|
||||
---
|
||||
|
||||
## Thesis Generation
|
||||
|
||||
Every recommendation includes a human-readable thesis that explains the reasoning behind the action. Thesis generation happens in two layers: a deterministic assembly that is always present, and an optional LLM rewrite that polishes the wording for trading-eligible recommendations.
|
||||
|
||||
### Deterministic Thesis Assembly
|
||||
|
||||
The `build_thesis()` function in `services/recommendation/worker.py` constructs a thesis string entirely from the trend data and eligibility result, with no model involvement. The thesis is assembled from several components in order:
|
||||
|
||||
1. **Opening** — States the ticker, trend direction, window, strength, and confidence. For example: "AAPL shows a bearish trend over the 7d window with strength 0.35 and confidence 0.62."
|
||||
|
||||
2. **Catalysts** — Lists the top three dominant catalysts from the `TrendSummary`, drawn from the evidence ranking described in [Page 4](04-trend-aggregation-and-accumulating-signals.md).
|
||||
|
||||
3. **Contradiction note** — If the contradiction score exceeds `0.15`, a note flags the signal disagreement and its magnitude.
|
||||
|
||||
4. **Trend projection** — When a `TrendProjection` is available and not flagged as low-confidence, the thesis incorporates the projected direction, strength, and top driving factors. If the projection diverges from the current trend, a divergence note is appended.
|
||||
|
||||
5. **Risks** — Lists the top two material risks from the `TrendSummary`.
|
||||
|
||||
6. **Evidence count** — States the number of supporting and opposing evidence documents.
|
||||
|
||||
7. **Prescriptive action** — States the recommended action and mode (e.g., "Recommendation: SELL (paper eligible).").
|
||||
|
||||
The deterministic thesis is always generated and serves as the audit reference. Even when the LLM rewrites the thesis, the deterministic version is preserved in the model metadata for traceability.
|
||||
|
||||
### Optional LLM Rewrite via the Thesis-Rewriter Agent
|
||||
|
||||
For recommendations that are both eligible and not suppressed, the worker optionally invokes the thesis-rewriter agent to polish the deterministic thesis into analyst-quality prose. The LLM rewrite is implemented in `services/recommendation/thesis_llm.py` and uses the `thesis-rewriter` agent slug, resolved at runtime through the `AgentConfigResolver` in `services/shared/agent_config.py`.
|
||||
|
||||
The `AgentConfigResolver` queries the `ai_agents` and `agent_variants` database tables to resolve the active configuration for the `thesis-rewriter` slug, preferring an active variant's model, timeout, and retry settings when one exists. The resolver uses a 60-second TTL in-memory cache to avoid hitting the database on every recommendation. This is the same resolution mechanism used by the document extractor and event classifier agents described in [Page 2](02-ai-agent-processing-and-extraction.md).
|
||||
|
||||
The `rewrite_thesis_with_llm()` function builds a prompt from the deterministic thesis and trend context (ticker, window, direction, strength, confidence, contradiction score, catalysts, risks), sends it to the local Ollama instance via HTTP, and returns the rewritten text. The system prompt enforces strict rules: no fabricated information, no numbers or facts not present in the input, under 150 words, neutral professional tone, and only the rewritten thesis text in the response.
|
||||
|
||||
The LLM layer is purely additive — if the call fails for any reason (network error, timeout, empty response, token budget exceeded), the original deterministic thesis is returned unchanged. The worker in `services/recommendation/main.py` resolves the thesis-rewriter configuration at startup and refreshes it every 50 jobs to pick up configuration changes without requiring a restart. When no database configuration exists for the `thesis-rewriter` slug, thesis rewriting is silently disabled.
|
||||
|
||||
Performance logging for the thesis-rewriter is written to the `agent_performance_log` table, recording success/failure, duration, estimated token counts, and the variant ID. Token budget enforcement checks hourly usage against the variant's configured budget before making the LLM call, preventing runaway costs from high-volume recommendation cycles.
|
||||
|
||||
### Risk Classification Prefix
|
||||
|
||||
Before the thesis is stored, the `classify_risk()` function in `services/recommendation/worker.py` assigns a risk classification label that is prepended to the thesis text as a `[risk:<level>]` prefix. The classification is computed from a composite score:
|
||||
|
||||
| Factor | Contribution |
|
||||
|--------|-------------|
|
||||
| Contradiction score | `contradiction × 2.0` |
|
||||
| Low confidence | `(1.0 − confidence) × 1.5` |
|
||||
| Low evidence count | `+1.0` if < 3 docs, `+0.5` if < 5 docs |
|
||||
| Rejection reasons | `+0.5` per rejection reason |
|
||||
|
||||
The composite score maps to four levels:
|
||||
|
||||
| Score Range | Classification |
|
||||
|-------------|---------------|
|
||||
| ≥ 3.0 | `very_high` |
|
||||
| ≥ 2.0 | `high` |
|
||||
| ≥ 1.0 | `moderate` |
|
||||
| < 1.0 | `low` |
|
||||
|
||||
A recommendation with high contradiction (0.4 → contributes 0.8), moderate confidence (0.55 → contributes 0.675), and 4 evidence documents (contributes 0.5) would score 1.975, classifying as `moderate`. The same recommendation with only 2 evidence documents would score 2.475, pushing it to `high`. This classification gives downstream consumers — both the trading engine and human reviewers — a quick risk signal without needing to re-evaluate the underlying metrics.
|
||||
|
||||
---
|
||||
|
||||
## Persistence
|
||||
|
||||
The recommendation pipeline persists its output to three PostgreSQL tables, creating a complete audit trail from trend assessment through decision logic to the final recommendation.
|
||||
|
||||
### `recommendations` — The Core Record
|
||||
|
||||
The `persist_recommendation()` function in `services/recommendation/worker.py` inserts the `Recommendation` into the `recommendations` table. Each row captures the ticker, action, mode, confidence, time horizon, thesis (including the risk classification prefix and any suppression notes), invalidation conditions (as JSONB), position sizing (portfolio percentage and max loss percentage), model metadata (provider, model name, prompt version, schema version), risk classification, and generation timestamp. The insert returns the recommendation's UUID, which serves as the foreign key for the evidence and risk evaluation tables.
|
||||
|
||||
### `recommendation_evidence` — Evidence Citations
|
||||
|
||||
For each evidence document referenced in the recommendation, a row is inserted into the `recommendation_evidence` table linking the recommendation UUID to the document UUID, with an evidence type (`supporting` or `opposing`) and a position-based weight that decays with rank: `weight = 1.0 / (1.0 + index × 0.1)`. The first supporting document gets weight `1.0`, the second gets `0.91`, the third `0.83`, and so on. Non-UUID document IDs (such as synthetic pattern signal IDs like `pattern:AAPL:earnings:7d` from the competitive signal layer) are filtered out before insertion, since the table enforces a foreign key to the `documents` table.
|
||||
|
||||
### `risk_evaluations` — Decision Audit Trail
|
||||
|
||||
The `risk_evaluations` table records the full eligibility decision for each recommendation: whether the trend was eligible, the allowed mode, the list of rejection reasons (as JSONB), and a `risk_checks` JSONB object containing the time horizon, position sizing details, invalidation conditions, and risk classification. This table enables post-hoc analysis of why the system made a particular decision — auditors can trace from the recommendation back through the eligibility evaluation to the underlying trend metrics.
|
||||
|
||||
---
|
||||
|
||||
## Deduplication
|
||||
|
||||
Before running the full evaluation pipeline, the worker checks whether the latest recommendation for the same ticker and time horizon is effectively identical to what would be generated. The `_is_duplicate_recommendation()` function in `services/recommendation/worker.py` compares the previous recommendation's action, mode, and confidence (within a `0.01` tolerance) against the current eligibility result. If all three match, the recommendation is skipped — the underlying trend data has not changed meaningfully since the last cycle. This prevents the system from flooding the `recommendations` table with identical entries on every aggregation cycle, while still generating a new recommendation whenever the trend metrics shift enough to change the action, mode, or confidence.
|
||||
|
||||
---
|
||||
|
||||
## What Comes Next
|
||||
|
||||
At this point, the recommendation engine has translated trend assessments into concrete `Recommendation` objects — each with an action, execution mode, position sizing guideline, thesis, and risk classification — and persisted them alongside their evidence citations and eligibility audit trails. Recommendations marked as `paper_eligible` or `live_eligible` are now available for the trading engine to consume. [Page 6 — Trading Decisions and Execution](06-trading-decisions-and-execution.md) explains how the trading engine polls these recommendations, applies its own pre-trade check sequence (circuit breakers, trading windows, confidence gates, deduplication, declining positions, and max open positions), computes final position sizes with portfolio-level constraints, and submits orders through the broker adapter to Alpaca's paper trading API.
|
||||
@@ -0,0 +1,199 @@
|
||||
# Page 6 — Trading Decisions and Execution
|
||||
|
||||
The recommendation engine described in [Page 5](05-recommendation-generation.md) produces `Recommendation` objects with an action, execution mode, position sizing guideline, thesis, and risk classification. Recommendations marked as `paper_eligible` or `live_eligible` are persisted to the `recommendations` table and are now available for the final stage of the pipeline: autonomous trade execution. The trading engine in `services/trading/engine.py` is where intelligence becomes action. It polls eligible recommendations, subjects each one to a strict sequence of pre-trade safety checks, computes a portfolio-aware position size, and — if every gate passes — submits an order through the broker adapter to Alpaca's paper trading API. Every evaluation, whether it results in a trade or a skip, is recorded as a `TradingDecision` in the `trading_decisions` table, creating a complete audit trail from the original document signal through to the broker response.
|
||||
|
||||
For a visual overview of the decision flow, see the [Trading Engine Decision Loop diagram](diagrams/trading-engine-decision-loop.md).
|
||||
|
||||
---
|
||||
|
||||
## The Trading Engine Decision Loop
|
||||
|
||||
The `TradingEngine` class in `services/trading/engine.py` is the orchestrator. When `start()` is called, it loads the current portfolio state from PostgreSQL — open positions, reserve pool balance, sector exposure, portfolio heat — and then spawns five concurrent `asyncio` tasks that run for the lifetime of the engine:
|
||||
|
||||
1. **`_decision_loop()`** — The core polling loop. Every 60 seconds (configurable via `polling_interval_seconds`), it queries the `recommendations` table for rows where `action IN ('buy', 'sell')`, `mode IN ('paper_eligible', 'live_eligible')`, and `generated_at` is within the last two hours. Recommendations are ordered by confidence descending and capped at 50 per cycle. For each recommendation, the engine fetches the current market price (first from `market_snapshots`, falling back to the Polygon API), then runs the full pre-trade evaluation pipeline described below.
|
||||
|
||||
2. **`_stop_loss_monitor()`** — Periodically checks current prices against the stop-loss and take-profit levels maintained by the `StopLossManager` in `services/trading/stop_loss_manager.py`. When a price crosses a stop-loss or take-profit threshold, the monitor submits a sell order to the broker queue. The `StopLossManager` computes initial levels from ATR and risk tier parameters, re-evaluates them when volatility shifts materially (ATR change > 10%), activates trailing stops when the price moves more than 50% toward the take-profit target, and tightens stops proactively when portfolio heat exceeds 80% of the maximum.
|
||||
|
||||
3. **`_performance_loop()`** — Computes portfolio-wide performance metrics (total value, unrealized and realized P&L, win rate, Sharpe ratio, drawdown, portfolio heat), persists daily snapshots to `portfolio_snapshots`, checks for daily-loss circuit breaker triggers, evaluates profit-taking opportunities, and synchronizes positions with the database to detect closed positions and trigger reserve pool siphoning.
|
||||
|
||||
4. **`_risk_tier_scheduler()`** — Runs once daily at 16:00 ET (market close). It loads the latest `PerformanceMetrics` from `portfolio_snapshots`, computes the reserve pool as a fraction of total portfolio value, and delegates to the `RiskTierController` in `services/trading/risk_tier_controller.py` to determine whether the active risk tier should change. Tier changes are persisted to `risk_tier_history` and take effect immediately for subsequent decision cycles.
|
||||
|
||||
5. **`_rebalance_scheduler()`** — Runs weekly on Monday at 09:45 ET (shortly after market open). It loads current positions, evaluates them against the active risk tier's constraints using the `PortfolioRebalancer`, and pushes any rebalance sell orders to `stonks:queue:broker_orders`. The rebalancer respects the circuit breaker — if any breaker is active, the rebalance cycle is skipped entirely.
|
||||
|
||||
All five tasks run concurrently within a single `asyncio` event loop. Graceful shutdown via `stop()` cancels all tasks and awaits their completion. If any task encounters an unexpected exception, it logs the error and retries after a brief sleep rather than crashing the engine.
|
||||
|
||||
---
|
||||
|
||||
## Pre-Trade Check Sequence
|
||||
|
||||
When the decision loop picks up a buy recommendation, it calls `evaluate_recommendation()` — a synchronous method that runs the full pre-trade check sequence. The checks are applied in a strict order, and the first failure short-circuits the evaluation with a `skip` decision. This fail-fast design ensures that expensive downstream computations (like position sizing and correlation analysis) are never reached when a simple gate would have rejected the trade.
|
||||
|
||||
The six checks, in order:
|
||||
|
||||
**a. Circuit breaker check.** The engine calls `self.circuit_breaker.is_active()` on the current `CircuitBreakerState`. If any circuit breaker is active and its cooldown has not expired, the recommendation is skipped with reason `circuit_breaker_active`. The circuit breaker mechanism is described in detail below.
|
||||
|
||||
**b. Trading window check.** The `is_within_trading_window()` function verifies that the current time falls within US market hours. Outside the trading window, no orders are submitted — the recommendation is skipped with reason `outside_trading_window`.
|
||||
|
||||
**c. Confidence gate.** The recommendation's confidence score is compared against the active risk tier's `min_confidence` threshold. A conservative tier requires confidence ≥ 0.75, moderate requires ≥ 0.55, and aggressive requires ≥ 0.40. If the recommendation's confidence falls below the tier minimum, it is skipped with reason `insufficient_confidence`. This gate ensures that the risk tier's conservatism is enforced before any capital allocation is considered.
|
||||
|
||||
**d. Deduplication check.** The engine maintains an in-memory set of processed recommendation IDs (`processed_recommendation_ids`) and also checks Redis via `stonks:dedupe:trading:*` keys (with a 24-hour TTL). If the recommendation has already been evaluated in this engine session or by a previous instance, it is skipped with reason `duplicate_recommendation`. This prevents the same recommendation from generating multiple orders across polling cycles.
|
||||
|
||||
**e. Declining positions check.** The `check_declining_positions()` method examines all open positions. If more than 50% of positions have unrealized losses exceeding 2% of their entry value, the engine halts new entries with reason `multiple_declining_positions`. This is a portfolio-level safety valve — when the majority of existing positions are underwater, adding new exposure compounds the risk.
|
||||
|
||||
**f. Max open positions check.** The engine enforces a configurable maximum number of concurrent positions (default 10). If the portfolio is already at capacity, the recommendation is skipped with reason `max_positions_reached`.
|
||||
|
||||
For sell recommendations, the engine follows a separate, simpler path: it verifies the trading window, looks up the existing position for the ticker, and submits a market sell order for the full position quantity without running the position sizer. Sell decisions still generate a `TradingDecision` audit record and set the Redis deduplication key.
|
||||
|
||||
If all six checks pass for a buy recommendation, the engine proceeds to position sizing.
|
||||
|
||||
---
|
||||
|
||||
## Position Sizing
|
||||
|
||||
The `PositionSizer` in `services/trading/position_sizer.py` translates a recommendation's signal quality into a concrete dollar amount and share count, applying a sequential pipeline of adjustments that account for confidence, portfolio composition, sector concentration, correlation, and upcoming earnings events. The sizer operates on the *active pool* — the portion of the portfolio available for trading after subtracting the reserve pool balance.
|
||||
|
||||
### Base Sizing
|
||||
|
||||
The computation begins with a base allocation percentage derived from the risk tier:
|
||||
|
||||
```
|
||||
base_allocation_pct = risk_tier.max_position_pct × 0.5
|
||||
raw_pct = base_allocation_pct × (confidence / min_confidence)
|
||||
```
|
||||
|
||||
The base starts at half the tier's maximum position percentage, then scales linearly with how far the recommendation's confidence exceeds the tier minimum. A moderate-tier recommendation with confidence 0.70 against a minimum of 0.55 would produce a raw percentage of `0.05 × (0.70 / 0.55) ≈ 0.0636`, or about 6.4% of the active pool. The raw percentage is clamped to `max_position_pct` (5% for conservative, 10% for moderate, 15% for aggressive) and then converted to a dollar amount against the active pool. An absolute position cap (default $50) provides a hard ceiling regardless of pool size — a safety measure for the paper trading environment.
|
||||
|
||||
### Correlation-Aware Diversification
|
||||
|
||||
The sizer computes a weighted average correlation between the candidate ticker and all existing positions, using the pairwise correlation matrix that the engine refreshes from 30 days of daily close prices in `market_snapshots`. Each existing position's correlation is weighted by its market value, so larger positions have more influence on the diversification check.
|
||||
|
||||
If the weighted average correlation exceeds 0.8, the position is rejected outright — the portfolio already has too much exposure to correlated assets. Between 0.5 and 0.8, the dollar amount is reduced proportionally: a correlation of 0.65 produces a scale factor of `1.0 − (0.65 − 0.5) / (0.8 − 0.5) = 0.5`, halving the position size. Below 0.5, no reduction is applied.
|
||||
|
||||
### Sector Exposure Reduction
|
||||
|
||||
The sizer checks whether adding the new position would push the sector's total exposure beyond the risk tier's `max_sector_pct` (20% for conservative, 30% for moderate, 40% for aggressive). If the sector is already at its limit, the position is rejected. If the new position would exceed the limit, the dollar amount is reduced to exactly fill the remaining sector capacity.
|
||||
|
||||
### Diversification Bonus
|
||||
|
||||
When the portfolio holds fewer than three distinct sectors and the candidate ticker belongs to a new sector, the sizer applies a 1.2× bonus to the dollar amount. This incentivizes early diversification — the first few positions are encouraged to spread across sectors rather than concentrating in a single one. The bonus is re-clamped to `max_position_pct` after application to prevent oversized positions.
|
||||
|
||||
### Earnings Proximity Adjustment
|
||||
|
||||
The sizer checks the earnings calendar for the candidate ticker. If earnings are within one trading day, the position is rejected entirely — the binary risk of an earnings surprise is too high for automated entry. If earnings are within three trading days, the dollar amount is reduced by 50%. Beyond three days, no adjustment is applied.
|
||||
|
||||
### Portfolio Heat Check and Share Rounding
|
||||
|
||||
After all adjustments, the sizer estimates the new position's contribution to portfolio heat (the aggregate risk from stop-loss distances across all positions). If adding the position would push total heat beyond `max_portfolio_heat × active_pool` (10% for conservative, 20% for moderate, 30% for aggressive), the position is rejected.
|
||||
|
||||
Finally, the dollar amount is converted to whole shares via `floor(dollar_amount / current_price)`. If rounding produces zero shares (the position is too small for even one share at the current price), the position is rejected. The final dollar amount is recalculated from the whole-share quantity to reflect the actual capital deployed.
|
||||
|
||||
The `PositionSizeResult` returned to the engine includes the dollar amount, share quantity, allocation percentage, a list of human-readable adjustment notes, and a rejected flag with reason if any step failed. These adjustment notes are embedded in the `TradingDecision`'s `decision_trace` for full auditability.
|
||||
|
||||
---
|
||||
|
||||
## Circuit Breaker
|
||||
|
||||
The `CircuitBreaker` in `services/trading/circuit_breaker.py` is a pure computation module that evaluates three independent trigger conditions. It carries no state of its own — the engine manages the `CircuitBreakerState` dataclass and persists trigger events to the `circuit_breaker_events` table and Redis keys under `stonks:trading:circuit_breaker:*`.
|
||||
|
||||
### Three Trigger Types
|
||||
|
||||
**Daily loss trigger.** When the portfolio's daily P&L loss exceeds 5% of total portfolio value (`daily_loss_pct = 0.05`), the circuit breaker activates. The `check_daily_loss()` method compares the absolute loss ratio against the threshold. The cooldown duration is set to `volatility_pause_hours` (default 2 hours). The performance loop in the engine calls `_check_circuit_breaker_daily_loss()` periodically to evaluate this condition against the latest portfolio metrics. In extreme cases where the drawdown exceeds an emergency threshold, the reserve pool's emergency liquidation mechanism may also be triggered.
|
||||
|
||||
**Single position loss trigger.** When any individual position loses more than 15% of its entry value (`single_position_loss_pct = 0.15`), the circuit breaker activates with a ticker-specific cooldown. The `check_single_position()` method evaluates the loss percentage. The cooldown for the affected ticker is set to `ticker_cooldown_hours` (default 48 hours), during which the engine will not re-enter that ticker. The `is_ticker_cooled_down()` method checks whether a specific ticker is still within its cooldown window by consulting the `ticker_cooldowns` dictionary in the `CircuitBreakerState`.
|
||||
|
||||
**Volatility trigger (stop-loss clustering).** When three or more stop-losses fire within a 30-minute rolling window (`stop_loss_hits_threshold = 3`, `stop_loss_window_minutes = 30`), the circuit breaker activates. The `check_volatility()` method uses a sliding window algorithm: it sorts the stop-loss timestamps and checks every contiguous subsequence of length `stop_loss_hits_threshold` to see if it fits within the window. This detects rapid-fire stop-loss cascades that indicate extreme market volatility. The cooldown is `volatility_pause_hours` (default 2 hours).
|
||||
|
||||
### Cooldown Computation
|
||||
|
||||
The `compute_cooldown_expiry()` method calculates when a triggered breaker expires. For `daily_loss` and `volatility` triggers, the expiry is `triggered_at + volatility_pause_hours`. For `single_position` triggers, the expiry is `triggered_at + ticker_cooldown_hours`, giving the affected ticker a longer cooling-off period. The `is_active()` method returns `True` when the breaker is flagged active and the current time has not yet passed the cooldown expiry.
|
||||
|
||||
### Redis State Tracking
|
||||
|
||||
The engine persists circuit breaker state to Redis under the `stonks:trading:circuit_breaker:*` key pattern (constructed by `trading_cb_key()` in `services/shared/redis_keys.py`). Each trigger type gets its own key — for example, `stonks:trading:circuit_breaker:daily_loss` — storing the activation timestamp and cooldown expiry. This allows the state to survive engine restarts and enables external monitoring tools to query breaker status without accessing the engine's memory.
|
||||
|
||||
---
|
||||
|
||||
## Reserve Pool
|
||||
|
||||
The `ReservePoolController` in `services/trading/reserve_pool.py` manages an untouchable cash reserve that grows from realized trading profits. The reserve serves two purposes: it provides a buffer against drawdowns, and its size relative to the portfolio influences risk tier upgrade decisions.
|
||||
|
||||
### Profit Siphoning
|
||||
|
||||
When the engine detects a closed position with positive unrealized P&L (via `_sync_positions_and_siphon()` in the performance loop), it calls `siphon_profit()` on the controller. The method transfers a configurable fraction of the realized profit into the reserve — by default 20% (`siphon_pct = 0.20`). Only positive profits are siphoned; losses do not reduce the reserve balance. Each siphon event is recorded in the `reserve_pool_ledger` table with the transfer amount, resulting balance, trigger type (`profit_siphon`), the ticker as reference, and a timestamp.
|
||||
|
||||
### High-Water Mark Rebalancing
|
||||
|
||||
The `is_high_water()` method returns `True` when the reserve balance exceeds 30% of total portfolio value (`high_water_pct = 0.30`). This signal is consumed by the risk tier scheduler — when the reserve is healthy and other performance criteria are met, the controller may recommend upgrading to a more aggressive tier. The high-water mark acts as a confidence indicator: a large reserve means the system has been consistently profitable and can afford to take on more risk.
|
||||
|
||||
### Emergency Liquidation
|
||||
|
||||
The `should_emergency_liquidate()` method checks whether the current drawdown exceeds an emergency threshold. When triggered, `emergency_liquidate()` returns the full reserve balance for release back into the active pool. The caller (the engine) is responsible for zeroing the persisted balance and recording the ledger entry. Emergency liquidation is a last resort — it sacrifices the safety buffer to prevent the portfolio from hitting a catastrophic loss level.
|
||||
|
||||
### Active Pool Computation
|
||||
|
||||
The `compute_active_pool()` method calculates the capital available for trading: `active_pool = total_portfolio_value − reserve_balance`. All position sizing computations use the active pool rather than the total portfolio value, ensuring that the reserve is never inadvertently deployed into new positions.
|
||||
|
||||
---
|
||||
|
||||
## Risk Tier Auto-Adjustment
|
||||
|
||||
The `RiskTierController` in `services/trading/risk_tier_controller.py` evaluates portfolio performance and determines whether the active risk tier should shift. The system supports three tiers — conservative, moderate, and aggressive — each defined by a `RiskTierConfig` dataclass in `services/trading/models.py` with distinct parameter values:
|
||||
|
||||
| Parameter | Conservative | Moderate | Aggressive |
|
||||
|-----------|-------------|----------|------------|
|
||||
| `min_confidence` | 0.75 | 0.55 | 0.40 |
|
||||
| `max_position_pct` | 5% | 10% | 15% |
|
||||
| `stop_loss_atr_multiplier` | 1.5× | 2.0× | 2.5× |
|
||||
| `reward_risk_ratio` | 2.0 | 1.5 | 1.2 |
|
||||
| `max_sector_pct` | 20% | 30% | 40% |
|
||||
| `max_portfolio_heat` | 10% | 20% | 30% |
|
||||
|
||||
The tier controller's `evaluate()` method checks two conditions:
|
||||
|
||||
**Downgrade (any one triggers).** If the trailing 30-day win rate drops below 40% or the current drawdown exceeds 15%, the tier steps down by one level (e.g., aggressive → moderate). If the system is already at conservative, no further downgrade is possible.
|
||||
|
||||
**Upgrade (all must be true).** If the win rate exceeds 55%, the reserve pool exceeds 20% of total portfolio value, and the current drawdown is below 5%, the tier steps up by one level. The triple requirement ensures that upgrades only happen when the system is performing well, has built a safety cushion, and is not in a drawdown.
|
||||
|
||||
The risk tier scheduler in the engine evaluates these conditions daily at market close. When a tier change occurs, it is persisted to the `risk_tier_history` table with the previous tier, new tier, trigger source (`auto_adjustment`), and the metrics that drove the decision (win rate, drawdown, reserve percentage, Sharpe ratio). The new tier takes effect immediately — the engine updates its `_active_risk_tier` reference, and all subsequent decision cycles use the new tier's parameters for confidence gates, position sizing, stop-loss computation, and sector exposure limits.
|
||||
|
||||
---
|
||||
|
||||
## Order Submission Flow
|
||||
|
||||
When `evaluate_recommendation()` returns an `act` decision, the engine constructs an order job and pushes it through a multi-stage submission pipeline that spans two services.
|
||||
|
||||
### TradingDecision Persistence
|
||||
|
||||
Every evaluation — whether it results in `act` or `skip` — produces a `TradingDecision` dataclass that is persisted to the `trading_decisions` table via `_persist_decision()`. The record captures the recommendation ID, decision outcome, skip reason (if applicable), ticker, computed position size and share quantity, the risk tier at the time of decision, portfolio heat, active pool and reserve pool balances, circuit breaker status, correlation and sector exposure check results, earnings proximity flag, and a `decision_trace` JSONB field containing the full reasoning chain. This creates a complete audit record of every recommendation the engine evaluated and why it acted or declined.
|
||||
|
||||
### Order Enqueue
|
||||
|
||||
For `act` decisions, the engine builds an order job dictionary containing the trading decision ID, ticker, action (buy or sell), quantity, and order type (market). This job is pushed via `rpush` to the `stonks:queue:broker_orders` Redis queue (constructed by `queue_key(QUEUE_BROKER)` from `services/shared/redis_keys.py`). The engine immediately deducts the estimated order cost from the in-memory active pool to prevent over-allocation across concurrent recommendation evaluations within the same polling cycle.
|
||||
|
||||
### Broker Service Processing
|
||||
|
||||
The broker service in `services/adapters/broker_service.py` runs as a standalone worker that polls `stonks:queue:broker_orders` via `blpop`. For each order job, `process_order_job()` executes a multi-step pipeline:
|
||||
|
||||
1. **Idempotency check.** A deterministic idempotency key is generated from the job's ticker, action, quantity, and trading decision ID. The service checks Redis first (fast path) and then the `orders` table (durable fallback) to prevent duplicate submissions. If a matching key exists, the job is silently dropped.
|
||||
|
||||
2. **Risk evaluation.** The service loads the current `PortfolioRiskConfig` from the database and the account's risk state (open positions, daily P&L, sector exposure) from both the database and the Alpaca API. The `evaluate_order()` function runs the proposed order through a set of risk checks — position limits, sector concentration, daily loss thresholds — and produces an evaluation result. The evaluation is persisted to the `risk_evaluations` table regardless of outcome.
|
||||
|
||||
3. **Alpaca submission.** If the risk evaluation passes, the service calls `submit_order()` on the `AlpacaBrokerAdapter` in `services/adapters/broker_adapter.py`. The adapter constructs the Alpaca REST API payload (symbol, quantity, side, order type, time in force) and submits it to `paper-api.alpaca.markets/v2/orders` with an idempotency key header. The adapter follows a fail-closed policy: any network error or ambiguous response returns a rejected `OrderResponse` rather than risking duplicate orders.
|
||||
|
||||
4. **Persistence and audit trail.** The `persist_order()` function writes the order to the `orders` table with the full request and response details, risk evaluation results, and the recommendation ID for traceability. When the order is filled, the fill details (price, quantity) are recorded. Order events are published to the analytical lakehouse via MinIO for downstream analysis. The Redis idempotency marker is set after successful persistence to prevent reprocessing.
|
||||
|
||||
The result is a complete chain of custody: from the original document that produced a signal (Pages [1](01-data-ingestion-and-preparation.md)–[2](02-ai-agent-processing-and-extraction.md)), through signal scoring ([Page 3](03-signal-scoring-and-weighted-signals.md)) and trend aggregation ([Page 4](04-trend-aggregation-and-accumulating-signals.md)), to the recommendation ([Page 5](05-recommendation-generation.md)), the trading decision, the risk evaluation, and the broker response — every step is persisted and linked by foreign keys. The `trading_decisions` table links to `recommendations` via `recommendation_id`, the `orders` table links back to both, and the `positions` and `portfolio_snapshots` tables capture the portfolio impact over time.
|
||||
|
||||
For additional reference on the trading engine's configuration, queue topology, and database tables, see [docs/services.md](../services.md).
|
||||
|
||||
---
|
||||
|
||||
## Conclusion: From Raw Data to Trade Execution
|
||||
|
||||
This six-page series has traced the full intelligence-to-decision pipeline in Stonks Oracle, from the moment raw data enters the system to the moment an order reaches the broker.
|
||||
|
||||
It began with [Page 1](01-data-ingestion-and-preparation.md), where the scheduler orchestrates ingestion cycles across four data sources — Polygon news, SEC EDGAR filings, Polygon market data, and macro news APIs — and the parser normalizes raw content into structured documents ready for AI processing. [Page 2](02-ai-agent-processing-and-extraction.md) described how the Document Intelligence Extractor and Global Event Classifier agents use LLM inference to produce structured JSON intelligence, with hot-swappable model configurations and a robust JSON repair pipeline. [Page 3](03-signal-scoring-and-weighted-signals.md) explained how raw extraction output is transformed into `WeightedSignal` objects through a composite formula that balances recency, credibility, novelty, and market context across three independent signal layers. [Page 4](04-trend-aggregation-and-accumulating-signals.md) showed how the aggregation engine merges these signals across five time windows, detecting contradictions, ranking evidence, and computing trend projections — with consecutive same-direction signals accumulating to escalate the system's response from neutral through watch and hold to buy or sell. [Page 5](05-recommendation-generation.md) covered the translation of trend assessments into actionable recommendations through data quality suppression, eligibility evaluation, position sizing, thesis generation, and risk classification.
|
||||
|
||||
And here in Page 6, the pipeline reached its terminus: the trading engine's decision loop polling those recommendations, subjecting each to circuit breaker checks, confidence gates, deduplication, portfolio health assessments, and a multi-step position sizer — then submitting approved orders through the broker adapter to Alpaca's paper trading API, with every decision recorded in a fully auditable trail from signal to execution.
|
||||
|
||||
The pipeline is designed to be conservative by default and transparent throughout. Every stage applies its own safety checks — deduplication at ingestion, confidence gates at extraction, contradiction detection at aggregation, suppression at recommendation, and circuit breakers at trading. The system can be tuned through runtime configuration (risk tier parameters, suppression thresholds, signal layer toggles in `risk_configs`) without code changes or restarts. And the complete audit trail — from `documents` through `document_intelligence`, `document_impact_records`, `trend_windows`, `recommendations`, `trading_decisions`, and `orders` — means that any trade can be traced back to the specific documents, signals, and decisions that produced it.
|
||||
@@ -0,0 +1 @@
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
# Ingestion-to-Extraction Flow
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
subgraph Scheduler["Scheduler\nservices/scheduler/app.py"]
|
||||
S1["schedule_cycle()"]
|
||||
S2["Cadence check\nmarket_api: 300s\nnews_api: 300s\nfilings_api: 3600s\nmacro_news: 600s"]
|
||||
S3["Rate limit check\ncheck_rate_limit()"]
|
||||
S1 --> S2 --> S3
|
||||
end
|
||||
|
||||
S3 -->|"rpush"| Q_ING["stonks:queue:ingestion"]
|
||||
|
||||
Q_ING -->|"lpop"| ING
|
||||
|
||||
subgraph ING["Ingestion Worker\nservices/ingestion/worker.py"]
|
||||
direction TB
|
||||
AD["Adapter Dispatch\nprocess_job()"]
|
||||
AD --> PA["PolygonMarketAdapter\nservices/adapters/market_adapter.py"]
|
||||
AD --> PB["PolygonNewsAdapter\nservices/adapters/news_adapter.py"]
|
||||
AD --> PC["SECEdgarAdapter\nservices/adapters/filings_adapter.py"]
|
||||
AD --> PD["MacroNewsAdapter\nservices/adapters/macro_news_adapter.py"]
|
||||
AD --> PE["WebScrapeAdapter\nservices/adapters/web_scrape_adapter.py"]
|
||||
end
|
||||
|
||||
ING -->|"Content hash check\nstonks:dedupe:*\nTTL 24h"| REDIS_DEDUPE[("Redis\nDedupe Markers")]
|
||||
|
||||
ING -->|"upload_raw_artifact()"| MINIO_RAW
|
||||
|
||||
subgraph MINIO_RAW["MinIO Raw Storage"]
|
||||
B1["stonks-raw-market"]
|
||||
B2["stonks-raw-news"]
|
||||
B3["stonks-raw-filings"]
|
||||
end
|
||||
|
||||
ING -->|"persist_ingestion_items()"| PG_ING
|
||||
|
||||
subgraph PG_ING["PostgreSQL"]
|
||||
T1["documents"]
|
||||
T2["ingestion_runs"]
|
||||
T3["document_company_mentions"]
|
||||
end
|
||||
|
||||
ING -->|"rpush new doc IDs"| Q_PARSE["stonks:queue:parsing"]
|
||||
|
||||
Q_PARSE -->|"lpop"| PARSER
|
||||
|
||||
subgraph PARSER["Parser Worker\nservices/parser/worker.py"]
|
||||
P1["fetch_html() → parse_html()"]
|
||||
P2["Quality scoring\nconfidence: high / medium / low"]
|
||||
P3["Company mention detection\ndetect_company_mentions()"]
|
||||
P4["Routing decision"]
|
||||
P1 --> P2 --> P3 --> P4
|
||||
end
|
||||
|
||||
PARSER -->|"upload_normalized_text()\nupload_parser_output()"| MINIO_NORM["MinIO\nstonks-normalized"]
|
||||
PARSER -->|"update_document_parse_results()"| PG_ING
|
||||
|
||||
P4 -->|"doc_type = macro_event"| Q_MACRO["stonks:queue:macro_classification"]
|
||||
P4 -->|"doc_type ≠ macro_event"| Q_EXT["stonks:queue:extraction"]
|
||||
|
||||
Q_EXT -->|"lpop"| EXT
|
||||
Q_MACRO -->|"lpop"| EXT
|
||||
|
||||
subgraph EXT["Extractor Worker\nservices/extractor/main.py"]
|
||||
E1["Document Intelligence\nExtractor agent\nslug: document-extractor"]
|
||||
E2["Global Event Classifier\nslug: event-classifier\nservices/extractor/event_classifier.py"]
|
||||
E3["persist_extraction()\nservices/extractor/worker.py"]
|
||||
end
|
||||
|
||||
EXT -->|"persist to"| PG_EXT
|
||||
|
||||
subgraph PG_EXT["PostgreSQL"]
|
||||
T4["document_intelligence"]
|
||||
T5["document_impact_records"]
|
||||
T6["global_events"]
|
||||
T7["macro_impact_records"]
|
||||
end
|
||||
|
||||
EXT -->|"rpush"| Q_AGG["stonks:queue:aggregation"]
|
||||
```
|
||||
@@ -0,0 +1,80 @@
|
||||
# Recommendation Generation Flow
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Q_REC["stonks:queue:recommendation"] -->|"lpop"| WORKER["Recommendation Worker\nservices/recommendation/main.py"]
|
||||
|
||||
WORKER --> FETCH["Fetch TrendSummary\nfrom trend_windows\nfor ticker + window"]
|
||||
|
||||
FETCH --> SUPP
|
||||
|
||||
subgraph SUPP["Data Quality Suppression\nservices/recommendation/suppression.py"]
|
||||
S1["extraction confidence < 0.40?"]
|
||||
S2["evidence staleness > 168h?"]
|
||||
S3["source diversity < 1 type?"]
|
||||
S4["extraction failure rate > 50%?"]
|
||||
S5["valid documents < 2?"]
|
||||
S6["data quality score < 0.30?"]
|
||||
S7["Macro-only signal?\nevaluate_macro_only_suppression()"]
|
||||
S8["Pattern-only signal?\nevaluate_pattern_only_suppression()"]
|
||||
end
|
||||
|
||||
SUPP -->|"Any check fails:\nsuppressed = true\nmode → informational"| ELIG
|
||||
SUPP -->|"All checks pass"| ELIG
|
||||
|
||||
subgraph ELIG["Eligibility Evaluation\nservices/recommendation/eligibility.py"]
|
||||
direction TB
|
||||
G["Gate Checks"]
|
||||
G1["confidence ≥ 0.35"]
|
||||
G2["strength ≥ 0.10"]
|
||||
G3["contradiction ≤ 0.60"]
|
||||
G4["evidence ≥ 2"]
|
||||
G5["direction ≠ neutral"]
|
||||
G --> G1 & G2 & G3 & G4 & G5
|
||||
|
||||
G1 & G2 & G3 & G4 & G5 --> ACT["Action Mapping"]
|
||||
ACT --> A1["BUY: bullish + strength ≥ 0.25"]
|
||||
ACT --> A2["SELL: bearish + strength ≥ 0.25"]
|
||||
ACT --> A3["HOLD: directional + confidence ≥ 0.50"]
|
||||
ACT --> A4["WATCH: otherwise"]
|
||||
|
||||
A1 & A2 & A3 & A4 --> MODE["Mode Escalation"]
|
||||
MODE --> M1["informational\n(default for HOLD/WATCH)"]
|
||||
MODE --> M2["paper_eligible\nconfidence ≥ 0.50"]
|
||||
MODE --> M3["live_eligible\nconfidence ≥ 0.70\ncontradiction ≤ 0.25\nevidence ≥ 5"]
|
||||
end
|
||||
|
||||
ELIG --> SIZING
|
||||
|
||||
subgraph SIZING["Position Sizing\nservices/recommendation/eligibility.py"]
|
||||
PS1["base = 1% portfolio"]
|
||||
PS2["scale by confidence × strength\nup to 10% max"]
|
||||
PS3["contradiction penalty\n−0.5 × contradiction_score"]
|
||||
PS4["evidence count penalty\n< 3 docs → ×0.5\n< 5 docs → ×0.75"]
|
||||
end
|
||||
|
||||
SIZING --> THESIS
|
||||
|
||||
subgraph THESIS["Thesis Generation"]
|
||||
TH1["Deterministic thesis\nassembled from trend data"]
|
||||
TH2["Optional LLM rewrite\nthesis-rewriter agent\nservices/recommendation/thesis_llm.py"]
|
||||
TH1 --> TH2
|
||||
end
|
||||
|
||||
THESIS --> RISK
|
||||
|
||||
subgraph RISK["Risk Classification"]
|
||||
RC1["low"]
|
||||
RC2["moderate"]
|
||||
RC3["high"]
|
||||
RC4["very_high"]
|
||||
end
|
||||
|
||||
RISK --> PERSIST
|
||||
|
||||
subgraph PERSIST["Persistence — PostgreSQL"]
|
||||
P1["recommendations"]
|
||||
P2["recommendation_evidence"]
|
||||
P3["risk_evaluations"]
|
||||
end
|
||||
```
|
||||
@@ -0,0 +1,52 @@
|
||||
# Three-Layer Signal Merging
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
subgraph Layer1["Layer 1 — Company Signals"]
|
||||
DIR["document_impact_records\n(per-company extraction output)"]
|
||||
DIR -->|"build_weighted_signals()"| WS1["WeightedSignal[]\nweight = 1.0 (full)"]
|
||||
end
|
||||
|
||||
subgraph Layer2["Layer 2 — Macro Signals"]
|
||||
MIR["macro_impact_records\n(global event interpolation)"]
|
||||
MIR -->|"build_macro_weighted_signals()"| WS2["WeightedSignal[]\nimpact × MACRO_SIGNAL_WEIGHT\n(0.3)"]
|
||||
TOGGLE_M{"macro_enabled\nin risk_configs?"}
|
||||
TOGGLE_M -->|"true"| MIR
|
||||
TOGGLE_M -->|"false"| SKIP_M["Layer skipped\ngraceful degradation"]
|
||||
end
|
||||
|
||||
subgraph Layer3["Layer 3 — Competitive Signals"]
|
||||
CSR["competitive_signal_records\n(pattern mining + propagation)"]
|
||||
CSR -->|"build_pattern_weighted_signals()\nservices/aggregation/signal_propagation.py"| WS3["WeightedSignal[]\nimpact × COMPETITIVE_SIGNAL_WEIGHT\n(0.2)"]
|
||||
TOGGLE_C{"competitive_enabled\nin risk_configs?"}
|
||||
TOGGLE_C -->|"true"| CSR
|
||||
TOGGLE_C -->|"false"| SKIP_C["Layer skipped\ngraceful degradation"]
|
||||
end
|
||||
|
||||
WS1 --> MERGE["Concatenate all WeightedSignal lists"]
|
||||
WS2 --> MERGE
|
||||
WS3 --> MERGE
|
||||
|
||||
MERGE --> AGG
|
||||
|
||||
subgraph AGG["Aggregation Engine\nservices/aggregation/worker.py"]
|
||||
A1["weighted_sentiment_average()"]
|
||||
A2["detect_contradictions()\nservices/aggregation/contradiction.py"]
|
||||
A3["derive_trend_direction()"]
|
||||
A4["compute_trend_confidence()"]
|
||||
A5["rank_evidence()"]
|
||||
A1 --> A2 --> A3 --> A4 --> A5
|
||||
end
|
||||
|
||||
AGG -->|"assemble_trend_summary()"| TS["TrendSummary\nservices/shared/schemas.py"]
|
||||
|
||||
TS -->|"persist_trend_summary()"| PG_TREND
|
||||
|
||||
subgraph PG_TREND["PostgreSQL"]
|
||||
TW["trend_windows\n(upserted each cycle)"]
|
||||
TH["trend_history\n(time-series snapshots)"]
|
||||
TE["trend_evidence\n(per-document rankings)"]
|
||||
end
|
||||
|
||||
AGG -->|"rpush"| Q_REC["stonks:queue:recommendation"]
|
||||
```
|
||||
@@ -0,0 +1,94 @@
|
||||
# Trading Engine Decision Loop
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
subgraph ENGINE["Trading Engine\nservices/trading/engine.py"]
|
||||
direction TB
|
||||
TASKS["5 Concurrent Async Tasks"]
|
||||
T1["_decision_loop()\n60s polling interval"]
|
||||
T2["_stop_loss_monitor()"]
|
||||
T3["_performance_loop()"]
|
||||
T4["_risk_tier_scheduler()"]
|
||||
T5["_rebalance_scheduler()"]
|
||||
TASKS --> T1 & T2 & T3 & T4 & T5
|
||||
end
|
||||
|
||||
T1 --> POLL["Poll recommendations table\naction IN (buy, sell)\nmode IN (paper_eligible, live_eligible)\ngenerated_at > NOW() − 2h"]
|
||||
|
||||
POLL --> EVAL["evaluate_recommendation()"]
|
||||
|
||||
EVAL --> CHK_A
|
||||
|
||||
subgraph PRETRADE["Pre-Trade Check Sequence\n(first failure short-circuits)"]
|
||||
direction TB
|
||||
CHK_A["a. Circuit Breaker active?\nservices/trading/circuit_breaker.py\nTriggers: daily_loss, single_position, volatility"]
|
||||
CHK_B["b. Trading Window?\nis_within_trading_window()"]
|
||||
CHK_C["c. Confidence Gate\nconfidence ≥ risk_tier.min_confidence"]
|
||||
CHK_D["d. Deduplication\nRec ID in processed set?\nRedis: stonks:dedupe:trading:*"]
|
||||
CHK_E["e. Declining Positions\n> 50% positions down > 2%"]
|
||||
CHK_F["f. Max Open Positions\nopen_count ≥ max (default 10)"]
|
||||
|
||||
CHK_A -->|"pass"| CHK_B
|
||||
CHK_B -->|"pass"| CHK_C
|
||||
CHK_C -->|"pass"| CHK_D
|
||||
CHK_D -->|"pass"| CHK_E
|
||||
CHK_E -->|"pass"| CHK_F
|
||||
end
|
||||
|
||||
CHK_A & CHK_B & CHK_C & CHK_D & CHK_E & CHK_F -->|"fail"| SKIP["TradingDecision\ndecision = skip\n+ skip_reason"]
|
||||
|
||||
CHK_F -->|"pass"| SIZER
|
||||
|
||||
subgraph SIZER["Position Sizing\nservices/trading/position_sizer.py"]
|
||||
direction TB
|
||||
SZ1["Base sizing\nrisk_tier.max_position_pct × 0.5\n× (confidence / min_confidence)"]
|
||||
SZ2["Correlation reduction\nweighted avg corr > 0.8 → reject\n> 0.5 → proportional reduction"]
|
||||
SZ3["Sector exposure\ncap at risk_tier.max_sector_pct"]
|
||||
SZ4["Diversification bonus\n1.2× for new sector (< 3 sectors)"]
|
||||
SZ5["Earnings proximity\n≤ 1 day → reject\n≤ 3 days → 50% reduction"]
|
||||
SZ6["Absolute position cap"]
|
||||
SZ7["Portfolio heat check\nmax_portfolio_heat × active_pool"]
|
||||
SZ8["Share rounding\nfloor(dollar / price)"]
|
||||
|
||||
SZ1 --> SZ2 --> SZ3 --> SZ4 --> SZ5 --> SZ6 --> SZ7 --> SZ8
|
||||
end
|
||||
|
||||
SIZER -->|"rejected"| SKIP
|
||||
SIZER -->|"approved"| ACT["TradingDecision\ndecision = act\nshares, dollar amount"]
|
||||
|
||||
ACT --> PERSIST_TD["Persist to\ntrading_decisions"]
|
||||
|
||||
ACT --> ORDER["Build order job\n{ticker, action, side,\nquantity, order_type}"]
|
||||
|
||||
ORDER -->|"rpush"| Q_BROKER["stonks:queue:broker_orders"]
|
||||
|
||||
Q_BROKER --> BROKER["Broker Adapter\nAlpaca paper trading\nservices/adapters/broker_adapter.py"]
|
||||
|
||||
BROKER --> AUDIT
|
||||
|
||||
subgraph AUDIT["Audit Trail — PostgreSQL"]
|
||||
AU1["orders"]
|
||||
AU2["positions"]
|
||||
AU3["portfolio_snapshots"]
|
||||
end
|
||||
|
||||
subgraph CB_DETAIL["Circuit Breaker Detail\nservices/trading/circuit_breaker.py"]
|
||||
CB1["daily_loss\nportfolio loss > 5%\ncooldown: volatility_pause_hours"]
|
||||
CB2["single_position\nposition loss > 15%\ncooldown: ticker_cooldown_hours (48h)"]
|
||||
CB3["volatility\n≥ 3 stop-losses in 30min\ncooldown: volatility_pause_hours (2h)"]
|
||||
CB4["Redis state\nstonks:trading:circuit_breaker:*"]
|
||||
end
|
||||
|
||||
subgraph RESERVE["Reserve Pool\nservices/trading/reserve_pool.py"]
|
||||
RP1["Profit siphoning: 20%"]
|
||||
RP2["High-water rebalance: 30%"]
|
||||
RP3["Emergency liquidation"]
|
||||
RP4["reserve_pool_ledger"]
|
||||
end
|
||||
|
||||
subgraph RISK_TIER["Risk Tier Auto-Adjustment\nservices/trading/risk_tier_controller.py"]
|
||||
RT1["Evaluate: Sharpe ratio,\ndrawdown, win rate"]
|
||||
RT2["conservative → moderate → aggressive"]
|
||||
RT3["risk_tier_history"]
|
||||
end
|
||||
```
|
||||
@@ -0,0 +1,62 @@
|
||||
# Trend Accumulation and Escalation
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
subgraph Windows["Five Time Windows\nservices/aggregation/worker.py"]
|
||||
W1["intraday (12h)"]
|
||||
W2["1d (1 day)"]
|
||||
W3["7d (7 days)"]
|
||||
W4["30d (30 days)"]
|
||||
W5["90d (90 days)"]
|
||||
end
|
||||
|
||||
W1 & W2 & W3 & W4 & W5 --> SIGNALS
|
||||
|
||||
SIGNALS["Fetch signals per window\nCompany + Macro + Competitive\n→ WeightedSignal[]"]
|
||||
|
||||
SIGNALS --> SENT["weighted_sentiment_average()\nCompute avg sentiment across signals"]
|
||||
|
||||
SENT --> DIR
|
||||
|
||||
subgraph DIR["derive_trend_direction()"]
|
||||
D1["avg_sentiment ≥ 0.15 → BULLISH"]
|
||||
D2["avg_sentiment ≤ −0.15 → BEARISH"]
|
||||
D3["contradiction > 0.10\nAND |avg| < 0.30 → MIXED"]
|
||||
D4["otherwise → NEUTRAL"]
|
||||
end
|
||||
|
||||
DIR --> CONF
|
||||
|
||||
subgraph CONF["compute_trend_confidence()"]
|
||||
C1["Unique source count\ncaps at 15 → 0.8 contribution"]
|
||||
C2["Avg extraction credibility"]
|
||||
C3["Signal agreement ratio\ndampened by log₂(n+1)/log₂(8)\nsaturates ~7 unique sources"]
|
||||
C4["Contradiction penalty\n−0.4 × contradiction_score"]
|
||||
C5["confidence = 0.3×count + 0.3×credibility\n+ 0.4×agreement − penalty"]
|
||||
end
|
||||
|
||||
CONF --> STRENGTH["trend_strength = |avg_sentiment|\nclamped to [0, 1]"]
|
||||
|
||||
STRENGTH --> ESC
|
||||
|
||||
subgraph ESC["Escalation Path\n(via eligibility thresholds)"]
|
||||
direction TB
|
||||
E1["NEUTRAL\nconfidence < 0.35\nOR strength < 0.10\nOR direction = neutral"]
|
||||
E2["WATCH\nstrength < 0.25\nAND confidence < 0.50"]
|
||||
E3["HOLD\nstrength < 0.25\nAND confidence ≥ 0.50"]
|
||||
E4["BUY / SELL\nstrength ≥ 0.25\nAND direction = bullish/bearish"]
|
||||
|
||||
E1 -->|"More signals\nsame direction"| E2
|
||||
E2 -->|"Confidence grows\nmore unique sources"| E3
|
||||
E3 -->|"Strength exceeds 0.25\naccumulated evidence"| E4
|
||||
end
|
||||
|
||||
ESC --> PERSIST
|
||||
|
||||
subgraph PERSIST["Persistence"]
|
||||
P1["trend_windows\n(upserted each cycle)"]
|
||||
P2["trend_history\n(time-series snapshots)"]
|
||||
P3["trend_evidence\n(per-document rankings)"]
|
||||
P4["trend_projections\nservices/aggregation/projection.py"]
|
||||
end
|
||||
```
|
||||
@@ -0,0 +1,58 @@
|
||||
# Weighted Signal Computation
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
DOC["Document Signal Input\n(published_at, source_credibility,\nnovelty_score, extraction_confidence,\nmarket_ctx)"]
|
||||
|
||||
DOC --> GATE
|
||||
DOC --> REC
|
||||
DOC --> CRED
|
||||
DOC --> NOV
|
||||
DOC --> MKT
|
||||
|
||||
subgraph GATE["Confidence Gate"]
|
||||
G1["extraction_confidence ≥ 0.2?"]
|
||||
G1 -->|"Yes"| G2["gate = 1.0"]
|
||||
G1 -->|"No"| G3["gate = 0.0\n(signal zeroed out)"]
|
||||
end
|
||||
|
||||
subgraph REC["Recency Decay"]
|
||||
R1["w = 2^(−age_hours / half_life)"]
|
||||
R2["Half-lives per window:\nintraday: 2h\n1d: 12h\n7d: 72h\n30d: 240h\n90d: 720h"]
|
||||
R3["Floor: min_recency_weight = 0.01"]
|
||||
R1 --- R2
|
||||
R1 --- R3
|
||||
end
|
||||
|
||||
subgraph CRED["Source Credibility"]
|
||||
C1["Clamp to [0.1, 1.0]"]
|
||||
C2["Apply exponent\n(default 1.0)"]
|
||||
C1 --> C2
|
||||
end
|
||||
|
||||
subgraph NOV["Novelty Bonus"]
|
||||
N1["bonus = novelty_score × 0.25"]
|
||||
N2["Range: [0.0, 0.25]\n(up to 25% boost)"]
|
||||
N1 --- N2
|
||||
end
|
||||
|
||||
subgraph MKT["Market Context Multiplier"]
|
||||
M1["Volatility boost\nlog₁₊(excess) × 0.15\ncapped at 0.30"]
|
||||
M2["Volume surge boost\nvolume_change > 50% → +0.15"]
|
||||
M3["multiplier = 1.0 + boost\n(always ≥ 1.0)"]
|
||||
M1 --> M3
|
||||
M2 --> M3
|
||||
end
|
||||
|
||||
GATE --> FORMULA
|
||||
REC --> FORMULA
|
||||
CRED --> FORMULA
|
||||
NOV --> FORMULA
|
||||
MKT --> FORMULA
|
||||
|
||||
FORMULA["combined = gate × recency × credibility\n× (1 + novelty_bonus)\n× market_context_multiplier"]
|
||||
|
||||
FORMULA --> SW["SignalWeight\nservices/aggregation/scoring.py"]
|
||||
|
||||
SW --> WS["WeightedSignal\n{ document_id, weight: SignalWeight,\nsentiment_value, impact_score }"]
|
||||
```
|
||||
@@ -0,0 +1,40 @@
|
||||
# Intelligence Pipeline Deep Dive
|
||||
|
||||
This document series provides a narrative walkthrough of the full intelligence-to-decision pipeline in Stonks Oracle. Unlike the existing service reference and API documentation, these pages tell the story of how raw data enters the system, gets processed by AI agents, produces structured signals, accumulates into trend summaries, and ultimately drives autonomous trading decisions.
|
||||
|
||||
Each page covers one stage of the pipeline and ends with a transition to the next, so you can read the series end-to-end or jump directly to the stage you need. Diagrams are stored as standalone Mermaid files that can be rendered independently or embedded in other documents.
|
||||
|
||||
---
|
||||
|
||||
## Table of Contents
|
||||
|
||||
1. [Data Ingestion and Preparation](01-data-ingestion-and-preparation.md) — How raw data from Polygon.io, SEC EDGAR, and macro news APIs enters the system, gets deduplicated, stored, parsed, and routed for AI processing.
|
||||
2. [AI Agent Processing and Structured Extraction](02-ai-agent-processing-and-extraction.md) — How the Document Intelligence Extractor and Global Event Classifier agents use LLM inference to produce structured JSON intelligence from documents.
|
||||
3. [Signal Scoring and the WeightedSignal Abstraction](03-signal-scoring-and-weighted-signals.md) — How raw extraction output is transformed into weighted signals through confidence gating, recency decay, source credibility, novelty bonuses, and market context multipliers.
|
||||
4. [Trend Aggregation and Accumulating Signals](04-trend-aggregation-and-accumulating-signals.md) — How the aggregation engine merges weighted signals across five time windows, detects contradictions, ranks evidence, and escalates trend strength as consecutive signals accumulate.
|
||||
5. [Recommendation Generation](05-recommendation-generation.md) — How trend summaries pass through data quality suppression, eligibility evaluation, position sizing, thesis generation, and risk classification to produce actionable recommendations.
|
||||
6. [Trading Decisions and Execution](06-trading-decisions-and-execution.md) — How the trading engine polls recommendations, runs pre-trade checks, sizes positions, enforces circuit breakers, and submits orders through the broker adapter.
|
||||
|
||||
---
|
||||
|
||||
## Diagrams
|
||||
|
||||
The following Mermaid diagram files can be rendered independently or referenced from the narrative pages:
|
||||
|
||||
- [Ingestion to Extraction Flow](diagrams/ingestion-to-extraction-flow.md) — Flowchart from Scheduler through Ingestion, Parser, to Extractor with all queues and storage.
|
||||
- [Three-Layer Signal Merging](diagrams/three-layer-signal-merging.md) — Company, Macro, and Competitive signal layers converging into the Aggregation engine.
|
||||
- [Weighted Signal Computation](diagrams/weighted-signal-computation.md) — Component breakdown of the composite weight formula.
|
||||
- [Trend Accumulation and Escalation](diagrams/trend-accumulation-escalation.md) — How consecutive signals strengthen trends and escalate actions across time windows.
|
||||
- [Recommendation Generation Flow](diagrams/recommendation-generation-flow.md) — From TrendSummary through suppression, eligibility, thesis, risk classification, to persistence.
|
||||
- [Trading Engine Decision Loop](diagrams/trading-engine-decision-loop.md) — Pre-trade check sequence, position sizing, and order submission flow.
|
||||
|
||||
---
|
||||
|
||||
## Related Documentation
|
||||
|
||||
For reference-level detail on individual services, AI agent configuration, and infrastructure, see the existing documentation:
|
||||
|
||||
- [Services Reference](../services.md) — Per-service configuration, database tables, queues, and runtime behaviors.
|
||||
- [AI Agents Guide](../ai-agents.md) — AI agent configuration, variants, A/B testing, and the agent management API.
|
||||
- [Data Pipeline Architecture](../architecture-data-pipeline.md) — Queue topology, data store summary, and Mermaid flow diagrams for the full data pipeline.
|
||||
- [LLM-to-Trade Pipeline](../llm-to-trade-pipeline.md) — End-to-end data flow from model output through signal aggregation to trade execution.
|
||||
@@ -0,0 +1,363 @@
|
||||
# V3 Annotation Guidelines
|
||||
|
||||
**Schema version:** 1.0.0
|
||||
**Last updated:** 2025-01-15
|
||||
|
||||
## Purpose
|
||||
|
||||
These guidelines define how human annotators and automated systems label documents in the Intelligence Pipeline v3 Gold Corpus. Every annotation must be evidence-grounded — no label is valid without a supporting evidence span traceable to the source text.
|
||||
|
||||
## Core Principles
|
||||
|
||||
1. **Evidence first.** If you cannot point to exact text that supports a label, do not apply the label.
|
||||
2. **Explicit over inferred.** Mark only what the document explicitly states in primary annotations. Inferred exposure uses a separate, lower-confidence channel.
|
||||
3. **Precision over recall.** A missed entity is preferable to a fabricated one. The pipeline uses multiple stages — later stages catch omissions.
|
||||
4. **Reproducibility.** Two annotators given the same document should produce substantially the same labels. Ambiguous cases are marked, not resolved by guess.
|
||||
|
||||
---
|
||||
|
||||
## Evidence Spans
|
||||
|
||||
### Definition
|
||||
|
||||
An evidence span is the exact substring of the source document that supports an annotation. It uses zero-based character offsets into the original (pre-chunking) document text.
|
||||
|
||||
### Rules
|
||||
|
||||
- Every entity, event, relation, numeric fact, and sentiment annotation MUST reference at least one evidence span.
|
||||
- Spans should be minimal but complete — include enough context for the label to be verifiable without the full document.
|
||||
- Overlapping spans are permitted (e.g., the same sentence supports both an entity and an event).
|
||||
- The `text` field MUST exactly match `source_text[start_char:end_char]`.
|
||||
|
||||
### Positive example
|
||||
|
||||
```
|
||||
Source: "Apple Inc. reported quarterly earnings of $1.52 per share"
|
||||
Span: start_char=0, end_char=10, text="Apple Inc."
|
||||
```
|
||||
|
||||
### Negative example
|
||||
|
||||
```
|
||||
Source: "Apple Inc. reported quarterly earnings of $1.52 per share"
|
||||
Span: start_char=0, end_char=5, text="Apple"
|
||||
```
|
||||
❌ Truncating "Apple Inc." to "Apple" loses the corporate suffix needed to distinguish from Apple Records or the fruit.
|
||||
|
||||
---
|
||||
|
||||
## Entity Annotation
|
||||
|
||||
### Entity Types
|
||||
|
||||
| Type | When to use | Example |
|
||||
|------|-------------|---------|
|
||||
| `company` | Legal entity, publicly traded firm, government agency | "Apple Inc.", "The Federal Reserve" |
|
||||
| `person` | Named individual | "Tim Cook", "Jerome Powell" |
|
||||
| `product` | Named product or service | "iPhone 16", "Azure OpenAI Service" |
|
||||
| `event` | Named event instance | "Q1 2025 earnings call" |
|
||||
| `financial_metric` | Named metric class | "EPS", "revenue", "free cash flow" |
|
||||
| `date` | Temporal expression | "Q1 2025", "January 15, 2025" |
|
||||
| `percentage` | Percentage value | "4%", "25 basis points" |
|
||||
| `currency` | Monetary value | "$1.52", "$10 billion" |
|
||||
| `relationship` | Explicit relationship mention | "subsidiary", "joint venture partner" |
|
||||
|
||||
### Canonical Resolution
|
||||
|
||||
- If an entity maps to a company in the symbol registry, set `canonical_id` and `canonical_name` (ticker).
|
||||
- If an entity is ambiguous (e.g., "Apple" could be AAPL or a fruit company), mark an ambiguity marker and set confidence below 1.0.
|
||||
- Do NOT invent canonical IDs. If not in the registry, leave `canonical_id` as null.
|
||||
|
||||
### Positive example
|
||||
|
||||
```json
|
||||
{
|
||||
"entity_type": "company",
|
||||
"literal_text": "Alphabet",
|
||||
"canonical_id": "googl-uuid",
|
||||
"canonical_name": "GOOGL",
|
||||
"confidence": 0.97
|
||||
}
|
||||
```
|
||||
|
||||
### Negative example
|
||||
|
||||
```json
|
||||
{
|
||||
"entity_type": "company",
|
||||
"literal_text": "the company",
|
||||
"canonical_id": "aapl-uuid",
|
||||
"canonical_name": "AAPL",
|
||||
"confidence": 0.90
|
||||
}
|
||||
```
|
||||
❌ "the company" is a pronoun reference, not an entity mention. Resolve coreference but annotate the actual named mention, not the pronoun.
|
||||
|
||||
---
|
||||
|
||||
## Event Classification
|
||||
|
||||
### Event Classes
|
||||
|
||||
| Class | Definition | Distinguishing criteria |
|
||||
|-------|-----------|------------------------|
|
||||
| `earnings_beat` | Reported EPS or revenue exceeds consensus | Explicit comparison to estimates |
|
||||
| `earnings_miss` | Reported EPS or revenue below consensus | Explicit comparison to estimates |
|
||||
| `guidance_raise` | Forward guidance raised vs prior or consensus | Future-looking, not historical result |
|
||||
| `guidance_cut` | Forward guidance lowered | Future-looking, not historical result |
|
||||
| `ma_announcement` | Merger, acquisition, investment, or divestiture | Transaction between entities |
|
||||
| `legal_regulatory` | Lawsuit, fine, regulatory action, or settlement | Legal or regulatory body involved |
|
||||
| `product_launch` | New product, service, or major feature announced | Not routine updates |
|
||||
| `supply_chain` | Disruption, partnership, or change in supply relationships | Affects production/delivery |
|
||||
| `rating_change` | Analyst upgrade, downgrade, or target change | From research analyst/firm |
|
||||
| `management_change` | CEO/CFO/board appointment, resignation, or removal | C-suite or board level |
|
||||
| `macro_event` | Interest rates, policy, trade, geopolitical | Not specific to one company |
|
||||
| `dividend_change` | Dividend increase, decrease, or special dividend | Shareholder distribution |
|
||||
| `buyback` | Share repurchase program announcement or completion | Capital return via buyback |
|
||||
|
||||
### Adjudication triggers for events
|
||||
|
||||
Route to the 9B adjudicator when:
|
||||
- The same facts could be classified as multiple event types (e.g., guidance_raise during an earnings call could be either earnings_beat or guidance_raise — label the most specific applicable class).
|
||||
- The event is implied but not explicitly stated.
|
||||
- The primary company is unclear.
|
||||
|
||||
### Positive example
|
||||
|
||||
```
|
||||
Source: "Apple beat earnings expectations with EPS of $1.52 vs $1.43 expected"
|
||||
Event class: earnings_beat
|
||||
Confidence: 0.98
|
||||
```
|
||||
|
||||
### Negative example
|
||||
|
||||
```
|
||||
Source: "Apple reported EPS of $1.52"
|
||||
Event class: earnings_beat
|
||||
```
|
||||
❌ Without a comparison to consensus/estimates, this is a numeric fact report, not an earnings beat. The document must provide evidence of beating expectations.
|
||||
|
||||
---
|
||||
|
||||
## Relations
|
||||
|
||||
### Relation Types
|
||||
|
||||
| Type | Subject | Object | When to use |
|
||||
|------|---------|--------|-------------|
|
||||
| `directly_affects` | Event | Company | Event explicitly names or discusses the company |
|
||||
| `inferred_exposure` | Event | Company | Exposure inferred from sector, supply chain, or competition |
|
||||
| `competes_with` | Company | Company | Competitive relationship stated or clearly implied |
|
||||
| `supplies` | Company | Company | Supply chain relationship stated |
|
||||
|
||||
### Critical distinction: directly_affects vs inferred_exposure
|
||||
|
||||
- `directly_affects`: The document **explicitly states** the company is impacted. Evidence span exists.
|
||||
- `inferred_exposure`: The impact is **reasoned** from relationships, not stated. May have weak or no direct evidence span.
|
||||
|
||||
Only `directly_affects` enters primary company extraction. `inferred_exposure` flows through the separate interpolation/propagation architecture with distinct confidence and provenance.
|
||||
|
||||
### Positive example (directly_affects)
|
||||
|
||||
```
|
||||
Source: "Microsoft announced a $10 billion investment in OpenAI"
|
||||
Relation: directly_affects(event=ma_announcement, company=Microsoft)
|
||||
Evidence: "Microsoft announced"
|
||||
```
|
||||
|
||||
### Negative example (incorrectly using directly_affects)
|
||||
|
||||
```
|
||||
Source: "Microsoft announced a $10 billion investment in OpenAI"
|
||||
Relation: directly_affects(event=ma_announcement, company=Google)
|
||||
```
|
||||
❌ Google is not mentioned in the event sentence. This should be `inferred_exposure` based on competitive relationship, with appropriate lower confidence.
|
||||
|
||||
---
|
||||
|
||||
## Numeric Facts
|
||||
|
||||
### Annotation rules
|
||||
|
||||
1. Always store both `literal_value` (exact text) and `normalized_value` (parsed number).
|
||||
2. Include `unit` (USD, %, bps, shares, etc.).
|
||||
3. Link to the subject entity when determinable.
|
||||
4. Use `predicate` to capture the semantic role: reported, expected, raised_to, cut_to, beat_by, missed_by.
|
||||
5. Include `period` when the fact references a specific time frame.
|
||||
|
||||
### Normalization conventions
|
||||
|
||||
| Literal | Normalized | Unit |
|
||||
|---------|-----------|------|
|
||||
| "$1.52" | 1.52 | USD |
|
||||
| "$94.9 billion" | 94900000000 | USD |
|
||||
| "25 basis points" | 0.25 | percentage_points |
|
||||
| "4%" | 4.0 | % |
|
||||
| "$0.26 per share" | 0.26 | USD |
|
||||
|
||||
### Positive example
|
||||
|
||||
```json
|
||||
{
|
||||
"fact_type": "eps",
|
||||
"predicate": "reported",
|
||||
"literal_value": "$1.52 per share",
|
||||
"normalized_value": 1.52,
|
||||
"unit": "USD",
|
||||
"period": {"period_type": "fiscal_quarter", "fiscal_year": 2025, "fiscal_quarter": 1}
|
||||
}
|
||||
```
|
||||
|
||||
### Negative example
|
||||
|
||||
```json
|
||||
{
|
||||
"fact_type": "eps",
|
||||
"predicate": "reported",
|
||||
"literal_value": "$1.52 per share",
|
||||
"normalized_value": 152,
|
||||
"unit": "cents"
|
||||
}
|
||||
```
|
||||
❌ While $1.52 = 152 cents, always normalize to the unit stated in the source. Conversion to a different unit introduces potential confusion.
|
||||
|
||||
---
|
||||
|
||||
## Sentiment
|
||||
|
||||
### Rules
|
||||
|
||||
1. Sentiment is **company-specific**, not document-level. A single article can have positive sentiment for one company and negative for another.
|
||||
2. Annotate probability distributions (positive, negative, neutral) that sum to 1.0.
|
||||
3. `mixed` label is used when evidence groups disagree — it is computed from evidence-group-level disagreement, NOT an unconstrained fourth class.
|
||||
4. The label should reflect the dominant probability.
|
||||
|
||||
### When to label "mixed"
|
||||
|
||||
Label `mixed` when:
|
||||
- Different paragraphs contain opposing sentiment for the same company
|
||||
- The same fact has both positive and negative implications (e.g., restructuring = cost cuts but also layoffs)
|
||||
- Analyst opinions explicitly disagree within the document
|
||||
|
||||
Do NOT label `mixed` when:
|
||||
- Sentiment is merely uncertain or mild — that's `neutral` with lower confidence
|
||||
- The document discusses multiple companies with different sentiments — annotate separately per company
|
||||
|
||||
### Positive example
|
||||
|
||||
```json
|
||||
{
|
||||
"label": "mixed",
|
||||
"positive_probability": 0.40,
|
||||
"negative_probability": 0.45,
|
||||
"neutral_probability": 0.15,
|
||||
"evidence_ids": ["ev-pressure", "ev-validation"]
|
||||
}
|
||||
```
|
||||
(Article says AI investment pressures cloud revenue but validates the broader thesis)
|
||||
|
||||
### Negative example
|
||||
|
||||
```json
|
||||
{
|
||||
"label": "mixed",
|
||||
"positive_probability": 0.85,
|
||||
"negative_probability": 0.05,
|
||||
"neutral_probability": 0.10
|
||||
}
|
||||
```
|
||||
❌ When positive_probability dominates at 0.85, the label should be `positive`, not `mixed`. Mixed requires genuine disagreement in evidence.
|
||||
|
||||
---
|
||||
|
||||
## Direct Effects vs Inferred Exposure
|
||||
|
||||
### Direct Effects
|
||||
|
||||
A direct effect means the document **explicitly states or clearly demonstrates** that an event impacts a specific company.
|
||||
|
||||
**Criteria:**
|
||||
- The company is named in the same sentence or paragraph as the event
|
||||
- The causal link is stated, not inferred
|
||||
- Evidence span directly connects event to company
|
||||
|
||||
### Inferred Exposure
|
||||
|
||||
Inferred exposure captures **reasoned but unstated** impacts on companies.
|
||||
|
||||
**Criteria:**
|
||||
- The company is NOT explicitly linked to the event in the source text
|
||||
- The connection comes from known relationships (competitor, supplier, sector peer)
|
||||
- Confidence should be lower than direct effects (typically 0.5–0.8)
|
||||
- Requires `reasoning` field explaining the inference chain
|
||||
|
||||
### Adjudication routing
|
||||
|
||||
When it's unclear whether an effect is direct or inferred, mark an ambiguity marker with type `implied_causal_impact` and route to the 9B adjudicator.
|
||||
|
||||
---
|
||||
|
||||
## Ambiguity Markers
|
||||
|
||||
### When to flag
|
||||
|
||||
Flag ambiguity when:
|
||||
- An alias resolves to multiple candidate companies (`unresolved_alias`)
|
||||
- Multiple companies could be the primary subject (`multiple_primary_companies`)
|
||||
- Numeric facts within the same document contradict each other (`contradictory_numeric_facts`)
|
||||
- Sentiment evidence points in opposing directions for the same company (`conflicting_sentiment`)
|
||||
- Impact is implied through causal chain, not stated (`implied_causal_impact`)
|
||||
- Guidance must be compared to consensus to determine direction (`guidance_vs_consensus_requires_reasoning`)
|
||||
- A required field cannot be determined from available evidence (`material_field_missing`)
|
||||
- Evidence covers less than the minimum threshold for confident extraction (`evidence_coverage_below_threshold`)
|
||||
- Calibrated confidence falls below the routing threshold (`calibrated_confidence_below_threshold`)
|
||||
- A relation spans multiple document chunks (`long_document_cross_chunk_relation`)
|
||||
|
||||
### Severity levels
|
||||
|
||||
- **low**: The annotation is likely correct but has reduced certainty. Fast path may proceed with a confidence penalty.
|
||||
- **medium**: The annotation requires review. Routes to adjudication by default.
|
||||
- **high**: The annotation cannot be reliably made without semantic reasoning. Always routes to adjudication.
|
||||
|
||||
---
|
||||
|
||||
## Safety-Critical Fields
|
||||
|
||||
The following fields are **safety-critical** for promotion gates. Errors in these fields can directly cause incorrect trading decisions:
|
||||
|
||||
| Field | Why it's critical | Minimum promotion gate |
|
||||
|-------|-------------------|----------------------|
|
||||
| Company identity (ticker) | Wrong ticker = trade on wrong security | Precision ≥ 0.95, Recall ≥ 0.90 |
|
||||
| Event class | Misclassifying beat/miss inverts signal direction | Macro-F1 ≥ 0.85 |
|
||||
| Sentiment direction | Wrong sentiment → wrong position direction | Direction accuracy ≥ 0.90 |
|
||||
| Numeric fact values | Wrong magnitude affects impact estimation | Tolerance match ≥ 0.92 |
|
||||
| Direct effect attribution | Wrong company attribution creates false signals | Precision ≥ 0.93 |
|
||||
| Evidence support | Unsupported claims are unverifiable | Support rate ≥ 0.95 |
|
||||
| Confidence calibration | Overconfidence bypasses review | ECE ≤ 0.05 |
|
||||
|
||||
Annotators must pay special attention to these fields. During review, any error in a safety-critical field requires correction before the annotation can receive "gold" status.
|
||||
|
||||
---
|
||||
|
||||
## Annotation Workflow
|
||||
|
||||
1. **First pass:** Identify all entities and evidence spans
|
||||
2. **Second pass:** Classify events and link to companies
|
||||
3. **Third pass:** Extract numeric facts with periods
|
||||
4. **Fourth pass:** Assess per-company sentiment
|
||||
5. **Fifth pass:** Identify relations, direct effects, and inferred exposures
|
||||
6. **Sixth pass:** Flag ambiguities and set confidence levels
|
||||
7. **Review:** Senior annotator validates safety-critical fields
|
||||
|
||||
### Inter-annotator agreement
|
||||
|
||||
Hard cases (flagged with ambiguity markers) receive double annotation. Inter-annotator agreement is measured per field type using Cohen's kappa. Target: κ ≥ 0.80 for entity and event labels, κ ≥ 0.70 for relations and sentiment.
|
||||
|
||||
---
|
||||
|
||||
## Version History
|
||||
|
||||
| Version | Date | Changes |
|
||||
|---------|------|---------|
|
||||
| 1.0.0 | 2025-01-15 | Initial schema and guidelines |
|
||||
@@ -0,0 +1,84 @@
|
||||
# Session Context — July 11, 2026
|
||||
|
||||
## Current State
|
||||
|
||||
### Active Namespace: `stonks-beta`
|
||||
- This is the ONLY namespace that should be running
|
||||
- `stonks-oracle` namespace has been scaled to 0 replicas (all deployments)
|
||||
- Dashboard: `https://stonks-beta.celestium.life`
|
||||
- API: `https://stonks-api-beta.celestium.life`
|
||||
|
||||
### What Was Done This Session
|
||||
|
||||
#### 1. Pipeline Health Fixes (spec: `.kiro/specs/pipeline-health-fixes/`)
|
||||
All implemented and deployed:
|
||||
- **Stuck Parsed Docs**: `STALE_PARSED_THRESHOLD_MINUTES` 240→30, `LIMIT` 100→500, `_ENQUEUED_TTL` 14400→3600 (`services/scheduler/app.py`)
|
||||
- **Price Fallback**: Added 24h market_snapshots time-window fallback in `create_prediction_snapshot()` (`services/validation/prediction_snapshot.py`)
|
||||
- **Sentiment Normalization**: Added `normalize_impact_scores()` z-score function (`services/aggregation/scoring.py`) + integrated into `aggregate_company_window()` (`services/aggregation/worker.py`)
|
||||
- **Signal Engine**: Replicas set to 0 in all Helm values files
|
||||
- **Quality Gate**: `max_snapshot_age_hours` 24→48 (`services/trading/model_quality_gate.py`)
|
||||
- **Backfill script**: `scripts/backfill_snapshot_prices.py` (one-time, not yet run on beta)
|
||||
|
||||
#### 2. Extractor Null-Field Fix
|
||||
- `services/extractor/schemas.py`: `_normalize_extraction_data()` now handles `None` values (not just missing keys) and filters out company entries with empty ticker
|
||||
- Test updated: `tests/test_extractor_schemas.py::test_validate_semantic_missing_ticker_is_error`
|
||||
|
||||
#### 3. Macro Doc Status Fix
|
||||
- `services/extractor/main.py`: `_process_macro_classification()` now updates document status to 'extracted' on success, 'extraction_failed' on error
|
||||
- Beta DB: manually fixed 1849 stuck macro docs (UPDATE status='extracted' WHERE id IN global_events)
|
||||
|
||||
#### 4. Dashboard Fix
|
||||
- `frontend/src/pages/OpsPipeline.tsx`: Document Stages now uses time-filtered `/health` data (consistent with other sections), all-time from SSE stream shown as subtitle, time range labels added to all sections
|
||||
|
||||
#### 5. CI/CD DNS Fix
|
||||
- `.woodpecker/*.yml`: All 5 pipeline files now use `clone.git.settings.remote: http://10.43.73.77:3000/admin/stonks-oracle.git` (Gitea ClusterIP directly, bypasses DNS)
|
||||
- CoreDNS: scaled to 4 replicas, `forward . 192.168.42.1`, `dnsPolicy: None` with `nameservers: [192.168.42.1]`
|
||||
- Woodpecker: `WOODPECKER_BACKEND_K8S_DNS_CONFIG` has `nameservers:[10.43.0.10]` + searches including `git-server.svc.cluster.local`
|
||||
|
||||
### Known Issues / TODO
|
||||
|
||||
1. **`stonks-oracle` namespace**: Scaled to 0 but still exists with stale data (42K extraction queue in Redis DB 0). Could be cleaned up or deleted entirely.
|
||||
|
||||
2. **Thesis Rewriter agent**: Was hammering vLLM from stonks-oracle namespace (5600+ calls/24h). Now stopped since namespace is scaled down. If it was also running in beta, check if recommendation service is calling vLLM for thesis rewrites excessively.
|
||||
|
||||
3. **`AxionML/Qwen3.5-9B-NVFP4` requests**: Something external is hitting vLLM with a model that doesn't exist (404s). Not from our pipeline — likely Open WebUI or another tool on the network configured with wrong model name. Source IP: goes through `vllm-metrics` nginx proxy (`10.42.1.155`).
|
||||
|
||||
4. **GitHub mirror**: `finalize.yml` mirror-github step fails (SSH key or DNS). Has `failure: ignore` so non-blocking. Needs `github_ssh_key` secret configured in Woodpecker.
|
||||
|
||||
5. **OpsPipeline dashboard**: Numbers now show time-filtered data. The "Document Stages" section shows counts from the selected time window (default 24h), with all-time totals as subtle subtitles. Currently beta shows: extracted=5417, low_quality=1659, parsed=15.
|
||||
|
||||
6. **Aggregation not generating trends on weekends**: Expected — market hours check prevents weekend trend generation. Will resume Monday.
|
||||
|
||||
7. **15 docs still in `parsed` status**: These are likely fresh ingests waiting for the next extraction cycle. Not stuck.
|
||||
|
||||
### Agent Performance (beta, last 24h as of session end)
|
||||
- Document Intelligence Extractor: 33 calls, 94% success, avg 11.4s, conf 0.794
|
||||
- Global Event Classifier: 81 calls, 99% success, avg 4.2s, conf 0.745
|
||||
- Thesis Rewriter: 5603 calls, 100% success, avg 2.5s (from stonks-oracle before shutdown)
|
||||
- Report Summarizer: 6 calls, 100% success, avg 6.9s
|
||||
|
||||
### Infrastructure
|
||||
- k3s cluster: 4 NixOS nodes (gremlin-1 through gremlin-4)
|
||||
- vLLM: `vllm-service` namespace, model `numind/NuExtract3`, 4070 Ti Super 16GB
|
||||
- CoreDNS: 4 replicas, `forward . 192.168.42.1`
|
||||
- Redis: DB 0 = stonks-oracle (stale), DB 1 = stonks-beta (active)
|
||||
- PostgreSQL: shared instance, both namespaces use same DB server (different databases? or same? — needs verification)
|
||||
- Gitea: `git-server` namespace, ClusterIP 10.43.73.77:3000, NodePort 30300
|
||||
- Woodpecker: `woodpecker` namespace, kubernetes backend, 2 agents
|
||||
|
||||
### Key Files Modified
|
||||
```
|
||||
services/scheduler/app.py — recovery thresholds + batch limit
|
||||
services/validation/prediction_snapshot.py — 24h price fallback
|
||||
services/aggregation/scoring.py — normalize_impact_scores()
|
||||
services/aggregation/worker.py — normalization integration
|
||||
services/trading/model_quality_gate.py — 48h threshold
|
||||
services/extractor/schemas.py — null field handling
|
||||
services/extractor/main.py — macro doc status update
|
||||
frontend/src/pages/OpsPipeline.tsx — dashboard fix
|
||||
scripts/backfill_snapshot_prices.py — new script
|
||||
tests/test_pbt_pipeline_health_*.py — PBT tests
|
||||
tests/test_extractor_schemas.py — updated test
|
||||
infra/helm/stonks-oracle/values*.yaml — signal-engine replicas
|
||||
.woodpecker/*.yml — ClusterIP clone fix
|
||||
```
|
||||
@@ -0,0 +1,613 @@
|
||||
# Observability and Metrics Reference
|
||||
|
||||
This document covers the full observability stack for Stonks Oracle: Prometheus metrics, operational alerting, structured logging, dead-letter queues, and recommended monitoring queries.
|
||||
|
||||
## Prometheus Metrics Endpoint
|
||||
|
||||
The Query API exposes a `/metrics` endpoint that returns all registered Prometheus metrics in the standard text exposition format.
|
||||
|
||||
**Endpoint**: `GET /metrics` on the Query API service (port 8000)
|
||||
|
||||
**Response**: `text/plain; version=0.0.4; charset=utf-8` — standard Prometheus scrape format via `prometheus_client.generate_latest()`.
|
||||
|
||||
### Prometheus Scrape Configuration
|
||||
|
||||
Add the following job to your `prometheus.yml`:
|
||||
|
||||
```yaml
|
||||
scrape_configs:
|
||||
- job_name: "stonks-oracle"
|
||||
scrape_interval: 15s
|
||||
scrape_timeout: 10s
|
||||
metrics_path: /metrics
|
||||
static_configs:
|
||||
- targets:
|
||||
# Docker Compose
|
||||
- "query-api:8000"
|
||||
# Kubernetes
|
||||
# - "query-api.stonks-oracle.svc.cluster.local:8000"
|
||||
```
|
||||
|
||||
For Kubernetes deployments, you can also use a `ServiceMonitor` resource if the Prometheus Operator is installed:
|
||||
|
||||
```yaml
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: ServiceMonitor
|
||||
metadata:
|
||||
name: stonks-oracle
|
||||
namespace: stonks-oracle
|
||||
spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: query-api
|
||||
endpoints:
|
||||
- port: http
|
||||
path: /metrics
|
||||
interval: 15s
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Prometheus Metrics Reference
|
||||
|
||||
All metrics are defined in `services/shared/metrics.py`. Metric names use the `stonks_` prefix.
|
||||
|
||||
### Service Info
|
||||
|
||||
| Metric | Type | Description |
|
||||
|--------|------|-------------|
|
||||
| `stonks_oracle_info` | Info | Service metadata (build version, etc.) |
|
||||
|
||||
### Ingestion Metrics
|
||||
|
||||
| Metric | Type | Labels | Description |
|
||||
|--------|------|--------|-------------|
|
||||
| `stonks_ingestion_jobs_total` | Counter | `source_type`, `status` | Total ingestion jobs processed |
|
||||
| `stonks_ingestion_items_fetched_total` | Counter | `source_type` | Total items fetched from external sources |
|
||||
| `stonks_ingestion_items_new_total` | Counter | `source_type` | New (non-duplicate) items ingested |
|
||||
| `stonks_ingestion_items_deduped_total` | Counter | `source_type` | Items skipped due to deduplication |
|
||||
| `stonks_ingestion_errors_total` | Counter | `source_type` | Ingestion errors by source type |
|
||||
| `stonks_ingestion_adapter_duration_seconds` | Histogram | `source_type` | Adapter fetch latency (buckets: 0.1, 0.5, 1, 2, 5, 10, 30, 60s) |
|
||||
|
||||
### Parsing Metrics
|
||||
|
||||
| Metric | Type | Labels | Description |
|
||||
|--------|------|--------|-------------|
|
||||
| `stonks_parse_jobs_total` | Counter | `status` | Total parse jobs processed |
|
||||
| `stonks_parse_quality_score` | Histogram | — | Distribution of parser quality scores (buckets: 0.1–1.0 in 0.1 steps) |
|
||||
| `stonks_parse_low_quality_total` | Counter | — | Documents flagged as low quality by the parser |
|
||||
| `stonks_parse_duration_seconds` | Histogram | — | Parse job duration (buckets: 0.05, 0.1, 0.25, 0.5, 1, 2, 5, 10s) |
|
||||
|
||||
### Extraction Metrics
|
||||
|
||||
| Metric | Type | Labels | Description |
|
||||
|--------|------|--------|-------------|
|
||||
| `stonks_extraction_jobs_total` | Counter | `status` | Total extraction jobs processed |
|
||||
| `stonks_extraction_attempts_total` | Counter | — | Total Ollama extraction attempts (including retries) |
|
||||
| `stonks_extraction_retries_total` | Counter | — | Extraction retry count |
|
||||
| `stonks_extraction_duration_seconds` | Histogram | — | Extraction total duration (buckets: 1, 2, 5, 10, 20, 30, 60, 120s) |
|
||||
| `stonks_extraction_confidence` | Histogram | — | Distribution of extraction confidence scores (buckets: 0.1–1.0) |
|
||||
| `stonks_extraction_validation_errors_total` | Counter | — | Total validation errors across extractions |
|
||||
| `stonks_extraction_tokens_total` | Counter | `direction` | Estimated token usage (labels: `input`, `output`) |
|
||||
|
||||
### Aggregation Metrics
|
||||
|
||||
| Metric | Type | Labels | Description |
|
||||
|--------|------|--------|-------------|
|
||||
| `stonks_aggregation_windows_total` | Counter | `window` | Trend windows computed |
|
||||
| `stonks_aggregation_signals_total` | Counter | `window` | Signals processed during aggregation |
|
||||
| `stonks_aggregation_contradiction_score` | Histogram | — | Distribution of contradiction scores in trend windows (buckets: 0.0–1.0) |
|
||||
| `stonks_aggregation_duration_seconds` | Histogram | `window` | Aggregation job duration (buckets: 0.05, 0.1, 0.25, 0.5, 1, 2, 5, 10s) |
|
||||
|
||||
### Recommendation Metrics
|
||||
|
||||
| Metric | Type | Labels | Description |
|
||||
|--------|------|--------|-------------|
|
||||
| `stonks_recommendations_total` | Counter | `action`, `mode` | Recommendations generated |
|
||||
| `stonks_recommendations_suppressed_total` | Counter | — | Recommendations suppressed due to low data quality |
|
||||
| `stonks_recommendation_confidence` | Histogram | — | Distribution of recommendation confidence scores (buckets: 0.1–1.0) |
|
||||
|
||||
### Lake Publication Metrics
|
||||
|
||||
| Metric | Type | Labels | Description |
|
||||
|--------|------|--------|-------------|
|
||||
| `stonks_lake_facts_published_total` | Counter | `table_name` | Analytical facts published to the lakehouse |
|
||||
| `stonks_lake_publish_duration_seconds` | Histogram | `table_name` | Lake publication write latency (buckets: 0.01, 0.05, 0.1, 0.25, 0.5, 1, 2, 5s) |
|
||||
| `stonks_lake_publish_errors_total` | Counter | `table_name` | Lake publication errors |
|
||||
| `stonks_lake_publish_bytes_total` | Counter | `table_name` | Total bytes written to the lakehouse |
|
||||
|
||||
### Trading and Broker Metrics
|
||||
|
||||
| Metric | Type | Labels | Description |
|
||||
|--------|------|--------|-------------|
|
||||
| `stonks_orders_submitted_total` | Counter | `side`, `order_type`, `mode` | Orders submitted to broker |
|
||||
| `stonks_orders_rejected_total` | Counter | `reason_category` | Orders rejected before broker submission |
|
||||
| `stonks_orders_filled_total` | Counter | `side` | Orders filled by broker |
|
||||
| `stonks_orders_duplicates_prevented_total` | Counter | `detected_via` | Duplicate orders prevented by idempotency checks |
|
||||
| `stonks_orders_clamped_total` | Counter | — | Orders auto-clamped to fit within position limits |
|
||||
| `stonks_risk_evaluations_total` | Counter | `result` | Risk evaluations performed |
|
||||
| `stonks_risk_check_failures_total` | Counter | `check_name` | Individual risk check failures |
|
||||
| `stonks_positions_synced_total` | Counter | — | Position sync operations completed |
|
||||
|
||||
### Alerting Metrics
|
||||
|
||||
| Metric | Type | Labels | Description |
|
||||
|--------|------|--------|-------------|
|
||||
| `stonks_alerts_fired_total` | Counter | `rule`, `severity` | Total alerts fired by rule |
|
||||
| `stonks_alerts_resolved_total` | Counter | `rule` | Total alerts resolved by rule |
|
||||
| `stonks_alert_check_duration_seconds` | Histogram | — | Duration of alert evaluation cycle (buckets: 0.01–5s) |
|
||||
| `stonks_alert_active` | Gauge | `rule` | Whether an alert rule is currently firing (1) or resolved (0) |
|
||||
|
||||
### Dead-Letter Queue Metrics
|
||||
|
||||
| Metric | Type | Labels | Description |
|
||||
|--------|------|--------|-------------|
|
||||
| `stonks_dlq_items_total` | Counter | `queue` | Jobs sent to dead-letter queues |
|
||||
| `stonks_dlq_replayed_total` | Counter | `queue` | Jobs replayed from dead-letter queues |
|
||||
| `stonks_dlq_depth` | Gauge | `queue` | Current dead-letter queue depth |
|
||||
|
||||
### Active Jobs Gauge
|
||||
|
||||
| Metric | Type | Labels | Description |
|
||||
|--------|------|--------|-------------|
|
||||
| `stonks_active_jobs` | Gauge | `stage` | Currently processing jobs by pipeline stage |
|
||||
|
||||
---
|
||||
|
||||
## Alerting Module
|
||||
|
||||
The alerting module (`services/shared/alerting.py`) evaluates four operational alert rules against PostgreSQL state on a configurable interval. When a threshold is breached, the module emits structured log events and increments Prometheus counters. When a previously firing alert clears, it logs a resolution event.
|
||||
|
||||
### Alert Rules
|
||||
|
||||
#### 1. `source_failures` — Sustained Source Retrieval Failures
|
||||
|
||||
Detects sources where the last N ingestion runs all failed within the lookback window.
|
||||
|
||||
| Parameter | ConfigMap Variable | Default | Description |
|
||||
|-----------|--------------------|---------|-------------|
|
||||
| Consecutive failure threshold | `ALERT_SOURCE_FAILURE_THRESHOLD` | `3` | Number of consecutive failures before alert fires |
|
||||
| Lookback window | `ALERT_SOURCE_FAILURE_WINDOW_HOURS` | `6` hours | How far back to check ingestion_runs |
|
||||
|
||||
**Severity**: `warning`
|
||||
|
||||
**Query**: Checks `ingestion_runs` for sources where the most recent N runs (within the window) all have `status = 'failed'`.
|
||||
|
||||
**Details emitted**: `source_id`, `source_type`, `source_name`, `ticker`, `consecutive_failures`
|
||||
|
||||
#### 2. `schema_failure_spike` — Extraction Validation Failure Rate
|
||||
|
||||
Detects when the extraction schema validation failure rate exceeds a threshold.
|
||||
|
||||
| Parameter | ConfigMap Variable | Default | Description |
|
||||
|-----------|--------------------|---------|-------------|
|
||||
| Failure rate threshold | `ALERT_SCHEMA_FAILURE_RATE_THRESHOLD` | `0.3` (30%) | Failure rate that triggers the alert |
|
||||
| Lookback window | `ALERT_SCHEMA_FAILURE_WINDOW_HOURS` | `1` hour | Window for computing failure rate |
|
||||
|
||||
**Severity**: `warning` if rate ≥ 30%, `critical` if rate ≥ 50%
|
||||
|
||||
**Query**: Computes `failed / total` from `model_performance_metrics` within the window.
|
||||
|
||||
**Details emitted**: `total_extractions`, `failed_extractions`, `failure_rate`, `threshold`, `window_hours`
|
||||
|
||||
#### 3. `analytical_lag` — Lake Publication Lag
|
||||
|
||||
Detects when lake publication has not completed within the threshold for any table.
|
||||
|
||||
| Parameter | ConfigMap Variable | Default | Description |
|
||||
|-----------|--------------------|---------|-------------|
|
||||
| Lag threshold | `ALERT_LAKE_LAG_THRESHOLD_MINUTES` | `60` minutes | Maximum acceptable time since last successful publish |
|
||||
|
||||
**Severity**: `warning`
|
||||
|
||||
**Query**: Checks `audit_events` for the most recent successful `lake_publish` event per table, alerts if any are older than the threshold.
|
||||
|
||||
**Details emitted**: `table_name`, `last_publish`, `lag_minutes`, `threshold_minutes`
|
||||
|
||||
#### 4. `broker_issues` — Consecutive Broker Errors
|
||||
|
||||
Detects consecutive broker submission errors (rejections, timeouts, connection failures).
|
||||
|
||||
| Parameter | ConfigMap Variable | Default | Description |
|
||||
|-----------|--------------------|---------|-------------|
|
||||
| Error threshold | `ALERT_BROKER_ERROR_THRESHOLD` | `3` | Consecutive broker errors before alert fires |
|
||||
| Lookback window | `ALERT_BROKER_ERROR_WINDOW_HOURS` | `1` hour | Window for checking order_events |
|
||||
|
||||
**Severity**: `critical`
|
||||
|
||||
**Query**: Counts recent `order_events` with `event_type IN ('broker_error', 'broker_timeout', 'connection_failed')`.
|
||||
|
||||
**Details emitted**: `error_count`, `threshold`, `window_hours`
|
||||
|
||||
### Evaluation Cycle
|
||||
|
||||
The alerting module runs on a configurable interval (default: every 120 seconds, controlled by `ALERT_CHECK_INTERVAL_SECONDS`). Each cycle:
|
||||
|
||||
1. Runs all four alert rules against PostgreSQL
|
||||
2. Compares results to the current `AlertState` to detect new firings and resolutions
|
||||
3. For new firings: increments `stonks_alerts_fired_total`, sets `stonks_alert_active` gauge to 1, logs a `WARNING`
|
||||
4. For resolutions: increments `stonks_alerts_resolved_total`, sets `stonks_alert_active` gauge to 0, logs an `INFO`
|
||||
5. Records the evaluation duration in `stonks_alert_check_duration_seconds`
|
||||
|
||||
Each rule check is wrapped in a try/except so a failure in one rule does not block the others.
|
||||
|
||||
### ConfigMap Variables Summary
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `ALERT_SOURCE_FAILURE_THRESHOLD` | `3` | Consecutive source failures before alert |
|
||||
| `ALERT_SOURCE_FAILURE_WINDOW_HOURS` | `6` | Source failure lookback window (hours) |
|
||||
| `ALERT_SCHEMA_FAILURE_RATE_THRESHOLD` | `0.3` | Extraction failure rate threshold (0.0–1.0) |
|
||||
| `ALERT_SCHEMA_FAILURE_WINDOW_HOURS` | `1` | Schema failure lookback window (hours) |
|
||||
| `ALERT_LAKE_LAG_THRESHOLD_MINUTES` | `60` | Max minutes since last lake publish |
|
||||
| `ALERT_BROKER_ERROR_THRESHOLD` | `3` | Consecutive broker errors before alert |
|
||||
| `ALERT_BROKER_ERROR_WINDOW_HOURS` | `1` | Broker error lookback window (hours) |
|
||||
| `ALERT_CHECK_INTERVAL_SECONDS` | `120` | Seconds between alert evaluation cycles |
|
||||
|
||||
---
|
||||
|
||||
## Structured Logging
|
||||
|
||||
All services use structured JSON logging configured via `services/shared/logging.py`. Call `setup_logging(service_name)` once at service startup.
|
||||
|
||||
### JSON Log Format
|
||||
|
||||
Each log line is a single JSON object with the following fields:
|
||||
|
||||
```json
|
||||
{
|
||||
"timestamp": "2025-01-15T12:34:56.789012+00:00",
|
||||
"level": "INFO",
|
||||
"logger": "ingestion_worker",
|
||||
"message": "Processed job for AAPL",
|
||||
"service": "ingestion_worker",
|
||||
"trace_id": "a1b2c3d4e5f67890",
|
||||
"span_id": "1a2b3c4d"
|
||||
}
|
||||
```
|
||||
|
||||
| Field | Type | Description |
|
||||
|-------|------|-------------|
|
||||
| `timestamp` | string (ISO 8601) | UTC timestamp of the log event |
|
||||
| `level` | string | Log level: `DEBUG`, `INFO`, `WARNING`, `ERROR`, `CRITICAL` |
|
||||
| `logger` | string | Python logger name |
|
||||
| `message` | string | Human-readable log message |
|
||||
| `service` | string | Service name set at startup (e.g., `ingestion_worker`, `scheduler`) |
|
||||
| `trace_id` | string | 16-character hex trace ID for distributed tracing |
|
||||
| `span_id` | string | 8-character hex span ID for the current operation |
|
||||
|
||||
### Additional Context Fields
|
||||
|
||||
When present, these fields are merged into the JSON output:
|
||||
|
||||
| Field | Source | Description |
|
||||
|-------|--------|-------------|
|
||||
| `span_operation` | `Span` context manager | Name of the traced operation |
|
||||
| `span_status` | `Span` context manager | `ok` or `error` |
|
||||
| `span_duration_ms` | `Span` context manager | Duration of the span in milliseconds |
|
||||
| `span_parent_id` | `Span` context manager | Parent span ID for nested spans |
|
||||
| `span_attributes` | `Span` context manager | Arbitrary key-value attributes set on the span |
|
||||
| `ticker` | Manual `extra={}` | Company ticker symbol |
|
||||
| `document_id` | Manual `extra={}` | Document UUID |
|
||||
| `source_type` | Manual `extra={}` | Source type (e.g., `polygon`, `news_api`) |
|
||||
| `job_id` | Manual `extra={}` | Job identifier |
|
||||
| `duration_ms` | Manual `extra={}` | Operation duration |
|
||||
| `error` | Manual `extra={}` | Error description |
|
||||
| `count` | Manual `extra={}` | Item count |
|
||||
| `exception` | Automatic | Formatted exception traceback (when `exc_info` is set) |
|
||||
|
||||
### Trace Context Propagation
|
||||
|
||||
Trace context flows through the pipeline via job payloads:
|
||||
|
||||
1. **Inject**: Before enqueuing a job to Redis, call `inject_trace_context(payload)` to add `_trace_id` to the payload dict.
|
||||
2. **Extract**: At the start of job processing, call `extract_trace_context(payload)` to restore the trace context (or generate a new one if absent).
|
||||
3. **Span**: Use the `Span` context manager to create child spans within a service:
|
||||
|
||||
```python
|
||||
from services.shared.logging import Span
|
||||
|
||||
with Span("process_document", ticker="AAPL") as span:
|
||||
# ... do work ...
|
||||
span.set_attribute("doc_count", 5)
|
||||
```
|
||||
|
||||
This produces a structured log entry on span exit with duration, status, and attributes.
|
||||
|
||||
### Log Querying
|
||||
|
||||
To trace a request through the pipeline, filter by `trace_id`:
|
||||
|
||||
```bash
|
||||
# Kubernetes — find all logs for a specific trace
|
||||
kubectl logs -n stonks-oracle -l app.kubernetes.io/part-of=stonks-oracle --all-containers \
|
||||
| jq -r 'select(.trace_id == "a1b2c3d4e5f67890")'
|
||||
|
||||
# Docker Compose — search across all services
|
||||
docker compose logs --no-color | grep '"trace_id":"a1b2c3d4e5f67890"'
|
||||
```
|
||||
|
||||
To find errors in a specific service:
|
||||
|
||||
```bash
|
||||
# Kubernetes
|
||||
kubectl logs -n stonks-oracle deployment/extractor --tail=500 \
|
||||
| jq 'select(.level == "ERROR")'
|
||||
|
||||
# Docker Compose
|
||||
docker compose logs extractor --no-color --tail=500 \
|
||||
| jq 'select(.level == "ERROR")'
|
||||
```
|
||||
|
||||
To find slow extraction spans:
|
||||
|
||||
```bash
|
||||
kubectl logs -n stonks-oracle deployment/extractor --tail=1000 \
|
||||
| jq 'select(.span_operation == "extract_document" and .span_duration_ms > 30000)'
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Dead-Letter Queue System
|
||||
|
||||
When a worker fails to process a job after exhausting retries (default: 3 attempts), the job is pushed to a per-queue dead-letter list in Redis. The DLQ system is implemented in `services/shared/dead_letter.py`.
|
||||
|
||||
### Queue Names
|
||||
|
||||
Dead-letter queues follow the naming pattern `stonks:dlq:<queue_name>`:
|
||||
|
||||
| DLQ Key | Source Queue | Description |
|
||||
|---------|-------------|-------------|
|
||||
| `stonks:dlq:ingestion` | `stonks:queue:ingestion` | Failed ingestion jobs (adapter errors, API failures) |
|
||||
| `stonks:dlq:parsing` | `stonks:queue:parsing` | Failed parse jobs |
|
||||
| `stonks:dlq:extraction` | `stonks:queue:extraction` | Failed extraction jobs (LLM errors, validation failures) |
|
||||
| `stonks:dlq:aggregation` | `stonks:queue:aggregation` | Failed aggregation jobs |
|
||||
| `stonks:dlq:recommendation` | `stonks:queue:recommendation` | Failed recommendation jobs |
|
||||
| `stonks:dlq:broker_orders` | `stonks:queue:broker_orders` | Failed broker order submissions |
|
||||
|
||||
When `DEPLOY_STAGE` is set, the prefix becomes `stonks:<stage>:dlq:<queue_name>`.
|
||||
|
||||
### DLQ Entry Format
|
||||
|
||||
Each DLQ entry wraps the original job payload with failure metadata:
|
||||
|
||||
```json
|
||||
{
|
||||
"original_payload": {
|
||||
"source_id": "...",
|
||||
"source_type": "polygon",
|
||||
"ticker": "AAPL",
|
||||
"company_id": "...",
|
||||
"config": {}
|
||||
},
|
||||
"queue": "ingestion",
|
||||
"error": "ConnectionError: API timeout after 30s",
|
||||
"attempt": 3,
|
||||
"worker": "ingestion_worker",
|
||||
"dead_lettered_at": "2025-01-15T12:34:56.789012+00:00"
|
||||
}
|
||||
```
|
||||
|
||||
| Field | Type | Description |
|
||||
|-------|------|-------------|
|
||||
| `original_payload` | object | The original job payload as it was enqueued |
|
||||
| `queue` | string | Source queue name |
|
||||
| `error` | string | Error message from the final failed attempt |
|
||||
| `attempt` | integer | Number of attempts made before dead-lettering |
|
||||
| `worker` | string | Worker identifier that dead-lettered the job |
|
||||
| `dead_lettered_at` | string (ISO 8601) | UTC timestamp when the job was dead-lettered |
|
||||
|
||||
### Routing
|
||||
|
||||
Jobs are routed to the DLQ by calling `send_to_dlq()` from worker code after retry exhaustion:
|
||||
|
||||
```python
|
||||
from services.shared.dead_letter import send_to_dlq
|
||||
|
||||
await send_to_dlq(
|
||||
rds=redis_client,
|
||||
queue_name="ingestion",
|
||||
original_payload=job,
|
||||
error=str(exception),
|
||||
attempt=3,
|
||||
worker="ingestion_worker",
|
||||
)
|
||||
```
|
||||
|
||||
The default maximum attempts before dead-lettering is `DEFAULT_MAX_ATTEMPTS = 3`.
|
||||
|
||||
### Replay Tooling
|
||||
|
||||
The `services/shared/dead_letter.py` module provides functions for inspecting and replaying DLQ items:
|
||||
|
||||
| Function | Description |
|
||||
|----------|-------------|
|
||||
| `peek_dlq(rds, queue_name, start=0, count=10)` | Inspect DLQ entries without removing them |
|
||||
| `replay_one(rds, queue_name)` | Pop the oldest DLQ entry and re-enqueue its original payload to the source queue |
|
||||
| `replay_all(rds, queue_name)` | Replay every item in the DLQ back to the source queue. Returns the count replayed |
|
||||
| `dlq_length(rds, queue_name)` | Return the number of items in the DLQ |
|
||||
| `dlq_summary(rds, queue_names)` | Return a mapping of queue_name → DLQ depth for multiple queues |
|
||||
| `purge_dlq(rds, queue_name)` | Delete all items from the DLQ. Returns count removed |
|
||||
|
||||
### Monitoring DLQ Depth
|
||||
|
||||
Use the `scripts/check_queues.py` script to inspect queue and DLQ depths from the command line:
|
||||
|
||||
```bash
|
||||
# Docker Compose
|
||||
REDIS_HOST=localhost REDIS_PORT=6379 REDIS_PASSWORD="" \
|
||||
python scripts/check_queues.py
|
||||
|
||||
# Kubernetes
|
||||
kubectl exec -n stonks-oracle deployment/query-api -- \
|
||||
python scripts/check_queues.py
|
||||
```
|
||||
|
||||
The Query API also exposes DLQ depths in the `/api/ops/pipeline/stream` SSE endpoint and the DevOps metrics endpoints, reporting `dlq:<queue_name>` keys alongside regular queue depths.
|
||||
|
||||
The `stonks_dlq_depth` Prometheus gauge tracks DLQ depth per queue for dashboard alerting.
|
||||
|
||||
---
|
||||
|
||||
## Recommended Prometheus/Grafana Queries
|
||||
|
||||
### Ingestion Throughput
|
||||
|
||||
```promql
|
||||
# Ingestion jobs per minute by source type and status
|
||||
sum(rate(stonks_ingestion_jobs_total[5m])) by (source_type, status) * 60
|
||||
|
||||
# New items ingested per minute
|
||||
sum(rate(stonks_ingestion_items_new_total[5m])) * 60
|
||||
|
||||
# Deduplication ratio (higher = more duplicates being filtered)
|
||||
sum(rate(stonks_ingestion_items_deduped_total[5m]))
|
||||
/ sum(rate(stonks_ingestion_items_fetched_total[5m]))
|
||||
|
||||
# Adapter latency p95 by source type
|
||||
histogram_quantile(0.95, sum(rate(stonks_ingestion_adapter_duration_seconds_bucket[5m])) by (le, source_type))
|
||||
|
||||
# Ingestion error rate
|
||||
sum(rate(stonks_ingestion_errors_total[5m])) by (source_type)
|
||||
```
|
||||
|
||||
### Extraction Latency and Quality
|
||||
|
||||
```promql
|
||||
# Extraction duration p50 and p95
|
||||
histogram_quantile(0.5, sum(rate(stonks_extraction_duration_seconds_bucket[5m])) by (le))
|
||||
histogram_quantile(0.95, sum(rate(stonks_extraction_duration_seconds_bucket[5m])) by (le))
|
||||
|
||||
# Extraction success rate
|
||||
sum(rate(stonks_extraction_jobs_total{status="success"}[5m]))
|
||||
/ sum(rate(stonks_extraction_jobs_total[5m]))
|
||||
|
||||
# Average extraction confidence
|
||||
histogram_quantile(0.5, sum(rate(stonks_extraction_confidence_bucket[5m])) by (le))
|
||||
|
||||
# Validation error rate
|
||||
sum(rate(stonks_extraction_validation_errors_total[5m]))
|
||||
|
||||
# Token usage rate (input vs output)
|
||||
sum(rate(stonks_extraction_tokens_total[5m])) by (direction)
|
||||
```
|
||||
|
||||
### Aggregation Volume
|
||||
|
||||
```promql
|
||||
# Trend windows computed per minute by window size
|
||||
sum(rate(stonks_aggregation_windows_total[5m])) by (window) * 60
|
||||
|
||||
# Signals processed per minute
|
||||
sum(rate(stonks_aggregation_signals_total[5m])) by (window) * 60
|
||||
|
||||
# Average contradiction score (higher = more conflicting signals)
|
||||
histogram_quantile(0.5, sum(rate(stonks_aggregation_contradiction_score_bucket[5m])) by (le))
|
||||
|
||||
# Aggregation duration p95
|
||||
histogram_quantile(0.95, sum(rate(stonks_aggregation_duration_seconds_bucket[5m])) by (le, window))
|
||||
```
|
||||
|
||||
### Recommendation Generation
|
||||
|
||||
```promql
|
||||
# Recommendations generated per minute by action
|
||||
sum(rate(stonks_recommendations_total[5m])) by (action, mode) * 60
|
||||
|
||||
# Suppression rate
|
||||
sum(rate(stonks_recommendations_suppressed_total[5m]))
|
||||
/ sum(rate(stonks_recommendations_total[5m]))
|
||||
|
||||
# Recommendation confidence distribution
|
||||
histogram_quantile(0.5, sum(rate(stonks_recommendation_confidence_bucket[5m])) by (le))
|
||||
```
|
||||
|
||||
### Trading Engine Activity
|
||||
|
||||
```promql
|
||||
# Orders submitted per minute by side
|
||||
sum(rate(stonks_orders_submitted_total[5m])) by (side, mode) * 60
|
||||
|
||||
# Order rejection rate by reason
|
||||
sum(rate(stonks_orders_rejected_total[5m])) by (reason_category)
|
||||
|
||||
# Fill rate
|
||||
sum(rate(stonks_orders_filled_total[5m]))
|
||||
/ sum(rate(stonks_orders_submitted_total[5m]))
|
||||
|
||||
# Duplicate orders prevented
|
||||
sum(rate(stonks_orders_duplicates_prevented_total[5m])) by (detected_via)
|
||||
|
||||
# Risk evaluation outcomes
|
||||
sum(rate(stonks_risk_evaluations_total[5m])) by (result)
|
||||
|
||||
# Risk check failure breakdown
|
||||
sum(rate(stonks_risk_check_failures_total[5m])) by (check_name)
|
||||
```
|
||||
|
||||
### Lake Publication
|
||||
|
||||
```promql
|
||||
# Facts published per minute by table
|
||||
sum(rate(stonks_lake_facts_published_total[5m])) by (table_name) * 60
|
||||
|
||||
# Write latency p95 by table
|
||||
histogram_quantile(0.95, sum(rate(stonks_lake_publish_duration_seconds_bucket[5m])) by (le, table_name))
|
||||
|
||||
# Publication error rate
|
||||
sum(rate(stonks_lake_publish_errors_total[5m])) by (table_name)
|
||||
|
||||
# Bytes written per minute
|
||||
sum(rate(stonks_lake_publish_bytes_total[5m])) by (table_name) * 60
|
||||
```
|
||||
|
||||
### Alerting Health
|
||||
|
||||
```promql
|
||||
# Currently active alerts by rule
|
||||
stonks_alert_active
|
||||
|
||||
# Alert firing rate
|
||||
sum(rate(stonks_alerts_fired_total[1h])) by (rule, severity)
|
||||
|
||||
# Alert evaluation duration
|
||||
histogram_quantile(0.95, sum(rate(stonks_alert_check_duration_seconds_bucket[5m])) by (le))
|
||||
```
|
||||
|
||||
### Dead-Letter Queue Health
|
||||
|
||||
```promql
|
||||
# Current DLQ depth by queue
|
||||
stonks_dlq_depth
|
||||
|
||||
# DLQ inflow rate (jobs dead-lettered per minute)
|
||||
sum(rate(stonks_dlq_items_total[5m])) by (queue) * 60
|
||||
|
||||
# DLQ replay rate
|
||||
sum(rate(stonks_dlq_replayed_total[5m])) by (queue) * 60
|
||||
```
|
||||
|
||||
### Pipeline Overview (Active Jobs)
|
||||
|
||||
```promql
|
||||
# Currently active jobs by pipeline stage
|
||||
stonks_active_jobs
|
||||
|
||||
# Parse quality score distribution
|
||||
histogram_quantile(0.5, sum(rate(stonks_parse_quality_score_bucket[5m])) by (le))
|
||||
|
||||
# Low quality document rate
|
||||
sum(rate(stonks_parse_low_quality_total[5m]))
|
||||
/ sum(rate(stonks_parse_jobs_total[5m]))
|
||||
```
|
||||
|
||||
### Recommended Grafana Alert Rules
|
||||
|
||||
| Alert | Expression | For | Severity |
|
||||
|-------|-----------|-----|----------|
|
||||
| High DLQ depth | `stonks_dlq_depth > 10` | 5m | warning |
|
||||
| Ingestion error spike | `sum(rate(stonks_ingestion_errors_total[5m])) > 0.5` | 5m | warning |
|
||||
| Extraction latency high | `histogram_quantile(0.95, sum(rate(stonks_extraction_duration_seconds_bucket[5m])) by (le)) > 60` | 10m | warning |
|
||||
| Lake publication stale | `stonks_alert_active{rule="analytical_lag"} == 1` | 5m | warning |
|
||||
| Broker errors active | `stonks_alert_active{rule="broker_issues"} == 1` | 1m | critical |
|
||||
| Zero ingestion throughput | `sum(rate(stonks_ingestion_jobs_total[15m])) == 0` | 15m | critical |
|
||||
@@ -0,0 +1,144 @@
|
||||
# Stonks Oracle — What It Is and What It Does
|
||||
|
||||
## The One-Liner
|
||||
|
||||
Stonks Oracle is an autonomous market intelligence system that reads the news so you don't have to, forms a view on 50 publicly traded companies, and paper-trades that view — then grades its own homework.
|
||||
|
||||
---
|
||||
|
||||
## The Problem It Solves
|
||||
|
||||
Markets are noisy. Every day, hundreds of news articles, SEC filings, earnings transcripts, and geopolitical headlines hit the wire. A human analyst covering even a dozen names struggles to weigh all of it in real time. Most retail and even some institutional desks end up reacting to headlines rather than synthesizing the full picture.
|
||||
|
||||
Stonks Oracle replaces that manual synthesis with an always-on pipeline:
|
||||
|
||||
1. **It reads everything.** News articles, 10-K/10-Q filings, earnings calls, press releases, and macro/geopolitical headlines — ingested automatically on a schedule.
|
||||
2. **It extracts structured intelligence.** A local AI model reads each document and pulls out: which companies are mentioned, the sentiment (bullish / bearish / neutral), the catalyst type (earnings, product launch, regulatory action, M&A, etc.), impact horizon (same-day through 90 days), key facts, and material risks.
|
||||
3. **It forms a view.** Those individual extractions are aggregated into rolling trend summaries per company, refreshed continuously. The system flags contradictions (e.g., one filing is bullish but a news article is bearish) and tracks confidence based on evidence depth.
|
||||
4. **It decides whether to trade.** When confidence is high enough, contradiction is low, and evidence is fresh, it issues a buy or sell recommendation — with a full written thesis explaining why.
|
||||
5. **It executes paper trades.** An autonomous trading engine places orders through Alpaca's paper-trading system. Position sizing, stop-losses, take-profits, sector concentration limits, and circuit breakers are all built in.
|
||||
6. **It measures itself.** Every prediction is frozen at the moment it's made, then checked against actual price movements days and weeks later. The system tracks its own win rate, calibration, and whether it's beating SPY.
|
||||
|
||||
---
|
||||
|
||||
## The Universe
|
||||
|
||||
50 companies across 10 sectors:
|
||||
|
||||
| Sector | Examples |
|
||||
|--------|----------|
|
||||
| Technology | AAPL, MSFT, NVDA, GOOGL, META |
|
||||
| Consumer Cyclical | AMZN, TSLA, NKE, SBUX |
|
||||
| Financial Services | JPM, GS, V, MA |
|
||||
| Healthcare | JNJ, UNH, PFE, LLY |
|
||||
| Energy | XOM, CVX, COP |
|
||||
| Communication Services | NFLX, DIS, T |
|
||||
| Industrials | CAT, BA, UPS |
|
||||
| Consumer Defensive | PG, KO, WMT |
|
||||
| Real Estate | AMT, PLD |
|
||||
| Utilities | NEE, DUK |
|
||||
|
||||
46 competitor relationships are defined (direct rivals, same-sector peers, overlapping products, supply chain adjacencies) so the system can propagate signals — e.g., if a semiconductor shortage hits one chipmaker, the system assesses exposure for its competitors and supply chain partners.
|
||||
|
||||
---
|
||||
|
||||
## The Three Signal Layers
|
||||
|
||||
Think of these as three analysts sitting at the same desk, each watching a different feed:
|
||||
|
||||
### Layer 1 — Company-Specific Intelligence
|
||||
|
||||
The bread and butter. Every news article and filing about a specific company gets scored for sentiment, impact magnitude, and time horizon. These signals are weighted by recency (yesterday's earnings matter more than last month's), source credibility, and novelty (the fifth article repeating the same news adds less information than the first).
|
||||
|
||||
Trend summaries roll up across five windows: intraday, 1 day, 7 days, 30 days, and 90 days — giving both a "what's happening right now" and a "what's the longer arc" view.
|
||||
|
||||
### Layer 2 — Macro & Geopolitical
|
||||
|
||||
Global events (trade wars, rate decisions, geopolitical crises, commodity shocks) are classified by impact type and severity. Each company has an exposure profile — geographic revenue mix, supply chain regions, commodity dependencies — that maps macro events down to company-level impact scores.
|
||||
|
||||
A tariff announcement on Chinese imports doesn't affect all 50 companies equally. Apple with its Chinese manufacturing exposure gets a higher impact score than Procter & Gamble with largely domestic supply chains.
|
||||
|
||||
### Layer 3 — Competitive & Historical Patterns
|
||||
|
||||
The system mines its own history: when this type of catalyst (say, an earnings beat) happened to this company in the past, what happened to the stock? What happened to its competitors? If NVIDIA reports a blowout quarter, does AMD tend to sell off or rally in sympathy?
|
||||
|
||||
This layer also tracks major corporate actions (M&A, restructurings, leadership changes) and propagates their implications across the competitive web.
|
||||
|
||||
**Safety rule:** The system never trades on macro or competitive signals alone. If there's no company-specific evidence supporting the thesis, the recommendation is downgraded to informational only.
|
||||
|
||||
---
|
||||
|
||||
## How a Trade Happens
|
||||
|
||||
Here's the chain from "news article published" to "paper order placed":
|
||||
|
||||
1. **Ingestion** — The article is fetched, deduplicated, and stored.
|
||||
2. **Parsing** — Raw HTML is cleaned, boilerplate is stripped, quality is scored.
|
||||
3. **Extraction** — The AI model reads the cleaned text and produces structured JSON: tickers mentioned, sentiment, catalysts, key facts, risks.
|
||||
4. **Aggregation** — The new extraction is merged into rolling trend summaries for each mentioned company. Confidence, contradiction, and evidence depth are recalculated.
|
||||
5. **Recommendation** — If the trend passes quality filters (enough evidence, high enough confidence, low enough contradiction, not stale), a BUY or SELL recommendation is generated with a written thesis.
|
||||
6. **Risk checks** — The trading engine asks: Is the circuit breaker tripped? Is the market open? Do I already have too many positions? Is this sector already overweight? Are earnings in the next 48 hours?
|
||||
7. **Position sizing** — Dollar amount is computed from confidence, portfolio heat, and the current risk tier (conservative / moderate / aggressive — auto-adjusted based on trailing performance).
|
||||
8. **Execution** — The order goes to Alpaca's paper-trading API. Stop-loss and take-profit levels are set automatically based on the stock's recent volatility.
|
||||
9. **Monitoring** — Open positions are tracked with trailing stops. If a position declines past its stop, it's closed. If it hits the take-profit target, it's closed.
|
||||
10. **Scoring** — Days later, the prediction is evaluated against the actual price move. Did the call go the right way? Did the confidence track reality?
|
||||
|
||||
---
|
||||
|
||||
## Risk Management (Built In, Not Bolted On)
|
||||
|
||||
- **Circuit breakers** — If daily losses exceed a threshold or a single position loses too much, all trading halts automatically.
|
||||
- **Position caps** — No single position can consume more than a set percentage of the portfolio.
|
||||
- **Sector concentration limits** — The system won't pile into one sector even if all signals are bullish.
|
||||
- **Correlation awareness** — New positions are rejected if they'd push portfolio correlation too high.
|
||||
- **Earnings blackout** — Position sizes are reduced or skipped entirely within 48 hours of an earnings announcement.
|
||||
- **Reserve pool** — Profits are partially siphoned into an emergency liquidity reserve.
|
||||
- **Risk tier auto-adjustment** — The system evaluates its own Sharpe ratio, drawdown, and win rate daily and shifts between conservative, moderate, and aggressive modes.
|
||||
|
||||
---
|
||||
|
||||
## Self-Grading: The Validation Loop
|
||||
|
||||
Most trading systems tell you their view. Few systematically check whether that view was right.
|
||||
|
||||
Stonks Oracle captures every prediction as an immutable snapshot — the thesis, the confidence, the price at the time, the evidence cited. Then it waits. After the prediction's time horizon elapses (1 day, 7 days, 30 days), it compares the predicted direction against the actual price movement and computes:
|
||||
|
||||
- **Win rate** — What fraction of directional calls were correct?
|
||||
- **Calibration** — When the system says "70% confident bullish," does the stock actually go up ~70% of the time? (If it only goes up 50% of the time, the system is overconfident.)
|
||||
- **Information coefficient** — Does the system's score have any linear correlation with actual returns?
|
||||
- **Excess return vs. SPY** — Is it adding alpha, or would you be better off in an index fund?
|
||||
- **Source attribution** — Which news sources and signal types actually contribute to correct predictions? Which are noise?
|
||||
|
||||
If model quality drops below defined thresholds, a safety gate prevents the system from upgrading recommendations from "informational" to "paper eligible" — it forces itself to the sidelines until accuracy recovers.
|
||||
|
||||
---
|
||||
|
||||
## The Dashboard
|
||||
|
||||
A web-based interface lets you see everything the system sees:
|
||||
|
||||
- **Home** — Portfolio value, daily P&L, risk tier, active alerts.
|
||||
- **Companies** — The tracked universe with current trend summaries and signal strength.
|
||||
- **Documents** — Every ingested article and filing, with the AI's structured extraction visible.
|
||||
- **Trends** — Per-company trend charts across all time windows, with evidence chains you can click through.
|
||||
- **Recommendations** — Active and historical recommendations with full theses and risk classifications.
|
||||
- **Trading** — The engine's status: open positions, reserve pool, circuit breaker state, portfolio heat map.
|
||||
- **Orders & Positions** — Full trade blotter with execution details.
|
||||
- **Macro Events** — Global event timeline showing what the system is tracking at the geopolitical level.
|
||||
- **Reports** — AI-generated daily and weekly performance summaries.
|
||||
- **Model Performance** — Calibration curves, win rate trends, source reliability scores.
|
||||
- **SQL Explorer** — Ad-hoc queries against the full analytical data warehouse, with a chart builder.
|
||||
|
||||
---
|
||||
|
||||
## What It Is Not
|
||||
|
||||
- **Not a live trading system (yet).** All trades are paper trades through Alpaca's sandbox. The architecture supports live execution, but safety gates and validation must demonstrate consistent edge before real money is at risk.
|
||||
- **Not a black box.** Every recommendation includes a full thesis, every trade has a decision trace, every prediction links back to the specific evidence that drove it.
|
||||
- **Not a prediction guarantee.** Markets are hard. The system's value is in disciplined synthesis, consistent process, and honest self-measurement — not in claiming to always be right.
|
||||
|
||||
---
|
||||
|
||||
## Where It's Headed
|
||||
|
||||
Active development is upgrading the signal math from rule-based heuristics to probabilistic Bayesian inference — running both approaches in parallel, comparing their verdicts, and using the disagreements as training signals for continuous improvement. The goal is a system that not only reads the market but learns from its own track record which types of evidence, in which market regimes, actually predict future price moves.
|
||||
@@ -0,0 +1,130 @@
|
||||
# Page 1 — Data Ingestion and Preparation
|
||||
|
||||
Every signal that the platform eventually acts on begins its life as raw data pulled from an external source. Before any AI agent can extract structured intelligence, before any trend can accumulate, and before any decision can be executed, the platform must first discover new content, fetch it reliably, eliminate duplicates, store the raw artifacts for audit, and normalize the text into a form suitable for downstream processing. This page traces that journey from external API to parser output, covering the Scheduler, Ingestion Worker, deduplication layer, raw storage, and Parser in detail.
|
||||
|
||||
For a visual overview of the full flow described here, see the [Ingestion to Extraction Flow diagram](diagrams/ingestion-to-extraction-flow.md).
|
||||
|
||||
---
|
||||
|
||||
## Four Categories of Input Data
|
||||
|
||||
The platform tracks 50 entities across 10 sectors, and it draws intelligence from four distinct categories of external data. Each category has its own adapter, its own API conventions, and its own scheduling cadence, but all of them feed into the same ingestion pipeline.
|
||||
|
||||
The first category is **entity news**, sourced from the external data provider's news endpoint (`/v2/reference/news`). The `ExternalNewsAdapter` in `services/adapters/news_adapter.py` fetches articles linked to a specific entity identifier, returning structured results that include title, publisher, article URL, description, keywords, and publication timestamp. Each request can return up to 1,000 articles, though the default limit is 20 per fetch. The adapter tracks the most recent `published_utc` value and uses it on subsequent fetches to avoid re-retrieving articles the system has already seen.
|
||||
|
||||
The second category is **regulatory filings**, sourced from the public records API full-text search system (regulatory filings source). The `RegulatoryFilingsAdapter` in `services/adapters/filings_adapter.py` queries the `/LATEST/search-index` endpoint for regulatory filing types and other form types associated with an entity's identifier or CIK number. Unlike the external data provider endpoints, the public records API requires no key — only a descriptive `User-Agent` header per the API's fair-access policy. The adapter deduplicates results by accession number (`adsh`), filters out non-primary documents like XML fragments and graphics, and constructs the public records API filing index URL for each hit so downstream services can fetch the full document.
|
||||
|
||||
The third category is **data feeds**, also sourced from the external data provider. The `ExternalDataAdapter` in `services/adapters/market_adapter.py` supports multiple endpoints: previous-day aggregate bars (`/v2/aggs/ticker/{ticker}/prev`), range bars for custom date windows, intraday hourly bars, grouped daily bars that return data for all entities in a single call (`/v2/aggs/grouped/locale/us/market/stocks/{date}`), and entity detail lookups. Data feeds follow a different path than textual content — they do not pass through the Parser or Extractor, since the structured numeric data is already in a usable form.
|
||||
|
||||
The fourth category is **macro and geopolitical news**, fetched by the `MacroNewsAdapter` in `services/adapters/macro_news_adapter.py`. Unlike the other three categories, macro news is not entity-specific. These sources have `source_type='macro_news'` in the `sources` database table and may have a `NULL` `company_id`. The adapter fetches from a configurable HTTP endpoint (typically the external data provider's news API filtered for broad topics) and returns articles that describe global events — policy shifts, central bank decisions, geopolitical conflicts — rather than entity-specific developments. Macro news articles are eventually classified by the Global Event Classifier agent and routed through a separate queue, as described in [Page 2](02-ai-agent-processing-and-extraction.md).
|
||||
|
||||
All four adapter classes inherit from `BaseAdapter` defined in `services/adapters/base.py` and return an `AdapterResult` dataclass containing the raw payload bytes, a SHA-256 content hash, a list of parsed item dicts, HTTP metadata (status code, response time), and an error field that is `None` on success. This uniform interface allows the Ingestion Worker to handle all source types through a single dispatch mechanism.
|
||||
|
||||
---
|
||||
|
||||
## The Scheduler: Orchestrating Ingestion Cycles
|
||||
|
||||
The Scheduler (`services/scheduler/app.py`) is the heartbeat of the ingestion pipeline. It runs a continuous loop that ticks every 15 seconds (`SCHEDULER_TICK = 15`), and on each tick it evaluates which sources are due for their next fetch. The Scheduler does not fetch data itself — it enqueues jobs onto the `app:queue:ingestion` Redis list for the Ingestion Worker to process.
|
||||
|
||||
Each source type has a default polling cadence defined in the `DEFAULT_CADENCES` dictionary:
|
||||
|
||||
| Source Type | Default Cadence |
|
||||
|------------------|-----------------|
|
||||
| `market_api` | 300 seconds |
|
||||
| `news_api` | 300 seconds |
|
||||
| `filings_api` | 3,600 seconds |
|
||||
| `macro_news` | 600 seconds |
|
||||
| `web_scrape` | 1,800 seconds |
|
||||
| `execution_api` | 30 seconds |
|
||||
|
||||
Individual sources can override their cadence via the `polling_interval_seconds` field in their `config` JSONB column in the `sources` table. The `get_cadence_for_source()` function checks for this override first, falling back to the default if none is set, and enforces a minimum interval of 10 seconds.
|
||||
|
||||
The Scheduler determines whether a source is due by calling `is_source_due()`, which considers several conditions. If a source has never run before (no entry in the `ingestion_runs` table), it is immediately due. If the last run failed, the Scheduler respects an exponential backoff computed by `compute_backoff()`: the delay starts at 60 seconds (`DEFAULT_BACKOFF_BASE`) and doubles with each retry up to a maximum of 3,600 seconds (`MAX_BACKOFF`). If a source has failed 10 consecutive times (`MAX_RETRY_COUNT`), the Scheduler stops scheduling it entirely until an operator manually resets the retry state. If the last run is still marked as `running`, the source is skipped to prevent double-scheduling. Otherwise, the Scheduler checks whether enough time has elapsed since the last completed run based on the source's cadence.
|
||||
|
||||
Rate limiting adds another layer of protection. The `check_rate_limit()` function enforces two constraints. First, each source type has a per-type limit defined in `DEFAULT_RATE_LIMITS` — for example, `market_api` and `news_api` are each capped at 20 requests per minute, while `filings_api` and `macro_news` are capped at 10. Second, because `market_api` and `news_api` both use the same external data provider API key, a global provider rate limit of 45 requests per minute (`PROVIDER_GLOBAL_RATE_LIMIT`) is enforced across both types combined. Rate limit state is tracked in Redis using keys of the form `app:ratelimit:{source_type}:{window}`, where the window is a minute-granularity timestamp. If a source type exceeds its limit, the Scheduler logs a warning and skips that source for the current tick.
|
||||
|
||||
The Scheduler handles three categories of sources in each cycle. First, it fetches all active entity-specific sources (excluding `macro_news`) by joining the `sources` and `companies` tables. Second, it fetches active macro news sources separately, since these may not have a `company_id`. Third, it fetches global data sources — those with `source_type='market_api'` and `company_id IS NULL` — which represent endpoints like the grouped daily bars that return data for all entities in a single API call. For intraday bar sources, the Scheduler expands a single global source into per-entity jobs for every active entity.
|
||||
|
||||
Each enqueued job payload includes the `source_id`, `company_id`, `ticker`, `legal_name`, `source_type`, `source_name`, `config`, `credibility_score`, a list of company `aliases` (fetched from the `company_aliases` table), and a `scheduled_at` timestamp. The job is pushed onto `app:queue:ingestion` via Redis `RPUSH`.
|
||||
|
||||
Beyond scheduling, the Scheduler also performs periodic maintenance. Every ~20 cycles (~5 minutes), it runs `recover_stale_documents()` to re-enqueue documents that have been stuck in `parsed` status for longer than 240 minutes — a safety net for cases where Redis loses queue entries due to pod restarts or OOM events. Every ~40 cycles (~10 minutes), it runs `retry_failed_extractions()` to give documents in `extraction_failed` status another chance, resetting them to `parsed` and deleting the failed `document_intelligence` row so the Extractor treats them as fresh. Every ~100 cycles (~25 minutes), it runs `cleanup_all_tables()` to enforce retention policies across tables like `competitive_signal_records` (30 days), `ingestion_runs` (14 days), and `execution_decisions` (90 days).
|
||||
|
||||
For more detail on the Scheduler's configuration and operational behavior, see the [Services Reference](../services.md).
|
||||
|
||||
---
|
||||
|
||||
## The Ingestion Worker: Adapter Dispatch and Persistence
|
||||
|
||||
The Ingestion Worker (`services/ingestion/worker.py`) is a long-running process that continuously pops jobs from the `app:queue:ingestion` Redis list and processes them. On startup, it initializes one instance of each adapter class and stores them in a dispatch dictionary keyed by `source_type`:
|
||||
|
||||
```
|
||||
adapters = {
|
||||
"market_api": ExternalDataAdapter(...),
|
||||
"news_api": ExternalNewsAdapter(...),
|
||||
"filings_api": RegulatoryFilingsAdapter(),
|
||||
"web_scrape": WebScrapeAdapter(),
|
||||
"execution_api": ExecutionAdapter(...),
|
||||
"macro_news": MacroNewsAdapter(...),
|
||||
}
|
||||
```
|
||||
|
||||
When a job arrives, the `process_job()` function looks up the appropriate adapter by `source_type` and calls its `fetch()` method with the ticker and source config. Before fetching, it records a new row in the `ingestion_runs` table with status `running`. If the adapter returns an error, the worker calls `record_retrieval_failure()` to update the run status and increment the source's retry counter with exponential backoff timing.
|
||||
|
||||
On a successful fetch, the worker performs several steps in sequence. First, it uploads the raw payload to MinIO via `upload_raw_artifact()` in `services/shared/storage.py`. The target bucket is determined by the source type through the `SOURCE_BUCKET_MAP`: `market_api` payloads go to `app-raw-data`, `news_api` and `macro_news` payloads go to `app-raw-content`, and `filings_api` payloads go to `app-raw-filings`. Objects are stored under a path that encodes the source type, entity identifier, date hierarchy, and document ID — for example, `news_api/Entity-A/2025/01/15/{run_id}/raw.json`.
|
||||
|
||||
---
|
||||
|
||||
## Content Deduplication via Redis
|
||||
|
||||
After storing the raw artifact, the Ingestion Worker checks for duplicate content. Deduplication operates at two levels.
|
||||
|
||||
At the payload level, the worker checks the overall `content_hash` (a SHA-256 digest of the raw API response) against Redis. The key pattern is `app:dedupe:{content_hash}` with a 24-hour TTL (86,400 seconds). If the hash is already present, the entire payload is skipped — the `ingestion_runs` row is marked as completed with `items_new=0`, and no downstream jobs are enqueued. If the hash is new, the worker sets the marker in Redis so future fetches of identical content are caught.
|
||||
|
||||
At the individual item level, for source types other than `market_api` and `execution_api`, the worker calls `dedupe_items()` from `services/shared/dedupe.py`. This function checks each item against a layered deduplication strategy. The fast path checks Redis for both content-hash markers (`app:dedupe:{hash}`) and canonical-URL markers (`app:dedupe:url:{url_hash}`), both with 24-hour TTLs. If the Redis check misses, the function falls back to PostgreSQL, querying the `documents` table by `content_hash` or `canonical_url` for durable cross-source matching. When a duplicate is found through the PostgreSQL fallback, the function warms the Redis cache so subsequent checks are fast.
|
||||
|
||||
Items identified as duplicates are not discarded entirely. If the duplicate document was originally ingested for a different entity, the worker creates a cross-source mention link in the `document_company_mentions` table via `persist_document_company_mention()`. This ensures that a news article mentioning both Entity-A and Entity-F is linked to both entities even if it was first ingested through Entity-A's news source.
|
||||
|
||||
New (non-duplicate) items are persisted to PostgreSQL through `persist_ingestion_items()` in `services/shared/metadata.py`, which inserts rows into the `documents` table and records entity mentions in `document_company_mentions`. Each new document ID is then pushed onto `app:queue:parsing` for the Parser to process. After persistence, the worker calls `mark_as_seen()` to set Redis dedupe markers for both the content hash and canonical URL of each new item, ensuring that the next fetch cycle's deduplication checks are fast.
|
||||
|
||||
On successful completion, the worker updates the `ingestion_runs` row with the final counts (`items_fetched`, `items_new`) and calls `reset_source_retry_state()` to clear any accumulated backoff from previous failures. For news-type sources (`news_api` and `macro_news`), the worker also updates the source's `config` JSONB column with the latest `published_utc` value, so the next fetch only retrieves newer articles.
|
||||
|
||||
---
|
||||
|
||||
## The Parser: Normalization, Quality Scoring, and Routing
|
||||
|
||||
Documents that pass through ingestion arrive on the `app:queue:parsing` Redis list as JSON payloads containing a `document_id`, `ticker`, and `source_type`. The Parser Worker (`services/parser/worker.py`) pops these jobs and transforms raw HTML or text into normalized, quality-scored documents ready for AI extraction.
|
||||
|
||||
The parsing pipeline begins with HTML fetching. If the document has a URL (looked up from the `documents` table if not present in the job payload), the worker calls `fetch_html()` to retrieve the page content. Public records API URLs receive a specialized `User-Agent` header to comply with the API's fair-access policy. The raw HTML is then passed to `parse_html()` in `services/parser/html_parser.py`, which runs a multi-stage extraction pipeline.
|
||||
|
||||
The HTML parser first strips non-content tags — `script`, `style`, `nav`, `footer`, `header`, `aside`, `iframe`, and others — and removes boilerplate containers identified by CSS class or ID patterns (sidebars, ad slots, newsletter signups, social share bars, and similar UI elements). It then searches for the article body using a priority list of semantic selectors (`article`, `[role='main']`, `.article-body`, `.post-content`, and others). If no semantic match is found, it falls back to text-density scoring across candidate `div`, `section`, and `td` elements, selecting the block with the highest composite score based on text density, link density, paragraph count, and word count. The extracted text undergoes further cleaning: regex-based removal of residual boilerplate phrases (copyright notices, "subscribe to our newsletter" prompts, "share this article" fragments), removal of short orphan lines that are likely UI fragments, detection and collapse of repeated template blocks, and whitespace normalization.
|
||||
|
||||
Metadata extraction pulls the document title (from `og:title` or `<title>`), author, publisher (from `og:site_name` or hostname), publication date (from `article:published_time` or JSON-LD `datePublished`), canonical URL, language, description, and keywords from the HTML head elements.
|
||||
|
||||
If the parsed body text is shorter than 500 characters, the worker attempts to enrich it by reading the raw API payload from MinIO and extracting the data provider's article description, keywords, and author fields for the matching article. This enrichment step ensures that even articles with minimal scrapeable HTML still have enough textual content for meaningful AI extraction.
|
||||
|
||||
Quality scoring is performed by `score_parse_quality()` in `services/parser/html_parser.py`, which evaluates six weighted signals to produce a composite score between 0 and 0.95:
|
||||
|
||||
| Signal | Weight | What It Measures |
|
||||
|--------------------|--------|-----------------------------------------------------------------|
|
||||
| `word_count` | 0.30 | Length of extracted text (thresholds at 20, 50, 150, 300 words) |
|
||||
| `body_found` | 0.20 | Whether a semantic article body element was located |
|
||||
| `diversity` | 0.15 | Vocabulary richness (unique words / total words) |
|
||||
| `sentence` | 0.15 | Presence of proper sentence structure (terminal punctuation) |
|
||||
| `paragraph` | 0.10 | Multi-paragraph structure (blocks separated by blank lines) |
|
||||
| `metadata` | 0.10 | Presence of title, author, publisher, and publication date |
|
||||
|
||||
The composite score maps to a confidence label: scores below 0.35 are labeled `low`, scores between 0.35 and 0.65 are `medium`, and scores 0.65 and above are `high`. Documents with `low` confidence are marked with status `low_quality` in the `documents` table and are not enqueued for extraction — they are effectively filtered out of the pipeline at this stage.
|
||||
|
||||
Entity mention detection runs next. The worker fetches all known aliases from the `company_aliases` table (plus entity identifiers and legal names from the `companies` table) and calls `detect_company_mentions()` in `services/parser/html_parser.py`. The matching strategy varies by alias length: one-to-two character aliases use case-sensitive word-boundary matching to avoid false positives (the letter "A" should not match every occurrence of the word "a"), three-to-four character aliases use case-insensitive word-boundary matching (standard identifier format), and aliases of five or more characters use case-insensitive substring matching (entity names and brands). Confidence scores vary by alias type: identifier matches receive 0.9, legal name matches 0.85, general aliases 0.7, and brand matches 0.6. Multiple alias hits for the same entity are deduplicated, keeping the highest-confidence match and summing match counts. Detected mentions are persisted to the `document_company_mentions` table.
|
||||
|
||||
The normalized text and a structured parser output JSON (containing all metadata, quality signals, warnings, outbound links, tags, and mentions) are uploaded to the `app-normalized` MinIO bucket. The `documents` row is updated with the normalized storage reference, parser output reference, quality score, and confidence level.
|
||||
|
||||
Finally, the Parser makes a routing decision. If the document's `document_type` is `macro_event`, it is pushed onto `app:queue:macro_classification` for the Global Event Classifier agent. All other documents are pushed onto `app:queue:extraction` for the Document Intelligence Extractor agent. Both queues feed into the Extractor service described in [Page 2](02-ai-agent-processing-and-extraction.md). The job payload includes the `document_id`, `ticker`, and the first 32,000 characters of the normalized text, giving the downstream agent immediate access to the content without needing to fetch it from MinIO.
|
||||
|
||||
For additional detail on queue topology and data store layout, see the [Data Pipeline Architecture](../architecture-data-pipeline.md) documentation.
|
||||
|
||||
---
|
||||
|
||||
## What Comes Next
|
||||
|
||||
At this point, raw data has been fetched from four external sources, deduplicated, stored in MinIO, parsed into normalized text, scored for quality, tagged with entity mentions, and routed to the appropriate extraction queue. The documents sitting on `app:queue:extraction` and `app:queue:macro_classification` are clean, quality-filtered, and ready for AI processing. [Page 2 — AI Agent Processing and Structured Extraction](02-ai-agent-processing-and-extraction.md) picks up the story from here, explaining how the Document Intelligence Extractor and Global Event Classifier agents use LLM inference to transform these normalized documents into the structured JSON intelligence that feeds the rest of the pipeline.
|
||||
@@ -0,0 +1,164 @@
|
||||
# Page 2 — AI Agent Processing and Structured Extraction
|
||||
|
||||
Documents that arrive on the `app:queue:extraction` and `app:queue:macro_classification` Redis queues are clean, quality-filtered, and normalized — but they are still unstructured text. The job of the Extractor service is to transform that text into structured JSON intelligence that the rest of the pipeline can reason about quantitatively. Two AI agents share this responsibility: the Document Intelligence Extractor handles entity-specific content, filings, and transcripts, while the Global Event Classifier handles macro-level geopolitical and economic events. Both agents run through the same Ollama-based inference infrastructure, share a common JSON repair pipeline, and persist their results to PostgreSQL and MinIO for downstream consumption and audit.
|
||||
|
||||
This page explains how each agent works, what schemas they produce, how the system validates and repairs LLM output, how runtime configuration is resolved from the database, and how the final structured records are persisted. For a visual overview of the full flow from ingestion through extraction, see the [Ingestion to Extraction Flow diagram](diagrams/ingestion-to-extraction-flow.md). For reference-level detail on agent configuration and the variant management API, see the [AI Agents Guide](../ai-agents.md).
|
||||
|
||||
---
|
||||
|
||||
## The Document Intelligence Extractor
|
||||
|
||||
The Document Intelligence Extractor is the primary AI agent in the pipeline. Registered under the slug `document-extractor` in the `ai_agents` database table, it processes every non-macro document that passes through the Parser — news articles, regulatory filings, performance transcripts, and press releases. Its purpose is to read a normalized document and produce a structured JSON object that captures the document's summary, the entities it affects, the sentiment and impact for each entity, the catalysts driving that impact, and the evidence supporting the analysis.
|
||||
|
||||
The entry point is `services/extractor/main.py`, which runs a continuous worker loop polling the `app:queue:extraction` Redis list. When a job arrives, the worker extracts the `document_id`, `ticker`, and `text` fields from the JSON payload. If the job payload does not include the document text directly, the worker fetches it from MinIO using the `normalized_storage_ref` stored in the `documents` table — the Parser uploaded the normalized text to the `app-normalized` bucket during the previous pipeline stage (see [Page 1](01-data-ingestion-and-preparation.md)).
|
||||
|
||||
The actual LLM inference is handled by `OllamaClient` in `services/extractor/client.py`. The client sends the document to a local Ollama instance via the `/api/chat` HTTP endpoint with `stream=False` and `think=False`. The `think=False` flag is a deliberate performance choice — it disables the model's chain-of-thought reasoning phase, which would otherwise add two to four minutes of latency per document. The client does not use Ollama's `format` parameter for structured output because of a known Ollama bug (#14645) where the format constraint is silently ignored when `think=False` on qwen3.5 models. Instead, the system relies on prompt engineering to produce JSON and repairs any syntax issues after the fact.
|
||||
|
||||
The prompt sent to the model has two parts. The system prompt, defined in `services/extractor/prompts.py`, establishes the model's role as a document analyst and sets strict output rules: return only a single JSON object, no markdown fences, no explanation text, every schema field is required, use `"other"` for `catalyst_type` when unsure, keep evidence spans under 20 words, and limit key facts to three to five items. The user prompt, built by `build_extraction_prompt()` in the same module, provides the document text along with document-type-specific guidance. Four guidance variants exist — one each for articles, filings, transcripts, and press releases — each calibrated to the conventions and biases of that document type. For example, the filing guidance instructs the model to preserve the precise legal language of regulatory documents, while the press release guidance warns that sentiment may be biased positive and directs the model to focus on concrete metrics rather than marketing language.
|
||||
|
||||
The user prompt also includes a list of all tracked entity identifiers from the `companies` table, along with rules for how the model should use them. If a tracked entity identifier appears verbatim in the text, the model must include it in the output with at least one evidence span. If the article discusses a sector or theme that clearly affects a tracked entity (oil prices affecting Entity-D, AI chip demand affecting Entity-C), the model should include that entity as well. The model is explicitly told not to invent identifiers that are not in the provided list. Documents longer than 8,000 characters are truncated before being included in the prompt, with a `[... truncated for extraction ...]` marker appended.
|
||||
|
||||
The `OllamaClient` also supports a `context_window` override via the Ollama `num_ctx` option, which can be configured per agent variant through the `AgentConfigResolver` mechanism described later in this page.
|
||||
|
||||
---
|
||||
|
||||
## The ExtractionResult Schema
|
||||
|
||||
The structured output that the Document Intelligence Extractor produces is defined by the `ExtractionResult` Pydantic model in `services/extractor/schemas.py`. Every field is required — the model has no defaults — so the generated JSON schema forces the LLM to produce every field explicitly. The top-level fields are:
|
||||
|
||||
**`summary`** — a concise one-to-three sentence summary of the document's main point. This becomes the human-readable description stored in the `document_intelligence` table.
|
||||
|
||||
**`companies`** — an array of `CompanyExtractionItem` objects, one per affected entity. Each entity entry contains:
|
||||
|
||||
- `ticker` — the entity identifier (validated against a regex pattern of one to five uppercase letters).
|
||||
- `company_name` — the full entity name as referenced in the document.
|
||||
- `relevance` — a float between 0.0 and 1.0 indicating how relevant the document is to this entity, where 0 means tangential and 1 means the entity is the primary subject.
|
||||
- `sentiment` — one of `positive`, `negative`, `neutral`, or `mixed`, representing the overall sentiment toward this entity in the document.
|
||||
- `impact_score` — a float between 0.0 and 1.0 estimating the magnitude of impact, where 0 is negligible and 1 is highly material.
|
||||
- `impact_horizon` — one of `intraday`, `1d`, `1d_7d`, `1d_30d`, `30d_90d`, or `90d_plus`, indicating the expected timeframe over which the impact will play out.
|
||||
- `catalyst_type` — exactly one of `performance_report`, `product`, `legal`, `macro`, `supply_chain`, `m_and_a`, `rating_change`, or `other`. The prompt instructs the model to use `other` when none of the specific categories fit.
|
||||
- `key_facts` — a list of facts explicitly stated in the document. The prompt emphasizes that the model must not infer or fabricate facts.
|
||||
- `risks` — a list of risks explicitly mentioned in the document.
|
||||
- `evidence_spans` — short verbatim quotes from the document supporting the analysis. The prompt requests these be kept under 20 words each.
|
||||
|
||||
**`macro_themes`** — a list of broad economic or environmental themes mentioned in the document, such as `rates`, `inflation`, or `ai_capex`.
|
||||
|
||||
**`novelty_score`** — a float between 0.0 and 1.0 indicating how novel or surprising the information is. Routine performance reports score low; unexpected regulatory actions score high. This value feeds into the novelty bonus component of the signal weighting formula described in [Page 3](03-signal-scoring-and-weighted-signals.md).
|
||||
|
||||
**`confidence`** — a float between 0.0 and 1.0 representing the model's confidence in the accuracy of its extraction. Lower values indicate ambiguous or incomplete source text. This value becomes the confidence gate input for signal scoring.
|
||||
|
||||
**`extraction_warnings`** — a list of issues encountered during extraction, such as `ambiguous_ticker`, `incomplete_text`, or `low_confidence`. These warnings are persisted alongside the intelligence record for operational monitoring.
|
||||
|
||||
The JSON schema is generated programmatically from the Pydantic models via `generate_json_schema()` in `services/extractor/schemas.py`, which calls Pydantic's `model_json_schema()` and then inlines all `$defs` references so the schema is self-contained and Ollama-friendly.
|
||||
|
||||
---
|
||||
|
||||
## The Global Event Classifier
|
||||
|
||||
Not all documents describe entity-specific developments. Macro news articles — those tagged with `document_type='macro_event'` by the Parser — describe events that affect entire sectors or economies: trade disputes, central bank rate decisions, commodity supply disruptions, geopolitical conflicts. These documents are routed to the `app:queue:macro_classification` Redis queue and processed by the Global Event Classifier agent, registered under the slug `event-classifier` in the `ai_agents` table.
|
||||
|
||||
The classifier is implemented in `services/extractor/event_classifier.py`. When the extractor worker in `services/extractor/main.py` pops a job and determines that the document type is `macro_event` (either because the job came from the macro queue or because the `documents` table records it as such), it routes the document to `_process_macro_classification()` instead of the standard extraction pipeline. This function calls `classify_global_event()`, which builds a dedicated prompt, sends it to Ollama through the same `OllamaClient` infrastructure, parses the response, and persists the result.
|
||||
|
||||
The classifier's system prompt is distinct from the extractor's. It establishes the model's role as a macro-level news classifier and includes explicit anti-hallucination rules that are critical to preventing the classifier from overreaching. The prompt states that the model should only classify articles about macro events that affect entire sectors or economies — trade disputes, interest rate changes, commodity supply disruptions, regulatory changes, geopolitical conflicts, natural disasters. It explicitly lists what should not be classified as macro events: individual entity performance reports, lawsuits against a single entity, single-entity management changes, individual entity analysis, entity-specific debt or bankruptcy, and product launches by one entity. For these entity-specific articles that were incorrectly routed, the model is instructed to set severity to `"low"`, confidence below 0.3, and leave the `affected_regions`, `affected_sectors`, and `affected_commodities` arrays empty.
|
||||
|
||||
The user prompt, built by `build_event_classification_prompt()`, reinforces these anti-hallucination rules and provides additional guidance. It instructs the model to only extract facts explicitly stated in the text, to set confidence below 0.4 for vague or speculative content, to distinguish announced policy from rumored policy, and to reserve `"critical"` severity for events affecting multiple countries or entire global systems. Articles longer than 6,000 characters are truncated before inclusion in the prompt.
|
||||
|
||||
The output schema is the `GlobalEvent` dataclass, which contains:
|
||||
|
||||
- `event_types` — a list of impact type strings, drawn from a fixed set: `supply_disruption`, `demand_shift`, `cost_increase`, `regulatory_pressure`, `currency_impact`, `commodity_shock`, `trade_barrier`, and `geopolitical_risk`. The model is instructed to include all applicable types rather than collapsing to a single category.
|
||||
- `severity` — one of `low`, `moderate`, `high`, or `critical`.
|
||||
- `affected_regions` — ISO 3166-1 alpha-2 country codes or region names (e.g., `US`, `CN`, `EU`, `GB`, `JP`). Only regions explicitly mentioned or clearly implied should be included.
|
||||
- `affected_sectors` — GICS sector identifiers such as `Energy`, `Financials`, `Information Technology`, or `Industrials`.
|
||||
- `affected_commodities` — commodity identifiers like `crude_oil`, `natural_gas`, `gold`, `copper`, `wheat`, `lithium`, or `semiconductors`. An empty list if no commodities are directly affected.
|
||||
- `summary` — a one-to-three sentence summary of the event and its domain implications.
|
||||
- `key_facts` — facts explicitly stated in the article, limited to three to five items.
|
||||
- `estimated_duration` — one of `short_term` (days to weeks), `medium_term` (weeks to months), or `long_term` (months to years).
|
||||
- `confidence` — a float between 0.0 and 1.0, clamped during parsing.
|
||||
|
||||
Each `GlobalEvent` also carries a `model_metadata` object recording the provider (`ollama`), model name, prompt version (`event-classification-v1`), and schema version (`1.0.0`), plus a `source_document_id` linking back to the originating document.
|
||||
|
||||
After a successful classification, the system computes macro impact records for all tracked entities using the exposure-based interpolation engine in `services/aggregation/interpolation.py`. Each entity's exposure profile — geographic revenue mix, supply chain regions, key input commodities, regulatory jurisdictions, and position tier — determines how much a given macro event affects that entity. Entities with non-zero macro impact scores get `macro_impact_records` rows persisted to PostgreSQL, and aggregation jobs are enqueued to `app:queue:aggregation` for each affected entity identifier. The extractor worker tracks consecutive macro classification failures and emits a critical-level alert after three consecutive failures, continuing with entity-only signals in the meantime.
|
||||
|
||||
---
|
||||
|
||||
## The JSON Repair Pipeline
|
||||
|
||||
LLM output is inherently unreliable at the syntactic level. Models sometimes wrap JSON in markdown fences, produce trailing commas, leave strings unterminated, or truncate output mid-object when they hit token limits. The extractor addresses this with a three-stage JSON repair pipeline implemented across `services/extractor/client.py` and `services/extractor/schemas.py`.
|
||||
|
||||
The first stage is a direct `json.loads()` call. If the raw model output is already valid JSON, no repair is needed and the pipeline moves straight to validation. This is the fast path for well-behaved model responses.
|
||||
|
||||
The second stage strips markdown fences. Models frequently wrap their output in `` ```json ... ``` `` blocks despite being told not to. The `_strip_markdown_fences()` function in `services/extractor/client.py` uses a regex to detect and remove these wrappers before attempting another parse.
|
||||
|
||||
The third stage invokes the `json-repair` library as a fallback. The `_repair_json()` function in `services/extractor/client.py` calls `repair_json()` with `return_objects=False` to get a repaired JSON string. This library handles a wide range of common LLM JSON errors — trailing commas, missing quotes, unescaped characters — that would otherwise require custom repair logic.
|
||||
|
||||
The `services/extractor/schemas.py` module contains an additional layer of repair logic in its own `_repair_json()` function, which handles cases that the library might miss. It strips non-JSON prefixes (models sometimes prepend explanatory text before the opening brace), removes control characters that break parsing, fixes trailing commas before closing brackets, and as a last resort calls `_repair_truncated_json()` — a state-machine parser that walks the string tracking bracket depth and string state, then appends the necessary closing tokens to complete a truncated JSON object.
|
||||
|
||||
For the Global Event Classifier, the `_parse_classification_response()` function in `services/extractor/event_classifier.py` reuses the same `_strip_markdown_fences()` and `_repair_json()` functions from the client module, and additionally handles the case where the model wraps the output object in a single-element list — a quirk observed with some model configurations.
|
||||
|
||||
---
|
||||
|
||||
## Structural and Semantic Validation
|
||||
|
||||
Repairing JSON syntax is only the first step. The `validate_extraction()` function in `services/extractor/schemas.py` performs both structural and semantic validation on the parsed output, and the distinction between the two is important for understanding the retry logic.
|
||||
|
||||
Structural validation begins with normalization. The `_normalize_extraction_data()` function fills in missing top-level fields with sensible defaults (empty summary, empty companies array, 0.5 novelty score, 0.3 confidence), clamps numeric fields to the [0.0, 1.0] range, and normalizes per-entity fields. Catalyst types that the model produces as free-text alternatives — `"strategic pivot"`, `"acquisition"`, `"lawsuit"`, `"inflation"`, `"launch"` — are mapped to their canonical enum values through a comprehensive alias dictionary. Impact horizons like `"long-term"`, `"short"`, `"immediate"`, or `"near-term"` are similarly mapped to the valid set (`intraday`, `1d`, `1d_7d`, `1d_30d`, `30d_90d`, `90d_plus`). After normalization, the data is validated against the `ExtractionResult` Pydantic model, which enforces type constraints, enum membership, and range bounds.
|
||||
|
||||
Semantic validation catches issues that are structurally valid but logically suspect. The `_semantic_checks()` function runs a series of cross-field consistency checks that produce either errors (which trigger a retry) or warnings (which are logged but do not block acceptance). Semantic errors include duplicate entity identifiers across entity entries, missing identifier fields, and invalid impact horizon values. Semantic warnings include empty summaries, low confidence with entities present, invalid identifier formats (not matching the one-to-five uppercase letter pattern), missing evidence spans, evidence spans that are too short (under 8 characters) or too long (over 500 characters), high impact scores with no supporting key facts, very low relevance scores, and strong sentiment paired with negligible impact scores.
|
||||
|
||||
When the original document text is available, the validator also performs an evidence grounding check: each evidence span is searched for in the source text (case-insensitive), and spans not found in the document are flagged with a warning. This helps detect hallucinated evidence — quotes the model fabricated rather than extracted from the actual text.
|
||||
|
||||
If validation produces any semantic errors, the `ValidationReport` is marked as invalid and the `OllamaClient` retry loop treats it as a failed attempt. The retry logic uses exponential backoff with configurable parameters: a base delay (default from `OllamaConfig`), a multiplier applied on each retry, and a maximum delay cap. The number of retries is configurable per agent through the `max_retries` field in the `ai_agents` or `agent_variants` table. Non-retryable errors — HTTP 400, 401, 403, 404, and 422 responses from Ollama — short-circuit the retry loop immediately, since these indicate a problem with the request itself rather than a transient model failure.
|
||||
|
||||
Every attempt, whether successful or not, is recorded in an `ExtractionAttempt` dataclass that captures the raw output, validation report, error description, duration in milliseconds, model name, and whether the error was retryable. The full list of attempts is preserved in the `ExtractionResponse` for audit purposes and uploaded to MinIO by the persistence layer.
|
||||
|
||||
---
|
||||
|
||||
## The AgentConfigResolver: Hot-Swapping Models and Prompts
|
||||
|
||||
Both the Document Intelligence Extractor and the Global Event Classifier resolve their runtime configuration through the `AgentConfigResolver` in `services/shared/agent_config.py`. This mechanism allows operators to change models, prompts, timeouts, retry counts, and token budgets without restarting any service — changes take effect within 60 seconds.
|
||||
|
||||
The resolver works by querying the `ai_agents` and `agent_variants` PostgreSQL tables with a single SQL statement that uses `COALESCE` to prefer variant values over base agent values. When the extractor worker starts, it creates an `AgentConfigResolver` instance with a 60-second TTL cache and calls `resolver.resolve("document-extractor")` to get the active configuration. If an active variant exists for the agent (enforced by a unique partial index on `agent_variants` that allows at most one active variant per agent), the variant's `model_name`, `system_prompt`, `temperature`, `max_tokens`, `context_window`, `timeout_seconds`, and `max_retries` override the base agent's values wherever the variant provides a non-NULL value. If no active variant exists, the base agent's configuration is used. If the database query fails entirely, the resolver returns `None` and the worker falls back to environment-variable-based `OllamaConfig` defaults.
|
||||
|
||||
The resolved configuration is captured in a `ResolvedAgentConfig` frozen dataclass that includes the `agent_id`, `variant_id` (if any), `model_provider`, `model_name`, `system_prompt`, `user_prompt_template`, `prompt_version`, `temperature`, `max_tokens`, `context_window`, `input_token_limit`, `token_budget`, `timeout_seconds`, and `max_retries`. The extractor worker uses this to build an `OllamaConfig` that is passed to the `OllamaClient`.
|
||||
|
||||
The 60-second TTL cache means the resolver only hits the database once per minute per agent slug. Cache entries are keyed by slug and timestamped with `time.monotonic()`. When a cached entry expires, the next `resolve()` call re-queries the database and refreshes the cache. The `invalidate()` method can clear a single slug or the entire cache, though in practice the TTL-based expiry is sufficient for normal operations.
|
||||
|
||||
The extractor worker re-resolves its configuration every 100 jobs. If the resolved model name has changed (for example, because an operator activated a variant that uses a different model), the worker closes the old `OllamaClient` and creates a new one with the updated configuration. The event classifier is resolved separately and can use a different model than the document extractor — the worker maintains two independent `OllamaClient` instances when the models differ.
|
||||
|
||||
Token budget enforcement adds another layer of control. If a variant specifies a `token_budget` (total tokens per hour), the worker checks the `agent_performance_log` table before each invocation to see whether the budget has been exceeded. If so, the invocation is skipped entirely. Input token limits work similarly: if a variant sets an `input_token_limit`, the worker truncates the document text to approximately that many tokens (estimated at four characters per token) before sending it to the model.
|
||||
|
||||
For a complete guide to creating variants, activating them, and comparing their performance, see the [AI Agents Guide](../ai-agents.md).
|
||||
|
||||
---
|
||||
|
||||
## Persistence: From Extraction to Database
|
||||
|
||||
Once the LLM produces a valid extraction and it passes validation, the `persist_extraction()` function in `services/extractor/worker.py` orchestrates the full persistence pipeline. This function writes to both MinIO (for audit) and PostgreSQL (for downstream consumption), ensuring that every extraction attempt is fully traceable.
|
||||
|
||||
The MinIO persistence layer uploads four artifacts per extraction, all stored under date-partitioned paths in dedicated buckets. The prompt metadata (prompt version, schema version, model name) goes to `app-llm-prompts`. The raw model output for every attempt — including failed ones — goes to `app-llm-results`, preserving the full retry history. A validation report summarizing the final attempt's status, errors, and warnings is uploaded alongside the raw output. On success, the final parsed intelligence object (the `ExtractionResult` serialized as JSON) is uploaded to a separate path for easy retrieval.
|
||||
|
||||
The PostgreSQL persistence writes to two tables. The `document_intelligence` table receives one row per document, containing the summary, macro themes, novelty score, source credibility, extraction warnings, confidence, model metadata (provider, model name, prompt version, schema version), references to the MinIO artifacts (raw output ref, prompt ref), validation status (`valid` or `failed`), validation errors, and retry count. This row is the authoritative record of what the AI extracted from the document.
|
||||
|
||||
The `document_impact_records` table receives one row per entity mention within the extraction. Each impact record is linked to the parent `document_intelligence` row via `intelligence_id` and to the `companies` table via `company_id`. The record captures the entity identifier, relevance, sentiment, impact score, impact horizon, catalyst type, key facts, risks, and evidence spans for that specific entity. The `company_id` is resolved from an identifier-to-UUID mapping that the worker maintains by querying the `companies` table (refreshed every 100 jobs). If an identifier in the extraction output does not match any tracked entity, the impact record is skipped with a warning — the system only persists impact records for entities in its tracked universe.
|
||||
|
||||
After persisting the intelligence and impact records, the worker updates the document's status in the `documents` table to `extracted` (or `extraction_failed` if all retry attempts were exhausted). Even failed extractions get a `document_intelligence` row with `validation_status='failed'`, empty summary, zero confidence, and the accumulated error messages — this ensures the failure is visible in the database rather than silently lost.
|
||||
|
||||
Performance metrics are collected for every extraction via `collect_metrics()` in `services/extractor/metrics.py` and persisted to a metrics table. Prometheus counters and histograms track extraction attempts, duration, retries, confidence distribution, validation errors, and estimated token usage (input and output, estimated at four characters per token). When a resolved agent config is available, the worker also logs to the `agent_performance_log` table with variant attribution, enabling the A/B comparison queries described in the [AI Agents Guide](../ai-agents.md).
|
||||
|
||||
For the Global Event Classifier, persistence follows a parallel path. The prompt and raw output are uploaded to MinIO under an `event_classification/macro/` path prefix. The parsed `GlobalEvent` is persisted to the `global_events` PostgreSQL table, which stores the event types, severity, affected regions, affected sectors, affected commodities, summary, key facts, estimated duration, confidence, source document ID, and model metadata. Downstream, the macro interpolation engine computes `macro_impact_records` for each affected entity and persists those as well.
|
||||
|
||||
---
|
||||
|
||||
## Enqueuing Aggregation Jobs
|
||||
|
||||
The final step in the extraction pipeline is to notify the downstream aggregation engine that new intelligence is available. After a successful document extraction, the worker pushes a job onto the `app:queue:aggregation` Redis list containing the identifier of the affected entity. The aggregation engine (described in [Page 3](03-signal-scoring-and-weighted-signals.md)) will pick up this job and recompute the weighted signals and trend summaries for that entity, incorporating the freshly extracted intelligence.
|
||||
|
||||
For macro events, the enqueue logic is more expansive. After the Global Event Classifier produces a `GlobalEvent` and the interpolation engine computes macro impact records, the worker enqueues an aggregation job for every entity identifier that received a non-zero macro impact score. A single macro event — say, a new regulatory policy change affecting the Energy and Industrials sectors — can trigger aggregation recomputation for dozens of entities simultaneously. The aggregation job payload includes both the entity identifier and the `macro_event_id`, so the aggregation engine knows to incorporate the new macro signals.
|
||||
|
||||
The worker alternates between the extraction and macro classification queues to prevent starvation: every third job is pulled from `app:queue:macro_classification`, with the remaining two-thirds from `app:queue:extraction`. If the preferred queue is empty, the worker falls back to the other queue, ensuring that neither pipeline stalls while the other has work available.
|
||||
|
||||
---
|
||||
|
||||
## What Comes Next
|
||||
|
||||
At this point, documents have been transformed from unstructured text into structured JSON intelligence — `ExtractionResult` objects for entity-specific documents and `GlobalEvent` objects for macro news. These structured records are persisted in PostgreSQL and their entity identifiers have been enqueued for aggregation. But raw extraction output is not yet actionable for downstream decisions. The extraction tells us that a document is negative for Entity-A with an impact score of 0.7 and a confidence of 0.8, but it does not tell us how much weight that signal should carry relative to other signals about Entity-A, or how it compares to signals from different sources, time periods, or environmental conditions. [Page 3 — Signal Scoring and the WeightedSignal Abstraction](03-signal-scoring-and-weighted-signals.md) picks up the story from here, explaining how the aggregation engine transforms these raw extraction outputs into weighted signals through confidence gating, recency decay, source credibility scoring, novelty bonuses, and environmental context multipliers.
|
||||
@@ -0,0 +1,210 @@
|
||||
# Page 3 — Signal Scoring and the WeightedSignal Abstraction
|
||||
|
||||
The extraction pipeline described in [Page 2](02-ai-agent-processing-and-extraction.md) produces structured intelligence records — `document_impact_records` for entity-specific documents, `macro_impact_records` for global events, and `competitive_signal_records` for cross-entity pattern propagation. Each record carries a sentiment, an impact score, a confidence value, and a publication timestamp. But these raw values are not directly comparable. A high-confidence extraction from a reputable source published ten minutes ago should carry far more weight than a low-confidence extraction from an unknown source published three weeks ago. A document that breaks genuinely novel information should matter more than one that rehashes yesterday's performance report. And when conditions are changing fast — high volatility, surging volume — fresh signals become even more critical.
|
||||
|
||||
The signal scoring layer in `services/aggregation/scoring.py` solves this problem by transforming each raw intelligence record into a `WeightedSignal` object: a document reference paired with a composite aggregation weight that encodes recency, credibility, novelty, confidence, and environmental conditions into a single number. This page explains how that weight is computed, how sentiment labels become numeric values, and how three independent signal layers — Entity-Specific, Macro, and Competitive — each produce `WeightedSignal` objects that are concatenated into a unified list before the aggregation engine computes trend summaries. For a visual breakdown of the composite weight formula, see the [Weighted Signal Computation diagram](diagrams/weighted-signal-computation.md). For the full picture of how the three layers merge, see the [Three-Layer Signal Merging diagram](diagrams/three-layer-signal-merging.md).
|
||||
|
||||
---
|
||||
|
||||
## The WeightedSignal and SignalWeight Dataclasses
|
||||
|
||||
The core abstraction is the `WeightedSignal` dataclass, defined in `services/aggregation/scoring.py`. It pairs a document reference with the computed weight and the signal's sentiment and impact values:
|
||||
|
||||
- **`document_id`** — the UUID of the source document (for entity-specific and macro signals) or a synthetic identifier for pattern-derived signals (e.g., `pattern:Entity-A:performance_report:7d`).
|
||||
- **`weight`** — a `SignalWeight` object containing the component breakdown and the final combined score.
|
||||
- **`sentiment_value`** — a numeric sentiment value: `+1.0` for positive, `-1.0` for negative, `0.0` for neutral or mixed.
|
||||
- **`impact_score`** — the magnitude of impact, drawn from the extraction's per-entity impact score for entity-specific signals, or scaled by a layer-specific weight multiplier for macro and competitive signals.
|
||||
|
||||
The `SignalWeight` dataclass captures the individual components that feed into the combined weight, making the scoring decision fully transparent and auditable:
|
||||
|
||||
- **`recency`** — the exponential decay weight based on document age.
|
||||
- **`credibility`** — the source credibility weight after clamping and exponentiation.
|
||||
- **`novelty_bonus`** — the additive bonus derived from the document's novelty score.
|
||||
- **`confidence_gate`** — either `1.0` (signal passes) or `0.0` (signal is gated out).
|
||||
- **`market_ctx_multiplier`** — a multiplicative boost from environmental conditions, always `>= 1.0`.
|
||||
- **`combined`** — the final composite weight used by the aggregation engine.
|
||||
|
||||
The `ScoringConfig` frozen dataclass holds all tunable parameters for the scoring functions — half-life hours per window, credibility bounds, novelty bonus cap, confidence floor, and environmental context thresholds. A module-level `DEFAULT_CONFIG` singleton provides the production defaults, but every scoring function accepts an optional `config` parameter so that tests and alternative configurations can override any parameter without modifying global state.
|
||||
|
||||
---
|
||||
|
||||
## The Composite Weight Formula
|
||||
|
||||
The `compute_signal_weight()` function in `services/aggregation/scoring.py` computes the combined weight for a single document signal. The formula is:
|
||||
|
||||
```
|
||||
combined = gate × recency × credibility × (1 + novelty_bonus) × market_context_multiplier
|
||||
```
|
||||
|
||||
Each factor is computed independently and then multiplied together. This multiplicative structure means that any single factor can zero out the entire weight (the confidence gate) or amplify it (the market context multiplier), and the interaction between factors is naturally captured — a highly credible, very recent document with novel information in a volatile environment receives the maximum possible weight, while a stale, low-credibility document with routine information receives a weight close to zero.
|
||||
|
||||
The following sections describe each component in detail.
|
||||
|
||||
---
|
||||
|
||||
## Confidence Gate
|
||||
|
||||
The confidence gate is the first and most decisive filter. If the extraction confidence for a document falls below the `confidence_floor` threshold — set to `0.2` in the default `ScoringConfig` — the gate evaluates to `0.0` and the entire combined weight becomes zero. The document is effectively excluded from aggregation. If the confidence meets or exceeds the threshold, the gate evaluates to `1.0` and has no further effect on the weight.
|
||||
|
||||
This binary gate exists because documents with very low extraction confidence are too unreliable to aggregate. A confidence of 0.15 typically means the LLM struggled to parse the document — perhaps the text was truncated, the language was ambiguous, or the document type was unusual. Including such signals would add noise rather than information. The threshold of 0.2 is deliberately low; it filters only the most unreliable extractions while allowing moderately confident signals to participate (their lower confidence is reflected through the credibility component instead).
|
||||
|
||||
---
|
||||
|
||||
## Recency Decay
|
||||
|
||||
The `recency_weight()` function computes an exponential decay based on how old a document is relative to the aggregation anchor time. The formula is:
|
||||
|
||||
```
|
||||
w = 2^(−age_hours / half_life)
|
||||
```
|
||||
|
||||
A document published exactly one half-life ago receives a recency weight of `0.5`. A document published two half-lives ago receives `0.25`, and so on. A document published at or after the reference time receives the maximum weight of `1.0`.
|
||||
|
||||
The half-life varies by trend window, reflecting the intuition that shorter windows need faster decay to stay responsive, while longer windows should give older documents more influence. The default half-lives, configured in `ScoringConfig.half_life_hours`, are:
|
||||
|
||||
| Window | Half-Life |
|
||||
|--------|-----------|
|
||||
| `intraday` | 2 hours |
|
||||
| `1d` | 12 hours |
|
||||
| `7d` | 72 hours (3 days) |
|
||||
| `30d` | 240 hours (10 days) |
|
||||
| `90d` | 720 hours (30 days) |
|
||||
|
||||
For the intraday window, a document published four hours ago already has a recency weight of `0.25` — it is rapidly losing influence as newer information arrives. For the 90-day window, that same four-hour-old document still has a recency weight of essentially `1.0`, because the 30-day half-life means age only becomes significant over weeks.
|
||||
|
||||
A floor value of `min_recency_weight = 0.01` prevents very old documents from being completely zeroed out. Even a document from months ago retains a trace-level weight of 1%, ensuring it can still contribute to trend computation if no newer signals exist. Both timestamps are normalized to UTC; naive datetimes are treated as UTC to avoid timezone-related scoring errors.
|
||||
|
||||
---
|
||||
|
||||
## Source Credibility
|
||||
|
||||
The `credibility_weight()` function transforms a source's credibility score into a weight component. The raw credibility value — a float between 0.0 and 1.0 stored in the `document_intelligence` table — is first clamped to the range `[0.1, 1.0]` using the `credibility_floor` and `credibility_ceiling` parameters from `ScoringConfig`. This clamping ensures that even the least credible sources retain a minimum weight of 0.1 rather than being completely silenced, while preventing any source from exceeding a weight of 1.0.
|
||||
|
||||
After clamping, the value is raised to the `credibility_exponent` power. The default exponent is `1.0`, which means the clamped credibility passes through unchanged. Setting the exponent above 1.0 would penalize low-credibility sources more aggressively — for example, an exponent of 2.0 would reduce a credibility of 0.5 to a weight of 0.25. Setting it below 1.0 would flatten the curve, making the system more tolerant of lower-credibility sources. The exponent is configurable through `ScoringConfig` to allow operators to tune the credibility sensitivity without changing the scoring code.
|
||||
|
||||
---
|
||||
|
||||
## Novelty Bonus
|
||||
|
||||
The novelty bonus rewards documents that contain genuinely new information. The bonus is computed as:
|
||||
|
||||
```
|
||||
novelty_bonus = novelty_score × novelty_bonus_max
|
||||
```
|
||||
|
||||
where `novelty_score` is the 0.0-to-1.0 value produced by the extraction model (see the `ExtractionResult` schema in [Page 2](02-ai-agent-processing-and-extraction.md)) and `novelty_bonus_max` is `0.25` by default. This means the bonus ranges from `0.0` (completely routine information) to `0.25` (maximally novel information), providing up to a 25% boost to the signal weight.
|
||||
|
||||
The bonus enters the composite formula as `(1 + novelty_bonus)`, so it acts as a multiplicative amplifier on the base weight. A document with a novelty score of 1.0 gets its weight multiplied by 1.25; a document with a novelty score of 0.0 gets multiplied by 1.0 (no change). This design ensures that novelty can only increase a signal's weight, never decrease it — routine information is not penalized, it simply does not receive the bonus.
|
||||
|
||||
---
|
||||
|
||||
## Environmental Context Multiplier
|
||||
|
||||
The `market_context_multiplier()` function computes a boost factor based on real-time environmental conditions for the entity being aggregated. The multiplier is always `>= 1.0`, meaning environmental context can only amplify signal weights, never reduce them. When no environmental context data is available (the `MarketContext` object from `services/shared/schemas.py` has `has_data == False`), the multiplier defaults to `1.0`.
|
||||
|
||||
Two environmental features contribute to the boost:
|
||||
|
||||
**Volatility boost.** When the entity's price volatility exceeds the `volatility_recency_boost_threshold` (default `1.0` in price units), the excess volatility is transformed through a logarithmic scaling function: `log₁₊(excess) × 0.15`. The logarithmic scaling prevents extreme volatility from producing runaway weight amplification. The boost is capped at `volatility_recency_boost_max = 0.30`, so the maximum volatility contribution is a 30% weight increase. The rationale is that in highly volatile environments, fresh intelligence is disproportionately valuable — a signal about Entity-C matters more when Entity-C is swinging 5% intraday than when it is moving in a tight range.
|
||||
|
||||
**Volume surge boost.** When the entity's volume change percentage exceeds `volume_surge_threshold_pct = 50.0%` (meaning activity volume is at least 50% above the prior period's average), a flat `volume_surge_boost = 0.15` is added. Unlike the volatility boost, this is binary — either the volume threshold is met and the full 15% boost applies, or it is not and no boost is added. High-volume moves carry more conviction because they represent broader participation rather than thin-activity noise.
|
||||
|
||||
The two boosts are additive within the multiplier: `multiplier = 1.0 + volatility_boost + volume_surge_boost`. In the most extreme case — high volatility and a volume surge — the combined multiplier reaches `1.0 + 0.30 + 0.15 = 1.45`, amplifying the signal weight by 45%. The `MarketContext` data is fetched by `services/aggregation/market_context.py` from the data tables in PostgreSQL, using the same entity identifier and window parameters as the impact record query.
|
||||
|
||||
---
|
||||
|
||||
## Sentiment Mapping
|
||||
|
||||
Before signals can be aggregated into trend summaries, the categorical sentiment labels from the extraction output must be converted to numeric values. The `sentiment_to_numeric()` function in `services/aggregation/scoring.py` performs this mapping:
|
||||
|
||||
| Sentiment Label | Numeric Value |
|
||||
|----------------|---------------|
|
||||
| `positive` | `+1.0` |
|
||||
| `negative` | `-1.0` |
|
||||
| `neutral` | `0.0` |
|
||||
| `mixed` | `0.0` |
|
||||
|
||||
The mapping is case-insensitive. Any unrecognized label defaults to `0.0`. The choice to map both `neutral` and `mixed` to `0.0` is deliberate — a mixed-sentiment document (one that contains both positive and negative signals for the same entity) should not push the trend in either direction. The contradiction between the positive and negative aspects is captured separately by the contradiction detection system described in [Page 4](04-trend-aggregation-and-accumulating-signals.md), rather than being baked into the sentiment value itself.
|
||||
|
||||
For macro signals, the direction-to-sentiment mapping in `services/aggregation/worker.py` follows the same pattern: `positive` maps to `+1.0`, `negative` to `-1.0`, and both `mixed` and `neutral` to `0.0`. For competitive signals built by `build_pattern_weighted_signals()` in `services/aggregation/signal_propagation.py`, the sentiment is derived from the pattern's directional bias: `+1.0` if `positive_pct > negative_pct`, `-1.0` otherwise.
|
||||
|
||||
---
|
||||
|
||||
## Weighted Sentiment Average
|
||||
|
||||
The `weighted_sentiment_average()` function computes the central metric that drives trend direction: a weight-adjusted average sentiment across all signals for an entity in a given window. The formula is:
|
||||
|
||||
```
|
||||
weighted_avg = Σ(combined_weight × impact_score × sentiment_value) / Σ(combined_weight × impact_score)
|
||||
```
|
||||
|
||||
Each signal contributes its sentiment value scaled by both its composite weight and its impact score. The denominator normalizes by the total effective weight, producing a value in the range `[-1.0, +1.0]`. A result near `+1.0` means the weighted evidence is overwhelmingly positive; near `-1.0` means overwhelmingly negative; near `0.0` means either neutral or evenly split.
|
||||
|
||||
The use of `combined_weight × impact_score` as the effective weight means that high-impact, high-weight signals dominate the average. A single high-confidence, recent, credible document with a strong impact score can outweigh several older, lower-impact documents — which is the intended behavior. The aggregation engine in `services/aggregation/worker.py` passes this weighted average to `derive_trend_direction()`, which maps it to a `TrendDirection` enum value (positive, negative, mixed, or neutral) using the thresholds described in [Page 4](04-trend-aggregation-and-accumulating-signals.md).
|
||||
|
||||
If the total effective weight is zero — either because no signals exist or all signals were gated out by the confidence floor — the function returns `0.0`, which maps to a neutral trend direction.
|
||||
|
||||
---
|
||||
|
||||
## The Three Signal Layers
|
||||
|
||||
The aggregation engine in `services/aggregation/worker.py` does not treat all intelligence sources equally. Signals flow through three independent layers, each with a different relative weight, before being concatenated into a single `WeightedSignal` list for trend computation. This layered architecture allows the system to incorporate diverse intelligence sources while controlling how much influence each source type has on the final trend.
|
||||
|
||||
### Layer 1 — Entity-Specific Signals (Weight: 1.0)
|
||||
|
||||
Entity-specific signals are the primary layer. They are built by `build_weighted_signals()` in `services/aggregation/worker.py` from `document_impact_records` — the per-entity extraction output produced by the Document Intelligence Extractor (see [Page 2](02-ai-agent-processing-and-extraction.md)). Each impact record's sentiment is converted via `sentiment_to_numeric()`, and its impact score is used directly without any layer-level scaling. The `compute_signal_weight()` function produces the composite weight using the document's publication time, source credibility, novelty score, extraction confidence, and the entity's current environmental context.
|
||||
|
||||
Entity-specific signals carry a relative weight of `1.0` — they are the baseline against which other layers are measured. This reflects the design principle that direct, entity-specific intelligence (a performance report about Entity-A, a product launch by Entity-B, a lawsuit against Entity-E) is the most relevant and reliable signal for that entity's trend.
|
||||
|
||||
### Layer 2 — Macro Signals (Weight: 0.3)
|
||||
|
||||
Macro signals capture the indirect impact of global events on individual entities. They are built by `build_macro_weighted_signals()` in `services/aggregation/worker.py` from `macro_impact_records` — the per-entity impact scores computed by the exposure-based interpolation engine after the Global Event Classifier processes a macro news article. The sentiment is mapped from the `impact_direction` field (`positive` → `+1.0`, `negative` → `-1.0`, `mixed`/`neutral` → `0.0`), and the impact score is scaled by `MACRO_SIGNAL_WEIGHT`, which defaults to `0.3` in `AggregationConfig`.
|
||||
|
||||
The 0.3 weight means that a macro signal's impact score is reduced to 30% of its raw value before entering the aggregation. This attenuation reflects the inherent uncertainty in macro-to-entity impact estimation — a policy change might affect Entity-D's revenue, but the magnitude depends on exposure profiles, supply chain flexibility, and competitive dynamics that the interpolation engine can only approximate. By weighting macro signals at 0.3 relative to entity-specific signals at 1.0, the system ensures that macro intelligence informs the trend without overwhelming direct entity-specific evidence.
|
||||
|
||||
The recency decay, credibility, and confidence gating for macro signals use the same `compute_signal_weight()` function as entity-specific signals. The `published_at` timestamp comes from the global event's source document (the macro news article), and the `source_credibility` and `extraction_confidence` both use the macro impact record's `confidence` field.
|
||||
|
||||
### Layer 3 — Competitive Signals (Weight: 0.2)
|
||||
|
||||
Competitive signals capture cross-entity effects: when a catalyst hits one entity, historical patterns suggest how competitors might be affected. They are built by `build_pattern_weighted_signals()` in `services/aggregation/signal_propagation.py` from two sources: `HistoricalPattern` objects (self-entity patterns mined by `services/aggregation/pattern_matcher.py`) and `CompetitiveSignalRecord` objects (cross-entity propagation signals stored in `competitive_signal_records`).
|
||||
|
||||
For historical patterns, the sentiment is derived from the pattern's directional bias (`+1.0` if `positive_pct > negative_pct`, `-1.0` otherwise), and the impact score is the pattern's `avg_strength` multiplied by `competitive_signal_weight` (default `0.2` from `CompetitiveConfig`). The `published_at` for recency decay uses the pattern's `data_end` — the most recent data point in the pattern's sample — and the `extraction_confidence` uses the pattern's `pattern_confidence`. Source credibility is set to `1.0` because patterns are derived from validated historical data, and novelty is fixed at `0.5`.
|
||||
|
||||
For competitive signal records, the same structure applies: sentiment from `signal_direction`, impact from `signal_strength × competitive_signal_weight`, recency from `computed_at`, and confidence from `pattern_confidence`.
|
||||
|
||||
The 0.2 weight makes competitive signals the lightest layer. This is appropriate because competitive signal propagation involves the most inference — the system is predicting how Entity B will react based on what happened to Entity A in historically similar situations. The signal is valuable as supplementary evidence but should not drive trend direction on its own.
|
||||
|
||||
---
|
||||
|
||||
## Signal Merging in the Aggregation Engine
|
||||
|
||||
The `aggregate_company_window()` function in `services/aggregation/worker.py` orchestrates the merging of all three layers for a single entity and window. The process follows a clear sequence:
|
||||
|
||||
1. **Fetch entity-specific impact records** from `document_impact_records` for the entity within the window's time range.
|
||||
2. **Fetch environmental context** for the entity from data tables.
|
||||
3. **Build entity-specific weighted signals** via `build_weighted_signals()`.
|
||||
4. **Check the macro toggle** — query `risk_configs` for the `macro_enabled` flag, then fetch and merge macro signals if enabled.
|
||||
5. **Check the competitive toggle** — query `risk_configs` for the `competitive_enabled` flag, then fetch patterns, fetch competitive signals, and merge if enabled.
|
||||
6. **Concatenate** all `WeightedSignal` lists into a single list.
|
||||
7. **Assemble the `TrendSummary`** from the merged signals.
|
||||
|
||||
The concatenation in step 6 is a simple list append — `signals = signals + macro_signals` followed by `signals = signals + pattern_weighted`. There is no re-weighting or normalization at the merge point. The relative influence of each layer is already encoded in the impact scores (scaled by 0.3 for macro, 0.2 for competitive, 1.0 for entity-specific) and in the composite weights computed by `compute_signal_weight()`. The `weighted_sentiment_average()` function then naturally produces a sentiment average that reflects these relative weights.
|
||||
|
||||
---
|
||||
|
||||
## Runtime Toggles and Graceful Degradation
|
||||
|
||||
Both the macro and competitive signal layers can be enabled or disabled at runtime through the `risk_configs` PostgreSQL table, without restarting any service. The toggle state is read fresh from the database at the start of every aggregation cycle — there is no caching — so changes take effect on the very next cycle.
|
||||
|
||||
The `fetch_macro_enabled()` function in `services/aggregation/worker.py` queries the most recent active `risk_configs` row and reads the `config->>'macro_enabled'` JSON field. If the field is explicitly set to `"true"` or `"false"`, that value overrides the `AggregationConfig` default. If no config row exists or the field is absent, the function returns `None` and the engine falls back to the `AggregationConfig.macro_enabled` default (which is `True`). The `fetch_competitive_enabled()` function follows the identical pattern for the `competitive_enabled` field.
|
||||
|
||||
When a layer is disabled, the aggregation engine simply skips the fetch-and-merge step for that layer. Entity-specific signals are always computed — they cannot be toggled off. This means the system degrades gracefully: disabling the macro layer produces trends based on entity-specific signals alone (plus competitive signals if enabled), and disabling the competitive layer produces trends based on entity-specific and macro signals. Disabling both layers reduces the engine to its original single-layer behavior, using only direct document intelligence.
|
||||
|
||||
Crucially, disabling a layer does not stop upstream processing. When the macro layer is disabled, the Global Event Classifier continues to classify macro events and the interpolation engine continues to compute `macro_impact_records`. The data accumulates in PostgreSQL. When the layer is re-enabled, the aggregation engine immediately picks up all the macro impact records that were computed while the layer was disabled — there is no data loss or gap in coverage. The same applies to competitive signals: pattern mining and signal propagation continue regardless of the toggle state.
|
||||
|
||||
If the competitive signal fetch fails at runtime (for example, due to a database timeout), the aggregation engine catches the exception, logs it, and continues with entity-specific and macro signals only. This exception-based graceful degradation ensures that a transient failure in one layer does not block trend computation entirely.
|
||||
|
||||
---
|
||||
|
||||
## What Comes Next
|
||||
|
||||
At this point, every document intelligence record, macro impact record, and competitive signal record has been transformed into a `WeightedSignal` with a composite weight that encodes recency, credibility, novelty, confidence, and environmental conditions. The three signal layers have been merged into a single list, and the weighted sentiment average has been computed. But a single aggregation cycle produces only a snapshot — a point-in-time view of the evidence. The real power of the system emerges when these snapshots accumulate across multiple documents and time windows, building a case for action. [Page 4 — Trend Aggregation and Accumulating Signals](04-trend-aggregation-and-accumulating-signals.md) explains how the aggregation engine computes `TrendSummary` objects across five time windows, how consecutive same-direction signals strengthen trend confidence and escalate the system's response from neutral observation to actionable decision recommendations, and how contradiction detection and evidence ranking ensure that the trend reflects genuine consensus rather than noise.
|
||||
@@ -0,0 +1,267 @@
|
||||
# Page 4 — Trend Aggregation and Accumulating Signals
|
||||
|
||||
The scoring layer described in [Page 3](03-signal-scoring-and-weighted-signals.md) transforms every intelligence record into a `WeightedSignal` — a document reference paired with a composite weight that encodes recency, credibility, novelty, confidence, and environmental conditions. Three independent signal layers (Entity-Specific at weight 1.0, Environmental at 0.3, Relational at 0.2) each produce `WeightedSignal` objects that are concatenated into a single list. But a single list of weighted signals is still just raw material. The aggregation engine in `services/aggregation/worker.py` is where that raw material becomes a decision-grade assessment: a `TrendSummary` object that captures the direction, strength, confidence, contradiction level, and supporting evidence for an entity across a specific time window. This page explains how that transformation works — from weighted sentiment averages through trend direction derivation, contradiction detection, evidence ranking, and confidence computation — and, critically, how consecutive signals pointing in the same direction accumulate across documents and time windows to escalate the system's response from passive observation to actionable decision recommendations.
|
||||
|
||||
For a visual overview of the accumulation and escalation process, see the [Trend Accumulation and Escalation diagram](diagrams/trend-accumulation-escalation.md). For how the three signal layers merge into the aggregation engine, see the [Three-Layer Signal Merging diagram](diagrams/three-layer-signal-merging.md).
|
||||
|
||||
---
|
||||
|
||||
## Five Time Windows
|
||||
|
||||
The aggregation engine does not compute a single trend for each entity. It computes five, one for each time window defined in `services/aggregation/worker.py`:
|
||||
|
||||
| Window | Lookback Duration |
|
||||
|--------|-------------------|
|
||||
| `intraday` | 12 hours |
|
||||
| `1d` | 1 day |
|
||||
| `7d` | 7 days |
|
||||
| `30d` | 30 days |
|
||||
| `90d` | 90 days |
|
||||
|
||||
Each window produces an independent `TrendSummary` by fetching all impact records, macro impacts, and competitive signals for the entity within that window's time range. The `aggregate_company_window()` function in `services/aggregation/worker.py` orchestrates this per-window computation: it determines the time range from the window's lookback duration, fetches `document_impact_records` from PostgreSQL, retrieves environmental context, builds entity-specific weighted signals, checks the macro and competitive runtime toggles (see [Page 3](03-signal-scoring-and-weighted-signals.md) for toggle details), merges any enabled layer signals, and then assembles the `TrendSummary`.
|
||||
|
||||
The five-window design serves a specific purpose. Short windows (intraday, 1d) capture fast-moving sentiment shifts — a breaking negative performance disclosure, a sudden regulatory action — while long windows (30d, 90d) reveal sustained trends that persist across many documents and data cycles. An entity might show a negative intraday trend after a single unfavorable article, but a neutral 30-day trend because the broader evidence base is balanced. The recommendation engine downstream (described in [Page 5](05-recommendation-generation.md)) evaluates each window's `TrendSummary` independently, so the system can respond to both short-term catalysts and long-term directional shifts.
|
||||
|
||||
The `aggregate_company()` function iterates over all effective windows (configurable via `AggregationConfig.windows`, defaulting to all five) and calls `aggregate_company_window()` for each one. This means a single aggregation cycle for one entity produces up to five `TrendSummary` objects, each reflecting a different temporal perspective on the same underlying evidence.
|
||||
|
||||
---
|
||||
|
||||
## Trend Direction Derivation
|
||||
|
||||
Once the weighted sentiment average has been computed from the merged signal list (see the `weighted_sentiment_average()` function described in [Page 3](03-signal-scoring-and-weighted-signals.md)), the `derive_trend_direction()` function in `services/aggregation/worker.py` maps that numeric value to a `TrendDirection` enum. The rules are evaluated in a specific order, and the first matching rule wins:
|
||||
|
||||
1. **Mixed** — If the contradiction score exceeds `0.10` (the `MIXED_THRESHOLD` constant) *and* the absolute value of the average sentiment is below `0.30`, the direction is `MIXED`. This rule fires first because high contradiction with a weak directional signal indicates genuine disagreement in the evidence — the trend is not simply neutral, it is actively contested.
|
||||
|
||||
2. **Positive** — If the average sentiment is `≥ 0.15` (the `POSITIVE_THRESHOLD` constant), the direction is `POSITIVE`. This means the weight-adjusted evidence leans favorable with enough conviction to cross the threshold.
|
||||
|
||||
3. **Negative** — If the average sentiment is `≤ -0.15` (the `NEGATIVE_THRESHOLD` constant), the direction is `NEGATIVE`. The symmetric threshold ensures that positive and negative classifications require the same magnitude of evidence.
|
||||
|
||||
4. **Neutral** — If none of the above conditions are met, the direction is `NEUTRAL`. This covers the range where the average sentiment falls between -0.15 and +0.15 without high contradiction — the evidence is either balanced or insufficient to establish a directional lean.
|
||||
|
||||
The mixed-first evaluation order is important. Consider a scenario where five documents are positive and four are negative, all with similar weights. The weighted sentiment average might be slightly positive (say, +0.08), which would normally map to neutral. But the contradiction score — computed from the minority/majority weight split — would be high (close to 0.44). The mixed rule catches this case: the evidence is not neutral, it is conflicted. This distinction matters downstream because mixed trends receive different treatment in the recommendation engine than neutral trends.
|
||||
|
||||
---
|
||||
|
||||
## Contradiction Detection
|
||||
|
||||
The contradiction detection module in `services/aggregation/contradiction.py` provides a structured analysis of disagreement within the signal set. Rather than collapsing contradictory evidence into a single number, it produces a `ContradictionResult` containing both an overall score and a list of `DisagreementDetail` objects that explain *where* the disagreement lies.
|
||||
|
||||
The `detect_contradictions()` function runs two analyses:
|
||||
|
||||
### Sentiment Disagreement
|
||||
|
||||
The `_detect_sentiment_disagreement()` function examines whether both positive and negative sentiment signals exist in the signal set. For each signal with a non-zero effective weight (`combined_weight × impact_score > 0`), it classifies the signal as positive or negative based on its `sentiment_value` and accumulates the effective weight for each side. If both sides have at least one signal, it produces a `DisagreementDetail` with dimension `"sentiment"`, listing the document IDs and weights for each side, along with a human-readable description like "Sentiment split: 3 positive vs 2 negative signals (minority weight ratio 38%)".
|
||||
|
||||
### Catalyst-Level Disagreement
|
||||
|
||||
The `_detect_catalyst_disagreement()` function goes deeper. It groups signals by their `catalyst_type` (performance_report, product_launch, regulatory, etc.) using `CatalystEntry` objects built from the `document_impact_records`. Within each catalyst group, it checks whether both positive and negative signals exist. If they do, it produces a `DisagreementDetail` with dimension `"catalyst:<type>"` — for example, `"catalyst:performance_report"` when some documents interpret a periodic disclosure positively and others negatively. This catalyst-level analysis is valuable because it pinpoints the specific topic of disagreement rather than just flagging that disagreement exists somewhere in the evidence.
|
||||
|
||||
### The Overall Contradiction Score
|
||||
|
||||
The `_compute_overall_score()` function computes the backward-compatible scalar contradiction score using the minority/majority weight ratio formula:
|
||||
|
||||
```
|
||||
contradiction_score = minority_weight / total_weight
|
||||
```
|
||||
|
||||
where `minority_weight` is the smaller of the positive and negative effective weights, and `total_weight` is their sum. Signals with zero effective weight or neutral sentiment are excluded. The score ranges from `0.0` (complete agreement — all signals point the same direction) to `0.5` (perfect split — positive and negative weights are exactly equal). A score of `0.0` means no contradiction at all. A score above `0.10` combined with a weak average sentiment triggers the mixed direction classification in `derive_trend_direction()`.
|
||||
|
||||
The contradiction score also feeds directly into the confidence computation as a penalty, described in the next section. High contradiction reduces the system's confidence in the trend, which in turn affects whether the trend can escalate to actionable recommendations.
|
||||
|
||||
---
|
||||
|
||||
## Evidence Ranking
|
||||
|
||||
Not all documents contributing to a trend are equally important. The `rank_evidence()` function in `services/aggregation/worker.py` delegates to the evidence ranking module (`services/aggregation/evidence.py`) to produce ordered lists of the most influential supporting and opposing documents. The ranking uses a composite scoring approach configured by `EvidenceRankConfig`, considering multiple factors:
|
||||
|
||||
- **Weight** — the signal's composite weight from the scoring layer, reflecting recency, credibility, novelty, confidence, and environmental context.
|
||||
- **Impact** — the extraction's impact score for the entity, reflecting how significant the document's content is.
|
||||
- **Recency** — how recently the document was published, with more recent documents ranked higher.
|
||||
- **Confidence** — the extraction confidence, reflecting how reliably the LLM parsed the document.
|
||||
|
||||
Signals are split into supporting (positive sentiment) and opposing (negative sentiment) groups. Neutral and mixed sentiment signals are excluded from evidence lists — they do not argue for or against the trend direction. Within each group, signals are sorted by their composite rank score in descending order, and the top entries (up to `MAX_EVIDENCE_REFS = 10` per side) are returned as document ID lists.
|
||||
|
||||
The `assemble_trend_with_evidence()` function in `services/aggregation/worker.py` uses the detailed variant `rank_evidence_detailed()` to get `RankedEvidence` objects that include the individual scoring components (weight, impact, recency, confidence, sentiment value). These detailed rankings are persisted to the `trend_evidence` table for auditability, while the document ID lists are stored directly in the `TrendSummary` as `top_supporting_evidence` and `top_opposing_evidence`.
|
||||
|
||||
The evidence ranking serves two purposes. First, it provides the recommendation engine with the most relevant documents to cite in its thesis generation (see [Page 5](05-recommendation-generation.md)). Second, it gives human reviewers a quick way to understand *why* the system reached a particular trend assessment — the top-ranked documents are the ones that most influenced the direction and strength.
|
||||
|
||||
---
|
||||
|
||||
## Confidence Computation
|
||||
|
||||
The `compute_trend_confidence()` function in `services/aggregation/worker.py` produces the confidence score for a `TrendSummary`. This score is critical because it directly gates whether a trend can produce actionable recommendations — the eligibility evaluation in `services/recommendation/eligibility.py` requires a minimum confidence of `0.35` to generate any recommendation at all, and higher confidence thresholds control escalation to simulation and live execution modes.
|
||||
|
||||
Confidence is computed from four components:
|
||||
|
||||
### Unique Source Count
|
||||
|
||||
The function counts the number of unique document IDs across all active signals (those with `combined_weight > 0`). This count is divided by 15 and capped at `0.8`:
|
||||
|
||||
```
|
||||
count_factor = min(unique_sources / 15.0, 0.8)
|
||||
```
|
||||
|
||||
A trend backed by 15 or more unique source documents reaches the maximum count contribution of `0.8`. A trend backed by a single document gets only `0.067`. This component rewards breadth of evidence — a trend confirmed by many independent sources is more trustworthy than one driven by a single article, regardless of how high that article's individual weight might be.
|
||||
|
||||
### Average Extraction Credibility
|
||||
|
||||
The average credibility weight across all active signals provides a baseline quality measure. If most contributing documents come from high-credibility sources, this component is high. If the evidence is dominated by low-credibility sources, confidence is penalized accordingly.
|
||||
|
||||
### Signal Agreement with Sample-Size Dampening
|
||||
|
||||
The agreement ratio measures what fraction of directional signals (positive + negative, excluding neutral) agree on the majority direction. If 8 out of 10 directional signals are positive, the raw agreement is `0.8`. But raw agreement is misleading with small sample sizes — 1 out of 1 signals agreeing gives a perfect `1.0` agreement, which is not meaningful.
|
||||
|
||||
To address this, the agreement is dampened by a logarithmic sample-size factor:
|
||||
|
||||
```
|
||||
agreement_dampener = min(1.0, log₂(unique_sources + 1) / log₂(8))
|
||||
```
|
||||
|
||||
This dampener saturates at `1.0` when `unique_sources` reaches approximately 7 (since `log₂(8) = 3.0` and `log₂(8) = 3.0`). With fewer sources, the dampener reduces the agreement contribution: 1 source gives a dampener of `0.33`, 3 sources give `0.67`, and 7 sources give the full `1.0`. The log₂ scaling means that each additional source provides diminishing marginal improvement to the dampener, which matches the intuition that the jump from 1 to 3 sources is far more meaningful than the jump from 15 to 17.
|
||||
|
||||
### Contradiction Penalty
|
||||
|
||||
The contradiction score computed by `services/aggregation/contradiction.py` is applied as a direct penalty:
|
||||
|
||||
```
|
||||
contradiction_penalty = contradiction_score × 0.4
|
||||
```
|
||||
|
||||
A contradiction score of `0.5` (perfect split) produces a penalty of `0.2`, which is substantial enough to push a moderately confident trend below the eligibility threshold.
|
||||
|
||||
### The Combined Formula
|
||||
|
||||
The four components are combined as:
|
||||
|
||||
```
|
||||
confidence = 0.3 × count_factor + 0.3 × avg_credibility + 0.4 × agreement − contradiction_penalty
|
||||
```
|
||||
|
||||
The result is clamped to `[0.0, 1.0]`. The weighting gives signal agreement the largest share (40%), reflecting the principle that consensus among diverse sources is the strongest indicator of a reliable trend. Source count and credibility each contribute 30%, providing a balanced assessment of evidence breadth and quality. The contradiction penalty can reduce confidence significantly — a highly contradicted trend with a score of 0.4 loses 0.16 points of confidence, which can easily drop it below the 0.35 eligibility gate.
|
||||
|
||||
---
|
||||
|
||||
## How Accumulating Signals Escalate Decisions
|
||||
|
||||
The trend direction, strength, and confidence computed by the aggregation engine are not just descriptive — they directly determine what action the system takes. The escalation path from passive observation to active execution is governed by the eligibility thresholds defined in `services/recommendation/eligibility.py`, and the key insight is that consecutive signals pointing in the same direction naturally strengthen the trend metrics that control this escalation.
|
||||
|
||||
### The Escalation Ladder
|
||||
|
||||
The `EligibilityConfig` dataclass in `services/recommendation/eligibility.py` defines the thresholds that map trend metrics to actions:
|
||||
|
||||
**Neutral (no recommendation).** A trend fails the eligibility gates entirely when confidence is below `0.35`, trend strength is below `0.10`, contradiction exceeds `0.60`, evidence count is below `2`, or the direction is neutral. The `_check_gates()` function evaluates these hard gates — if any gate fails, no recommendation is generated for that window.
|
||||
|
||||
**Observe.** A trend that passes the gates but has a direction of mixed, or has strength below `0.25` with confidence below `0.50`, maps to an `OBSERVE` action via `_determine_action()`. This is the system's way of saying "something is happening, but the evidence is not strong enough to act on." Observe recommendations are always `informational` mode — they are logged for human review but never trigger decisions.
|
||||
|
||||
**Monitor.** When the trend has a clear direction (positive or negative) but strength remains below `0.25` while confidence reaches `0.50` or above, the action maps to `MONITOR`. This indicates that the directional signal is real but not yet strong enough for a commitment change. Like observe, monitor recommendations are `informational` mode.
|
||||
|
||||
**Act / Defer.** When trend strength reaches `0.25` or above with a positive direction, the action is `ACT`. With a negative direction at the same strength threshold, the action is `DEFER`. These are the only actions that can escalate beyond informational mode — `_determine_mode()` evaluates whether the recommendation qualifies for `simulation_eligible` (confidence ≥ `0.50`) or `production_eligible` (confidence ≥ `0.70`, contradiction ≤ `0.25`, evidence ≥ `5`).
|
||||
|
||||
### How Accumulation Drives Escalation
|
||||
|
||||
Consider an entity that starts with no recent intelligence. The first negative article arrives — a single document with negative sentiment. In the intraday window, this produces:
|
||||
|
||||
- **Trend strength** = `|avg_sentiment|` ≈ the absolute weighted sentiment from one signal, likely close to the impact score.
|
||||
- **Confidence** = low, because `count_factor = min(1/15, 0.8) = 0.067` and the agreement dampener is only `log₂(2)/log₂(8) = 0.33`.
|
||||
- **Direction** = negative (if the weighted sentiment is ≤ -0.15).
|
||||
|
||||
With confidence well below `0.35`, this trend fails the eligibility gate entirely. No recommendation is generated. The system is in the neutral state.
|
||||
|
||||
A second negative article arrives hours later. Now the intraday window has two signals:
|
||||
|
||||
- **Unique sources** = 2, so `count_factor = 0.133` and `agreement_dampener = log₂(3)/log₂(8) ≈ 0.53`.
|
||||
- **Agreement** = `1.0 × 0.53 = 0.53` (both signals agree on negative).
|
||||
- **Confidence** ≈ `0.3 × 0.133 + 0.3 × avg_cred + 0.4 × 0.53` — likely around `0.35-0.45` depending on credibility.
|
||||
|
||||
If confidence crosses `0.35` and strength exceeds `0.10`, the trend passes the eligibility gates. But with strength below `0.25`, the action is `OBSERVE` or `MONITOR` depending on confidence.
|
||||
|
||||
A third and fourth negative article arrive over the next day. The 1-day window now has four agreeing signals:
|
||||
|
||||
- **Unique sources** = 4, so `count_factor = 0.267` and `agreement_dampener = log₂(5)/log₂(8) ≈ 0.77`.
|
||||
- **Agreement** = `1.0 × 0.77 = 0.77`.
|
||||
- **Confidence** ≈ `0.3 × 0.267 + 0.3 × avg_cred + 0.4 × 0.77` — likely `0.50-0.60`.
|
||||
- **Strength** = `|avg_sentiment|` — with four negative signals and no contradicting evidence, this could easily exceed `0.25`.
|
||||
|
||||
Now the trend maps to `DEFER` with `simulation_eligible` mode (confidence ≥ `0.50`). The system has escalated from no recommendation to a simulation-eligible defer recommendation purely through the accumulation of consistent negative evidence.
|
||||
|
||||
If the negative evidence continues — more documents, more sources, higher credibility — confidence climbs further. At confidence ≥ `0.70` with contradiction ≤ `0.25` and evidence ≥ `5`, the recommendation reaches `production_eligible` mode, the highest escalation level.
|
||||
|
||||
The same process works in reverse for positive accumulation: consecutive favorable signals strengthen the positive trend, increase confidence through source diversity and agreement, and escalate from observe through monitor to act.
|
||||
|
||||
### The Role of Contradiction in Preventing False Escalation
|
||||
|
||||
Accumulation only works when signals agree. If the fifth article about an entity is positive while the previous four were negative, the contradiction score jumps — `minority_weight / total_weight` increases because the minority (positive) side now has non-zero weight. This has two effects: the contradiction penalty reduces confidence (potentially dropping it below an eligibility threshold), and if the contradiction exceeds `0.10` with `|avg_sentiment| < 0.30`, the direction flips to mixed, which maps to `OBSERVE` regardless of strength. The system effectively de-escalates when the evidence becomes contested, requiring a clearer consensus before re-escalating.
|
||||
|
||||
---
|
||||
|
||||
## Trend Projections
|
||||
|
||||
After the `TrendSummary` is assembled and persisted, the aggregation engine computes a forward-looking `TrendProjection` via `compute_projection()` in `services/aggregation/projection.py`. Projections estimate where the trend is heading based on current momentum, macro signal decay, and upcoming catalysts. They are advisory — they do not directly trigger recommendations — but they provide valuable context for human reviewers and can inform future automated decision-making.
|
||||
|
||||
### Momentum
|
||||
|
||||
The `compute_trend_momentum()` function computes the rate of change in signed trend strength between the current and previous aggregation cycles. If the current window shows a negative trend at strength `0.40` and the previous cycle showed negative at `0.30`, the momentum is `-0.10` (strengthening negative). If no previous data is available, the function uses a heuristic: momentum is estimated as half the current signed strength, providing a reasonable baseline for new trends.
|
||||
|
||||
Momentum enters the projection as a half-weighted adjustment to the current signed strength:
|
||||
|
||||
```
|
||||
momentum_projected_signed = direction_sign × current_strength + momentum × 0.5
|
||||
```
|
||||
|
||||
This means momentum influences the projection but does not dominate it — a strong current trend with weakening momentum still projects as directional, just with reduced strength.
|
||||
|
||||
### Macro Decay
|
||||
|
||||
The `project_macro_decay()` function estimates how active macro events will evolve over the projection horizon. Each macro event has an `estimated_duration` that maps to a decay half-life:
|
||||
|
||||
| Duration | Half-Life |
|
||||
|----------|-----------|
|
||||
| `short_term` | 1 day |
|
||||
| `medium_term` | 7 days |
|
||||
| `long_term` | 30 days |
|
||||
|
||||
For each event, the function computes the projected remaining impact at the end of the horizon using exponential decay: `future_factor = 2^(−future_age_days / half_life)`. The impact is further scaled by a severity weight (`critical`: 1.0, `high`: 0.75, `moderate`: 0.5, `low`: 0.25). Positive and negative macro impacts are accumulated separately, and the projected macro direction is determined by comparing the two sides — positive if the favorable side exceeds the unfavorable by 20%, negative if the reverse, mixed if both are present without a clear majority.
|
||||
|
||||
When the macro layer is enabled and macro events exist, the projection blends the entity-specific momentum projection with the macro trajectory. The macro weight is capped at `0.4` (40% of the blended projection), ensuring that macro signals inform but do not overwhelm the entity-specific trend. The blending formula combines the signed entity projection with the signed macro projection:
|
||||
|
||||
```
|
||||
blended = company_weight × momentum_projected + macro_weight × macro_signed
|
||||
```
|
||||
|
||||
### Driving Factors
|
||||
|
||||
The projection records a list of human-readable driving factors that explain what is influencing the projected direction. These include momentum descriptions ("Positive momentum (+0.150) in recent trend strength"), macro impact projections ("Macro signals project negative impact (strength 0.350) over 7d"), and upcoming catalysts drawn from the trend's `dominant_catalysts` list (limited to the top 3). If no specific factors are identified, a baseline continuation factor is recorded.
|
||||
|
||||
### Divergence Detection
|
||||
|
||||
After computing the projected direction, the function compares it to the current trend direction. If they differ — for example, the current trend is negative but the projection is positive due to decaying unfavorable macro events and favorable momentum — the projection is flagged with `diverges_from_current = True` and a divergence driving factor is appended. Divergence signals are particularly valuable because they indicate that the trend may be about to reverse, giving the recommendation engine and human reviewers an early warning.
|
||||
|
||||
The projection also flags low confidence when `projected_confidence` falls below the default threshold of `0.3`. Projection confidence starts at 80% of the current trend confidence (reflecting the inherent uncertainty of forward-looking estimates), with a small boost if macro data is available and a further reduction if the macro layer is disabled entirely.
|
||||
|
||||
---
|
||||
|
||||
## Persistence
|
||||
|
||||
Each aggregation cycle persists its results to four PostgreSQL tables, creating a durable record of the trend assessment and its supporting evidence.
|
||||
|
||||
### `trend_windows` — Current State
|
||||
|
||||
The `persist_trend_summary()` function in `services/aggregation/worker.py` upserts the `TrendSummary` into the `trend_windows` table, keyed by `(entity_type, entity_id, window)`. Each cycle overwrites the previous row for that entity and window, so `trend_windows` always reflects the most recent assessment. The row includes the trend direction, strength, confidence, contradiction score, disagreement details (as JSON), supporting and opposing evidence document IDs (as JSON arrays), dominant catalysts, material risks, environmental context, and the generation timestamp.
|
||||
|
||||
### `trend_history` — Time-Series Snapshots
|
||||
|
||||
Immediately after the upsert, `persist_trend_summary()` also inserts a snapshot row into the `trend_history` table. Unlike `trend_windows`, this table is append-only — every aggregation cycle adds a new row, creating a time-series of how the trend evolved over time. The history table stores the direction, strength, confidence, contradiction score, catalysts, risks, and timestamp. This time-series data powers the trend charts in the dashboard and enables the momentum computation in `services/aggregation/projection.py` by providing the previous cycle's strength and direction. If the history insert fails (for example, if the table does not yet exist in a development environment), the failure is logged at debug level and does not block the main upsert.
|
||||
|
||||
### `trend_evidence` — Per-Document Rankings
|
||||
|
||||
The `persist_trend_evidence()` function writes detailed evidence ranking rows to the `trend_evidence` table, linked to the `trend_windows` row by its UUID. Each row records a document ID, its role (supporting or opposing), and the individual scoring components: rank score, weight component, impact component, recency component, confidence component, and sentiment value. Non-UUID document IDs (such as synthetic pattern signal IDs like `pattern:Entity-A:performance_report:7d`) are filtered out before insertion, since the `trend_evidence` table enforces a foreign key to the `documents` table.
|
||||
|
||||
### `trend_projections` — Forward-Looking Estimates
|
||||
|
||||
The `persist_trend_projection()` function in `services/aggregation/projection.py` inserts the `TrendProjection` into the `trend_projections` table, linked to the `trend_windows` row. The row stores the projected direction, strength, confidence, projection horizon, driving factors (as JSON), macro contribution percentage, divergence flag, and computation timestamp. Like trend history, projections accumulate over time, allowing analysis of how well the system's forward-looking estimates matched subsequent reality.
|
||||
|
||||
---
|
||||
|
||||
## What Comes Next
|
||||
|
||||
At this point, the aggregation engine has transformed weighted signals into `TrendSummary` objects across five time windows, detected contradictions, ranked evidence, computed confidence, and persisted everything to PostgreSQL. The trend metrics — direction, strength, confidence, contradiction score — encode the accumulated weight of evidence for each entity. But a `TrendSummary` is still an assessment, not an action. The next stage translates these assessments into concrete recommendations: should the system act, defer, monitor, or simply observe? And with what conviction? [Page 5 — Recommendation Generation](05-recommendation-generation.md) explains how the recommendation engine applies data quality suppression, eligibility evaluation, commitment sizing, thesis generation, and risk classification to convert trend summaries into actionable `Recommendation` objects that the decision execution engine can execute.
|
||||
@@ -0,0 +1,226 @@
|
||||
# Page 5 — Recommendation Generation and Signal-to-Action Translation
|
||||
|
||||
The aggregation engine described in [Page 4](04-trend-aggregation-and-accumulating-signals.md) produces `TrendSummary` objects across five time windows for each entity identifier, encoding the direction, strength, confidence, contradiction level, and supporting evidence accumulated from all three signal layers. But a `TrendSummary` is an assessment — it describes what the evidence says, not what the system should do about it. The recommendation engine is where assessment becomes action. It takes each `TrendSummary`, subjects it to a series of deterministic evaluations, and produces a `Recommendation` object that specifies a concrete action (act, defer, monitor, or observe), an execution mode (informational, simulation-eligible, or production-eligible), a commitment sizing guideline, a human-readable thesis, and a risk classification. Every decision in this pipeline is rule-based and fully traceable — the LLM is only involved in an optional downstream step that rewrites the thesis wording.
|
||||
|
||||
The recommendation worker in `services/recommendation/main.py` polls the `app:queue:recommendation` Redis queue for jobs, each specifying an entity identifier and time window. For each job, it delegates to `generate_recommendation()` in `services/recommendation/worker.py`, which orchestrates the full pipeline: fetch the latest trend summary, check for duplicate recommendations, fetch any available trend projection, evaluate data quality suppression, evaluate eligibility, optionally rewrite the thesis via LLM, build the `Recommendation` object, and persist everything to PostgreSQL. For a visual overview of this flow, see the [Recommendation Generation Flow diagram](diagrams/recommendation-generation-flow.md).
|
||||
|
||||
---
|
||||
|
||||
## Data Quality Suppression
|
||||
|
||||
Before the eligibility engine evaluates whether a trend is strong enough to act on, the suppression layer in `services/recommendation/suppression.py` asks a more fundamental question: is the underlying data reliable enough to act on at all? A trend might show high confidence and strong directionality, but if the documents feeding it are stale, poorly extracted, or drawn from a single source type, the apparent signal quality is illusory. The suppression layer acts as a pre-filter on data quality, running before the eligibility engine and forcing any recommendation built on unreliable data to `informational` mode regardless of how strong the trend metrics look.
|
||||
|
||||
The `evaluate_suppression()` function accepts a `TrendSummary` and a `DataQualityContext` — a set of metrics about the documents underlying the trend, populated by querying `documents` and `document_intelligence` tables for the evidence document IDs stored in the trend summary. When full document-level metrics are not available (for example, in a development environment without the full document pipeline), the function falls back to `build_quality_context_from_summary()`, which estimates quality metrics from the trend summary's own evidence counts and confidence.
|
||||
|
||||
### The Six Data Quality Checks
|
||||
|
||||
The suppression evaluation runs six independent checks, each comparing a data quality metric against a configurable threshold defined in `SuppressionConfig`. If any single check fails, the recommendation is suppressed:
|
||||
|
||||
1. **Low extraction confidence** — If the average extraction confidence across the evidence documents falls below `0.40` (`min_avg_extraction_confidence`), the underlying LLM extractions are too unreliable. This catches cases where the extractor struggled with document formatting, ambiguous content, or low-quality source material, as described in [Page 2](02-ai-agent-processing-and-extraction.md).
|
||||
|
||||
2. **Evidence staleness** — If the most recent evidence document is older than `168` hours (7 days, `max_evidence_staleness_hours`), the trend is based on outdated information. Conditions change rapidly, and a week-old evidence base may no longer reflect the current state. When documents exist but no timestamp is available, the evidence is conservatively treated as stale.
|
||||
|
||||
3. **Low source diversity** — If fewer than `1` distinct source type (`min_source_types`) contributed to the evidence, the signal may be driven by a single unreliable source class. In practice, this check fires when the quality context has documents but all come from the same source type (for example, all news articles with no filings or supplementary data to corroborate).
|
||||
|
||||
4. **High extraction failure rate** — If more than `50%` (`max_extraction_failure_rate`) of the documents that should have contributed to the trend failed extraction entirely, the data pipeline is unreliable for this entity. A high failure rate means the trend summary is built from a biased subset of the available evidence — the failed documents might have told a different story.
|
||||
|
||||
5. **Insufficient valid documents** — If fewer than `2` valid (non-failed) documents (`min_valid_documents`) contributed to the trend, there simply is not enough data to act on. A single document, no matter how high-quality, does not provide the corroboration needed for automated execution decisions.
|
||||
|
||||
6. **Low data quality score** — The `_compute_data_quality_score()` function computes an overall quality score from three weighted components: extraction confidence (40% weight, normalized against a 0.8 baseline), evidence freshness (30% weight, linear decay over the staleness window), and document coverage (30% weight, combining the valid/total ratio with a count factor that saturates at 10 documents). If this composite score falls below `0.30` (`min_data_quality_score`) and the low-confidence check has not already fired, a general suppression reason is added.
|
||||
|
||||
When any check triggers, the `SuppressionResult` records the specific reasons (as `SuppressionReason` enum values) and the computed data quality score. The worker in `services/recommendation/worker.py` uses this result to force the recommendation's mode to `informational` and append a suppression note to the thesis text, ensuring the suppression decision is visible in the audit trail.
|
||||
|
||||
### Safety Suppressions: Macro-Only and Pattern-Only Signals
|
||||
|
||||
Beyond the six data quality checks, two additional safety suppressions protect against acting on signals that lack entity-specific corroboration:
|
||||
|
||||
**Macro-only suppression** (`evaluate_macro_only_suppression()`) fires when macro signals are the sole basis for a trend direction — no entity-specific signals contributed at all. As described in [Page 3](03-signal-scoring-and-weighted-signals.md), macro signals enter the aggregation engine at a reduced weight of `0.3` relative to entity-specific signals. But even at reduced weight, macro signals alone can shift a trend direction if no entity-specific evidence exists. When this happens, the recommendation is forced to `informational` mode with a caveat noting that the signal is macro-only and should not be used for automated execution.
|
||||
|
||||
**Pattern-only suppression** (`evaluate_pattern_only_suppression()`) applies the same logic to competitive/pattern signals. When pattern-based signals from `services/aggregation/pattern_matcher.py` and `services/aggregation/signal_propagation.py` are the sole contributors — no entity-specific or macro signals — the recommendation is suppressed. Historical patterns are valuable context, but acting on them without any current evidence is too speculative for automated execution.
|
||||
|
||||
Both safety suppressions are evaluated in the worker after the main suppression check, and both force the mode to `informational` when triggered.
|
||||
|
||||
---
|
||||
|
||||
## Eligibility Evaluation
|
||||
|
||||
Recommendations that survive the suppression layer enter the eligibility evaluation in `services/recommendation/eligibility.py`. This is the core decision logic — a set of deterministic rules that map trend metrics to actions, execution modes, and commitment sizing. The `evaluate_eligibility()` function is the single entry point, accepting a `TrendSummary` and an `EligibilityConfig` of tunable thresholds.
|
||||
|
||||
### Gate Checks
|
||||
|
||||
The `_check_gates()` function applies five hard gates. If any gate fails, the trend is ineligible for a recommendation (though the action and mode are still computed for the audit trace):
|
||||
|
||||
| Gate | Threshold | Rejection Reason |
|
||||
|------|-----------|-----------------|
|
||||
| Confidence | ≥ `0.35` | `low_confidence` |
|
||||
| Trend strength | ≥ `0.10` | `low_trend_strength` |
|
||||
| Contradiction score | ≤ `0.60` | `high_contradiction` |
|
||||
| Evidence count | ≥ `2` (supporting + opposing) | `insufficient_evidence` |
|
||||
| Direction | ≠ `neutral` | `neutral_direction` |
|
||||
|
||||
These gates are intentionally conservative. A confidence threshold of `0.35` means the system needs meaningful evidence breadth and agreement before generating any recommendation at all (see the confidence computation in [Page 4](04-trend-aggregation-and-accumulating-signals.md)). The contradiction ceiling of `0.60` allows moderately contested trends through — only when the evidence is deeply split does the gate reject. The evidence minimum of `2` ensures that no recommendation is ever based on a single document.
|
||||
|
||||
When a trend fails any gate, the resulting `EligibilityResult` has `eligible = False` and the mode is forced to `informational`, regardless of what the mode escalation logic would otherwise compute.
|
||||
|
||||
### Action Mapping
|
||||
|
||||
The `_determine_action()` function maps the trend's direction and strength to one of four action types. The logic evaluates in a specific order:
|
||||
|
||||
**Mixed or neutral direction → OBSERVE.** If the trend direction is `mixed` (high contradiction with weak directional signal) or `neutral`, the action is always `OBSERVE`. There is no directional conviction to act on.
|
||||
|
||||
**Strong directional signal → ACT or DEFER.** If the trend strength reaches `0.25` or above (`action_strength_threshold`), the action follows the direction: `ACT` for positive, `DEFER` for negative. This threshold ensures that only trends with meaningful magnitude trigger commitment-changing actions.
|
||||
|
||||
**Weak directional signal with decent confidence → MONITOR.** If the trend has a clear direction (positive or negative) but strength remains below `0.25`, the action depends on confidence. If confidence reaches `0.50` or above (`hold_confidence_threshold`), the action is `MONITOR` — the system recognizes the directional lean but does not have enough conviction to recommend a commitment change. Below `0.50` confidence, the action falls to `OBSERVE`.
|
||||
|
||||
This mapping creates the escalation ladder described in [Page 4](04-trend-aggregation-and-accumulating-signals.md): as consecutive signals accumulate and strengthen the trend metrics, the action naturally progresses from OBSERVE → MONITOR → ACT/DEFER.
|
||||
|
||||
### Mode Escalation
|
||||
|
||||
The `_determine_mode()` function determines the highest execution mode allowed for the recommendation. Mode controls whether the recommendation is purely informational, eligible for simulation mode, or eligible for live execution mode:
|
||||
|
||||
**OBSERVE and MONITOR → always informational.** These actions do not trigger executions, so they are always `informational` mode. They are logged for human review and dashboard display but never enter the decision execution engine.
|
||||
|
||||
**ACT and DEFER → escalation based on signal quality.** For actionable recommendations, mode escalates through three tiers:
|
||||
|
||||
- **`informational`** — The default when confidence is below `0.50`. The recommendation is recorded but not eligible for any execution.
|
||||
- **`simulation_eligible`** — When confidence reaches `0.50` or above (`paper_confidence_threshold`). The recommendation can be picked up by the simulation engine described in [Page 6](06-decision-execution.md).
|
||||
- **`production_eligible`** — The strictest tier, requiring confidence ≥ `0.70` (`live_confidence_threshold`), contradiction ≤ `0.25` (`live_max_contradiction`), and evidence count ≥ `5` (`live_min_evidence`). This triple gate ensures that only high-conviction, well-corroborated, low-contradiction recommendations can trigger live executions.
|
||||
|
||||
The evidence count for mode escalation is computed as the sum of supporting and opposing evidence documents, matching the same count used in the gate checks.
|
||||
|
||||
---
|
||||
|
||||
## Commitment Sizing
|
||||
|
||||
The `_compute_position_sizing()` function in `services/recommendation/eligibility.py` translates signal quality into an allocation pool guideline. Commitment sizing is not a fixed value — it scales dynamically with the confidence and strength of the underlying trend, penalized by contradiction and thin evidence.
|
||||
|
||||
### Base and Scaling
|
||||
|
||||
The computation starts with a base allocation of `1%` (`base_allocation_pct = 0.01`) and scales upward based on two factors:
|
||||
|
||||
- **Confidence factor** — `0.8 × confidence` (`confidence_sizing_weight`), reflecting how much the system trusts the trend assessment.
|
||||
- **Strength factor** — `0.5 + 0.5 × trend_strength`, ranging from `0.5` (weakest trend) to `1.0` (strongest trend).
|
||||
|
||||
The raw allocation percentage is computed as:
|
||||
|
||||
```
|
||||
raw_allocation = base + confidence_factor × strength_factor × (max - base)
|
||||
```
|
||||
|
||||
where `max` is `10%` (`max_allocation_pct = 0.10`). At maximum confidence (1.0) and maximum strength (1.0), the raw allocation reaches the full 10%. At typical values (confidence 0.6, strength 0.3), the raw allocation is considerably lower.
|
||||
|
||||
### Contradiction Penalty
|
||||
|
||||
The contradiction score applies a multiplicative penalty:
|
||||
|
||||
```
|
||||
allocation_pct = raw_allocation × (1.0 − 0.5 × contradiction_score)
|
||||
```
|
||||
|
||||
A contradiction score of `0.40` reduces the allocation by 20%. A score of `0.0` (no contradiction) applies no penalty. This ensures that contested trends receive smaller commitment sizes even when they pass the eligibility gates.
|
||||
|
||||
### Evidence Count Penalty
|
||||
|
||||
Thin evidence further reduces the allocation:
|
||||
|
||||
- Fewer than `3` evidence documents → multiply by `0.5` (halved).
|
||||
- Fewer than `5` evidence documents → multiply by `0.75`.
|
||||
- `5` or more documents → no penalty.
|
||||
|
||||
This penalty stacks with the contradiction penalty, so a trend with high contradiction and thin evidence receives a substantially reduced commitment size.
|
||||
|
||||
### Max Loss Scaling
|
||||
|
||||
The same scaling logic applies to the maximum loss percentage, which starts at a base of `0.3%` (`base_max_loss_pct = 0.003`) and scales up to `2%` (`max_max_loss_pct = 0.02`). Higher-conviction commitments are allowed larger loss tolerances, while low-conviction or contested commitments are constrained to tighter risk thresholds.
|
||||
|
||||
The final `PositionSizing` object (defined in `services/shared/schemas.py`) contains `allocation_pct` and `max_loss_pct`, both clamped to their respective bounds. This object is embedded in the `Recommendation` and later consumed by the decision execution engine's own commitment sizer (described in [Page 6](06-decision-execution.md)), which applies additional resource pool-level constraints.
|
||||
|
||||
---
|
||||
|
||||
## Thesis Generation
|
||||
|
||||
Every recommendation includes a human-readable thesis that explains the reasoning behind the action. Thesis generation happens in two layers: a deterministic assembly that is always present, and an optional LLM rewrite that polishes the wording for execution-eligible recommendations.
|
||||
|
||||
### Deterministic Thesis Assembly
|
||||
|
||||
The `build_thesis()` function in `services/recommendation/worker.py` constructs a thesis string entirely from the trend data and eligibility result, with no model involvement. The thesis is assembled from several components in order:
|
||||
|
||||
1. **Opening** — States the entity identifier, trend direction, window, strength, and confidence. For example: "Entity-A shows a negative trend over the 7d window with strength 0.35 and confidence 0.62."
|
||||
|
||||
2. **Catalysts** — Lists the top three dominant catalysts from the `TrendSummary`, drawn from the evidence ranking described in [Page 4](04-trend-aggregation-and-accumulating-signals.md).
|
||||
|
||||
3. **Contradiction note** — If the contradiction score exceeds `0.15`, a note flags the signal disagreement and its magnitude.
|
||||
|
||||
4. **Trend projection** — When a `TrendProjection` is available and not flagged as low-confidence, the thesis incorporates the projected direction, strength, and top driving factors. If the projection diverges from the current trend, a divergence note is appended.
|
||||
|
||||
5. **Risks** — Lists the top two material risks from the `TrendSummary`.
|
||||
|
||||
6. **Evidence count** — States the number of supporting and opposing evidence documents.
|
||||
|
||||
7. **Prescriptive action** — States the recommended action and mode (e.g., "Recommendation: DEFER (simulation eligible).").
|
||||
|
||||
The deterministic thesis is always generated and serves as the audit reference. Even when the LLM rewrites the thesis, the deterministic version is preserved in the model metadata for traceability.
|
||||
|
||||
### Optional LLM Rewrite via the Thesis-Rewriter Agent
|
||||
|
||||
For recommendations that are both eligible and not suppressed, the worker optionally invokes the thesis-rewriter agent to polish the deterministic thesis into professional-quality prose. The LLM rewrite is implemented in `services/recommendation/thesis_llm.py` and uses the `thesis-rewriter` agent slug, resolved at runtime through the `AgentConfigResolver` in `services/shared/agent_config.py`.
|
||||
|
||||
The `AgentConfigResolver` queries the `ai_agents` and `agent_variants` database tables to resolve the active configuration for the `thesis-rewriter` slug, preferring an active variant's model, timeout, and retry settings when one exists. The resolver uses a 60-second TTL in-memory cache to avoid hitting the database on every recommendation. This is the same resolution mechanism used by the document extractor and event classifier agents described in [Page 2](02-ai-agent-processing-and-extraction.md).
|
||||
|
||||
The `rewrite_thesis_with_llm()` function builds a prompt from the deterministic thesis and trend context (entity identifier, window, direction, strength, confidence, contradiction score, catalysts, risks), sends it to the local Ollama instance via HTTP, and returns the rewritten text. The system prompt enforces strict rules: no fabricated information, no numbers or facts not present in the input, under 150 words, neutral professional tone, and only the rewritten thesis text in the response.
|
||||
|
||||
The LLM layer is purely additive — if the call fails for any reason (network error, timeout, empty response, token budget exceeded), the original deterministic thesis is returned unchanged. The worker in `services/recommendation/main.py` resolves the thesis-rewriter configuration at startup and refreshes it every 50 jobs to pick up configuration changes without requiring a restart. When no database configuration exists for the `thesis-rewriter` slug, thesis rewriting is silently disabled.
|
||||
|
||||
Performance logging for the thesis-rewriter is written to the `agent_performance_log` table, recording success/failure, duration, estimated token counts, and the variant ID. Token budget enforcement checks hourly usage against the variant's configured budget before making the LLM call, preventing runaway costs from high-volume recommendation cycles.
|
||||
|
||||
### Risk Classification Prefix
|
||||
|
||||
Before the thesis is stored, the `classify_risk()` function in `services/recommendation/worker.py` assigns a risk classification label that is prepended to the thesis text as a `[risk:<level>]` prefix. The classification is computed from a composite score:
|
||||
|
||||
| Factor | Contribution |
|
||||
|--------|-------------|
|
||||
| Contradiction score | `contradiction × 2.0` |
|
||||
| Low confidence | `(1.0 − confidence) × 1.5` |
|
||||
| Low evidence count | `+1.0` if < 3 docs, `+0.5` if < 5 docs |
|
||||
| Rejection reasons | `+0.5` per rejection reason |
|
||||
|
||||
The composite score maps to four levels:
|
||||
|
||||
| Score Range | Classification |
|
||||
|-------------|---------------|
|
||||
| ≥ 3.0 | `very_high` |
|
||||
| ≥ 2.0 | `high` |
|
||||
| ≥ 1.0 | `moderate` |
|
||||
| < 1.0 | `low` |
|
||||
|
||||
A recommendation with high contradiction (0.4 → contributes 0.8), moderate confidence (0.55 → contributes 0.675), and 4 evidence documents (contributes 0.5) would score 1.975, classifying as `moderate`. The same recommendation with only 2 evidence documents would score 2.475, pushing it to `high`. This classification gives downstream consumers — both the decision execution engine and human reviewers — a quick risk signal without needing to re-evaluate the underlying metrics.
|
||||
|
||||
---
|
||||
|
||||
## Persistence
|
||||
|
||||
The recommendation pipeline persists its output to three PostgreSQL tables, creating a complete audit trail from trend assessment through decision logic to the final recommendation.
|
||||
|
||||
### `recommendations` — The Core Record
|
||||
|
||||
The `persist_recommendation()` function in `services/recommendation/worker.py` inserts the `Recommendation` into the `recommendations` table. Each row captures the entity identifier, action, mode, confidence, time horizon, thesis (including the risk classification prefix and any suppression notes), invalidation conditions (as JSONB), commitment sizing (allocation percentage and max loss percentage), model metadata (provider, model name, prompt version, schema version), risk classification, and generation timestamp. The insert returns the recommendation's UUID, which serves as the foreign key for the evidence and risk evaluation tables.
|
||||
|
||||
### `recommendation_evidence` — Evidence Citations
|
||||
|
||||
For each evidence document referenced in the recommendation, a row is inserted into the `recommendation_evidence` table linking the recommendation UUID to the document UUID, with an evidence type (`supporting` or `opposing`) and a position-based weight that decays with rank: `weight = 1.0 / (1.0 + index × 0.1)`. The first supporting document gets weight `1.0`, the second gets `0.91`, the third `0.83`, and so on. Non-UUID document IDs (such as synthetic pattern signal IDs like `pattern:Entity-A:performance_report:7d` from the competitive signal layer) are filtered out before insertion, since the table enforces a foreign key to the `documents` table.
|
||||
|
||||
### `risk_evaluations` — Decision Audit Trail
|
||||
|
||||
The `risk_evaluations` table records the full eligibility decision for each recommendation: whether the trend was eligible, the allowed mode, the list of rejection reasons (as JSONB), and a `risk_checks` JSONB object containing the time horizon, commitment sizing details, invalidation conditions, and risk classification. This table enables post-hoc analysis of why the system made a particular decision — auditors can trace from the recommendation back through the eligibility evaluation to the underlying trend metrics.
|
||||
|
||||
---
|
||||
|
||||
## Deduplication
|
||||
|
||||
Before running the full evaluation pipeline, the worker checks whether the latest recommendation for the same entity identifier and time horizon is effectively identical to what would be generated. The `_is_duplicate_recommendation()` function in `services/recommendation/worker.py` compares the previous recommendation's action, mode, and confidence (within a `0.01` tolerance) against the current eligibility result. If all three match, the recommendation is skipped — the underlying trend data has not changed meaningfully since the last cycle. This prevents the system from flooding the `recommendations` table with identical entries on every aggregation cycle, while still generating a new recommendation whenever the trend metrics shift enough to change the action, mode, or confidence.
|
||||
|
||||
---
|
||||
|
||||
## What Comes Next
|
||||
|
||||
At this point, the recommendation engine has translated trend assessments into concrete `Recommendation` objects — each with an action, execution mode, commitment sizing guideline, thesis, and risk classification — and persisted them alongside their evidence citations and eligibility audit trails. Recommendations marked as `simulation_eligible` or `production_eligible` are now available for the decision execution engine to consume. [Page 6 — Decision Execution](06-decision-execution.md) explains how the decision execution engine polls these recommendations, applies its own pre-execution check sequence (circuit breakers, execution windows, confidence gates, deduplication, declining commitments, and max open commitments), computes final commitment sizes with resource pool-level constraints, and submits execution requests through the execution adapter to the external execution API.
|
||||
@@ -0,0 +1,199 @@
|
||||
# Page 6 — Decision Execution
|
||||
|
||||
The recommendation engine described in [Page 5](05-recommendation-generation.md) produces `Recommendation` objects with an action, execution mode, commitment sizing guideline, thesis, and risk classification. Recommendations marked as `simulation_eligible` or `production_eligible` are persisted to the `recommendations` table and are now available for the final stage of the pipeline: autonomous decision execution. The decision execution engine in `services/trading/engine.py` is where intelligence becomes action. It polls eligible recommendations, subjects each one to a strict sequence of pre-execution safety checks, computes a pool-aware commitment size, and — if every gate passes — submits an execution request through the execution adapter to the external execution API. Every evaluation, whether it results in a decision or a skip, is recorded as a `DecisionRecord` in the `execution_decisions` table, creating a complete audit trail from the original document signal through to the execution response.
|
||||
|
||||
For a visual overview of the decision flow, see the [Decision Engine Loop diagram](diagrams/decision-engine-loop.md).
|
||||
|
||||
---
|
||||
|
||||
## The Decision Execution Engine Loop
|
||||
|
||||
The `DecisionEngine` class in `services/trading/engine.py` is the orchestrator. When `start()` is called, it loads the current resource pool state from PostgreSQL — active commitments, reserve pool balance, sector exposure, pool exposure — and then spawns five concurrent `asyncio` tasks that run for the lifetime of the engine:
|
||||
|
||||
1. **`_decision_loop()`** — The core polling loop. Every 60 seconds (configurable via `polling_interval_seconds`), it queries the `recommendations` table for rows where `action IN ('act', 'defer')`, `mode IN ('simulation_eligible', 'production_eligible')`, and `generated_at` is within the last two hours. Recommendations are ordered by confidence descending and capped at 50 per cycle. For each recommendation, the engine fetches the current data point (first from `market_snapshots`, falling back to the data source API), then runs the full pre-execution evaluation pipeline described below.
|
||||
|
||||
2. **`_risk_threshold_monitor()`** — Periodically checks current values against the risk threshold and gain target levels maintained by the `RiskThresholdManager` in `services/trading/stop_loss_manager.py`. When a value crosses a risk threshold or gain target, the monitor submits a defer execution request to the execution queue. The `RiskThresholdManager` computes initial levels from ATR and risk tier parameters, re-evaluates them when volatility shifts materially (ATR change > 10%), activates trailing thresholds when the value moves more than 50% toward the gain target, and tightens thresholds proactively when pool exposure exceeds 80% of the maximum.
|
||||
|
||||
3. **`_performance_loop()`** — Computes pool-wide performance metrics (total value, unrealized and realized gain/loss, success rate, risk-adjusted return ratio, peak-to-trough decline, pool exposure), persists daily snapshots to `pool_snapshots`, checks for daily-loss circuit breaker triggers, evaluates gain-taking opportunities, and synchronizes commitments with the database to detect closed commitments and trigger reserve pool siphoning.
|
||||
|
||||
4. **`_risk_tier_scheduler()`** — Runs once daily at 16:00 ET (session close). It loads the latest `PerformanceMetrics` from `pool_snapshots`, computes the reserve pool as a fraction of total resource pool value, and delegates to the `RiskTierController` in `services/trading/risk_tier_controller.py` to determine whether the active risk tier should change. Tier changes are persisted to `risk_tier_history` and take effect immediately for subsequent decision cycles.
|
||||
|
||||
5. **`_rebalance_scheduler()`** — Runs weekly on Monday at 09:45 ET (shortly after session open). It loads current commitments, evaluates them against the active risk tier's constraints using the `PoolRebalancer`, and pushes any rebalance defer execution requests to `app:queue:execution_orders`. The rebalancer respects the circuit breaker — if any breaker is active, the rebalance cycle is skipped entirely.
|
||||
|
||||
All five tasks run concurrently within a single `asyncio` event loop. Graceful shutdown via `stop()` cancels all tasks and awaits their completion. If any task encounters an unexpected exception, it logs the error and retries after a brief sleep rather than crashing the engine.
|
||||
|
||||
---
|
||||
|
||||
## Pre-Execution Check Sequence
|
||||
|
||||
When the decision loop picks up an act recommendation, it calls `evaluate_recommendation()` — a synchronous method that runs the full pre-execution check sequence. The checks are applied in a strict order, and the first failure short-circuits the evaluation with a `skip` decision. This fail-fast design ensures that expensive downstream computations (like commitment sizing and correlation analysis) are never reached when a simple gate would have rejected the decision.
|
||||
|
||||
The six checks, in order:
|
||||
|
||||
**a. Circuit breaker check.** The engine calls `self.circuit_breaker.is_active()` on the current `CircuitBreakerState`. If any circuit breaker is active and its cooldown has not expired, the recommendation is skipped with reason `circuit_breaker_active`. The circuit breaker mechanism is described in detail below.
|
||||
|
||||
**b. Execution window check.** The `is_within_execution_window()` function verifies that the current time falls within the active session hours. Outside the execution window, no execution requests are submitted — the recommendation is skipped with reason `outside_execution_window`.
|
||||
|
||||
**c. Confidence gate.** The recommendation's confidence score is compared against the active risk tier's `min_confidence` threshold. A conservative tier requires confidence ≥ 0.75, moderate requires ≥ 0.55, and aggressive requires ≥ 0.40. If the recommendation's confidence falls below the tier minimum, it is skipped with reason `insufficient_confidence`. This gate ensures that the risk tier's conservatism is enforced before any resource allocation is considered.
|
||||
|
||||
**d. Deduplication check.** The engine maintains an in-memory set of processed recommendation IDs (`processed_recommendation_ids`) and also checks Redis via `app:dedupe:execution:*` keys (with a 24-hour TTL). If the recommendation has already been evaluated in this engine session or by a previous instance, it is skipped with reason `duplicate_recommendation`. This prevents the same recommendation from generating multiple execution requests across polling cycles.
|
||||
|
||||
**e. Declining commitments check.** The `check_declining_commitments()` method examines all active commitments. If more than 50% of commitments have unrealized losses exceeding 2% of their entry value, the engine halts new entries with reason `multiple_declining_commitments`. This is a pool-level safety valve — when the majority of existing commitments are underwater, adding new exposure compounds the risk.
|
||||
|
||||
**f. Max active commitments check.** The engine enforces a configurable maximum number of concurrent commitments (default 10). If the resource pool is already at capacity, the recommendation is skipped with reason `max_commitments_reached`.
|
||||
|
||||
For defer recommendations, the engine follows a separate, simpler path: it verifies the execution window, looks up the existing commitment for the entity, and submits a full-quantity defer execution request without running the commitment sizer. Defer decisions still generate an audit record in `execution_decisions` and set the Redis deduplication key.
|
||||
|
||||
If all six checks pass for an act recommendation, the engine proceeds to commitment sizing.
|
||||
|
||||
---
|
||||
|
||||
## Commitment Sizing
|
||||
|
||||
The `CommitmentSizer` in `services/trading/position_sizer.py` translates a recommendation's signal quality into a concrete dollar amount and unit count, applying a sequential pipeline of adjustments that account for confidence, pool composition, sector concentration, correlation, and upcoming performance report events. The sizer operates on the *active pool* — the portion of the resource pool available for execution after subtracting the reserve pool balance.
|
||||
|
||||
### Base Sizing
|
||||
|
||||
The computation begins with a base allocation percentage derived from the risk tier:
|
||||
|
||||
```
|
||||
base_allocation_pct = risk_tier.max_position_pct × 0.5
|
||||
raw_pct = base_allocation_pct × (confidence / min_confidence)
|
||||
```
|
||||
|
||||
The base starts at half the tier's maximum commitment percentage, then scales linearly with how far the recommendation's confidence exceeds the tier minimum. A moderate-tier recommendation with confidence 0.70 against a minimum of 0.55 would produce a raw percentage of `0.05 × (0.70 / 0.55) ≈ 0.0636`, or about 6.4% of the active pool. The raw percentage is clamped to `max_position_pct` (5% for conservative, 10% for moderate, 15% for aggressive) and then converted to a dollar amount against the active pool. An absolute commitment cap (default $50) provides a hard ceiling regardless of pool size — a safety measure for the simulation mode environment.
|
||||
|
||||
### Correlation-Aware Diversification
|
||||
|
||||
The sizer computes a weighted average correlation between the candidate entity and all existing commitments, using the pairwise correlation matrix that the engine refreshes from 30 days of daily close values in `market_snapshots`. Each existing commitment's correlation is weighted by its value, so larger commitments have more influence on the diversification check.
|
||||
|
||||
If the weighted average correlation exceeds 0.8, the commitment is rejected outright — the resource pool already has too much exposure to correlated assets. Between 0.5 and 0.8, the dollar amount is reduced proportionally: a correlation of 0.65 produces a scale factor of `1.0 − (0.65 − 0.5) / (0.8 − 0.5) = 0.5`, halving the commitment size. Below 0.5, no reduction is applied.
|
||||
|
||||
### Sector Exposure Reduction
|
||||
|
||||
The sizer checks whether adding the new commitment would push the sector's total exposure beyond the risk tier's `max_sector_pct` (20% for conservative, 30% for moderate, 40% for aggressive). If the sector is already at its limit, the commitment is rejected. If the new commitment would exceed the limit, the dollar amount is reduced to exactly fill the remaining sector capacity.
|
||||
|
||||
### Diversification Bonus
|
||||
|
||||
When the resource pool holds fewer than three distinct sectors and the candidate entity belongs to a new sector, the sizer applies a 1.2× bonus to the dollar amount. This incentivizes early diversification — the first few commitments are encouraged to spread across sectors rather than concentrating in a single one. The bonus is re-clamped to `max_position_pct` after application to prevent oversized commitments.
|
||||
|
||||
### Performance Report Proximity Adjustment
|
||||
|
||||
The sizer checks the performance report calendar for the candidate entity. If a performance report is within one active session, the commitment is rejected entirely — the binary risk of a disclosure surprise is too high for automated entry. If a performance report is within three active sessions, the dollar amount is reduced by 50%. Beyond three sessions, no adjustment is applied.
|
||||
|
||||
### Pool Exposure Check and Unit Rounding
|
||||
|
||||
After all adjustments, the sizer estimates the new commitment's contribution to pool exposure (the aggregate risk from risk threshold distances across all commitments). If adding the commitment would push total exposure beyond `max_portfolio_heat × active_pool` (10% for conservative, 20% for moderate, 30% for aggressive), the commitment is rejected.
|
||||
|
||||
Finally, the dollar amount is converted to whole units via `floor(dollar_amount / current_value)`. If rounding produces zero units (the commitment is too small for even one unit at the current value), the commitment is rejected. The final dollar amount is recalculated from the whole-unit quantity to reflect the actual capital deployed.
|
||||
|
||||
The `CommitmentSizeResult` returned to the engine includes the dollar amount, unit quantity, allocation percentage, a list of human-readable adjustment notes, and a rejected flag with reason if any step failed. These adjustment notes are embedded in the decision record's `decision_trace` for full auditability.
|
||||
|
||||
---
|
||||
|
||||
## Circuit Breaker
|
||||
|
||||
The `CircuitBreaker` in `services/trading/circuit_breaker.py` is a pure computation module that evaluates three independent trigger conditions. It carries no state of its own — the engine manages the `CircuitBreakerState` dataclass and persists trigger events to the `circuit_breaker_events` table and Redis keys under `app:execution:circuit_breaker:*`.
|
||||
|
||||
### Three Trigger Types
|
||||
|
||||
**Daily loss trigger.** When the resource pool's daily gain/loss exceeds 5% of total resource pool value (`daily_loss_pct = 0.05`), the circuit breaker activates. The `check_daily_loss()` method compares the absolute loss ratio against the threshold. The cooldown duration is set to `volatility_pause_hours` (default 2 hours). The performance loop in the engine calls `_check_circuit_breaker_daily_loss()` periodically to evaluate this condition against the latest pool metrics. In extreme cases where the peak-to-trough decline exceeds an emergency threshold, the reserve pool's emergency liquidation mechanism may also be triggered.
|
||||
|
||||
**Single commitment loss trigger.** When any individual commitment loses more than 15% of its entry value (`single_position_loss_pct = 0.15`), the circuit breaker activates with an entity-specific cooldown. The `check_single_position()` method evaluates the loss percentage. The cooldown for the affected entity is set to `ticker_cooldown_hours` (default 48 hours), during which the engine will not re-enter that entity. The `is_ticker_cooled_down()` method checks whether a specific entity is still within its cooldown window by consulting the `ticker_cooldowns` dictionary in the `CircuitBreakerState`.
|
||||
|
||||
**Volatility trigger (risk threshold clustering).** When three or more risk thresholds fire within a 30-minute rolling window (`stop_loss_hits_threshold = 3`, `stop_loss_window_minutes = 30`), the circuit breaker activates. The `check_volatility()` method uses a sliding window algorithm: it sorts the risk threshold timestamps and checks every contiguous subsequence of length `stop_loss_hits_threshold` to see if it fits within the window. This detects rapid-fire risk threshold cascades that indicate extreme volatility. The cooldown is `volatility_pause_hours` (default 2 hours).
|
||||
|
||||
### Cooldown Computation
|
||||
|
||||
The `compute_cooldown_expiry()` method calculates when a triggered breaker expires. For `daily_loss` and `volatility` triggers, the expiry is `triggered_at + volatility_pause_hours`. For `single_position` triggers, the expiry is `triggered_at + ticker_cooldown_hours`, giving the affected entity a longer cooling-off period. The `is_active()` method returns `True` when the breaker is flagged active and the current time has not yet passed the cooldown expiry.
|
||||
|
||||
### Redis State Tracking
|
||||
|
||||
The engine persists circuit breaker state to Redis under the `app:execution:circuit_breaker:*` key pattern (constructed by `execution_cb_key()` in `services/shared/redis_keys.py`). Each trigger type gets its own key — for example, `app:execution:circuit_breaker:daily_loss` — storing the activation timestamp and cooldown expiry. This allows the state to survive engine restarts and enables external monitoring tools to query breaker status without accessing the engine's memory.
|
||||
|
||||
---
|
||||
|
||||
## Reserve Pool
|
||||
|
||||
The `ReservePoolController` in `services/trading/reserve_pool.py` manages an untouchable cash reserve that grows from realized execution gains. The reserve serves two purposes: it provides a buffer against peak-to-trough declines, and its size relative to the resource pool influences risk tier upgrade decisions.
|
||||
|
||||
### Profit Siphoning
|
||||
|
||||
When the engine detects a closed commitment with positive unrealized gain/loss (via `_sync_commitments_and_siphon()` in the performance loop), it calls `siphon_profit()` on the controller. The method transfers a configurable fraction of the realized gain into the reserve — by default 20% (`siphon_pct = 0.20`). Only positive gains are siphoned; losses do not reduce the reserve balance. Each siphon event is recorded in the `reserve_pool_ledger` table with the transfer amount, resulting balance, trigger type (`profit_siphon`), the entity as reference, and a timestamp.
|
||||
|
||||
### High-Water Mark Rebalancing
|
||||
|
||||
The `is_high_water()` method returns `True` when the reserve balance exceeds 30% of total resource pool value (`high_water_pct = 0.30`). This signal is consumed by the risk tier scheduler — when the reserve is healthy and other performance criteria are met, the controller may recommend upgrading to a more aggressive tier. The high-water mark acts as a confidence indicator: a large reserve means the system has been consistently successful and can afford to take on more risk.
|
||||
|
||||
### Emergency Liquidation
|
||||
|
||||
The `should_emergency_liquidate()` method checks whether the current peak-to-trough decline exceeds an emergency threshold. When triggered, `emergency_liquidate()` returns the full reserve balance for release back into the active pool. The caller (the engine) is responsible for zeroing the persisted balance and recording the ledger entry. Emergency liquidation is a last resort — it sacrifices the safety buffer to prevent the resource pool from hitting a catastrophic loss level.
|
||||
|
||||
### Active Pool Computation
|
||||
|
||||
The `compute_active_pool()` method calculates the capital available for execution: `active_pool = total_pool_value − reserve_balance`. All commitment sizing computations use the active pool rather than the total resource pool value, ensuring that the reserve is never inadvertently deployed into new commitments.
|
||||
|
||||
---
|
||||
|
||||
## Risk Tier Auto-Adjustment
|
||||
|
||||
The `RiskTierController` in `services/trading/risk_tier_controller.py` evaluates resource pool performance and determines whether the active risk tier should shift. The system supports three tiers — conservative, moderate, and aggressive — each defined by a `RiskTierConfig` dataclass in `services/trading/models.py` with distinct parameter values:
|
||||
|
||||
| Parameter | Conservative | Moderate | Aggressive |
|
||||
|-----------|-------------|----------|------------|
|
||||
| `min_confidence` | 0.75 | 0.55 | 0.40 |
|
||||
| `max_position_pct` | 5% | 10% | 15% |
|
||||
| `stop_loss_atr_multiplier` | 1.5× | 2.0× | 2.5× |
|
||||
| `reward_risk_ratio` | 2.0 | 1.5 | 1.2 |
|
||||
| `max_sector_pct` | 20% | 30% | 40% |
|
||||
| `max_portfolio_heat` | 10% | 20% | 30% |
|
||||
|
||||
The tier controller's `evaluate()` method checks two conditions:
|
||||
|
||||
**Downgrade (any one triggers).** If the trailing 30-day success rate drops below 40% or the current peak-to-trough decline exceeds 15%, the tier steps down by one level (e.g., aggressive → moderate). If the system is already at conservative, no further downgrade is possible.
|
||||
|
||||
**Upgrade (all must be true).** If the success rate exceeds 55%, the reserve pool exceeds 20% of total resource pool value, and the current peak-to-trough decline is below 5%, the tier steps up by one level. The triple requirement ensures that upgrades only happen when the system is performing well, has built a safety cushion, and is not in a decline.
|
||||
|
||||
The risk tier scheduler in the engine evaluates these conditions daily at session close. When a tier change occurs, it is persisted to the `risk_tier_history` table with the previous tier, new tier, trigger source (`auto_adjustment`), and the metrics that drove the decision (success rate, peak-to-trough decline, reserve percentage, risk-adjusted return ratio). The new tier takes effect immediately — the engine updates its `_active_risk_tier` reference, and all subsequent decision cycles use the new tier's parameters for confidence gates, commitment sizing, risk threshold computation, and sector exposure limits.
|
||||
|
||||
---
|
||||
|
||||
## Execution Request Submission Flow
|
||||
|
||||
When `evaluate_recommendation()` returns an `act` decision, the engine constructs an execution request job and pushes it through a multi-stage submission pipeline that spans two services.
|
||||
|
||||
### Decision Persistence
|
||||
|
||||
Every evaluation — whether it results in `act` or `skip` — produces a decision record that is persisted to the `execution_decisions` table via `_persist_decision()`. The record captures the recommendation ID, decision outcome, skip reason (if applicable), entity identifier, computed commitment size and unit quantity, the risk tier at the time of decision, pool exposure, active pool and reserve pool balances, circuit breaker status, correlation and sector exposure check results, performance report proximity flag, and a `decision_trace` JSONB field containing the full reasoning chain. This creates a complete audit record of every recommendation the engine evaluated and why it acted or declined.
|
||||
|
||||
### Execution Request Enqueue
|
||||
|
||||
For `act` decisions, the engine builds an execution request job dictionary containing the decision ID, entity identifier, action (act or defer), quantity, and request type (immediate). This job is pushed via `rpush` to the `app:queue:execution_orders` Redis queue (constructed by `queue_key(QUEUE_BROKER)` from `services/shared/redis_keys.py`). The engine immediately deducts the estimated execution cost from the in-memory active pool to prevent over-allocation across concurrent recommendation evaluations within the same polling cycle.
|
||||
|
||||
### Execution Service Processing
|
||||
|
||||
The execution service in `services/adapters/broker_service.py` runs as a standalone worker that polls `app:queue:execution_orders` via `blpop`. For each execution request job, `process_order_job()` executes a multi-step pipeline:
|
||||
|
||||
1. **Idempotency check.** A deterministic idempotency key is generated from the job's entity identifier, action, quantity, and decision ID. The service checks Redis first (fast path) and then the `orders` table (durable fallback) to prevent duplicate submissions. If a matching key exists, the job is silently dropped.
|
||||
|
||||
2. **Risk evaluation.** The service loads the current `PoolRiskConfig` from the database and the account's risk state (active commitments, daily gain/loss, sector exposure) from both the database and the external execution API. The `evaluate_order()` function runs the proposed execution request through a set of risk checks — commitment limits, sector concentration, daily loss thresholds — and produces an evaluation result. The evaluation is persisted to the `risk_evaluations` table regardless of outcome.
|
||||
|
||||
3. **External API submission.** If the risk evaluation passes, the service calls `submit_order()` on the `ExecutionAdapter` in `services/adapters/broker_adapter.py`. The adapter constructs the external execution API payload (entity identifier, quantity, side, request type, time in force) and submits it to `execution-api.example.com/v2/orders` with an idempotency key header. The adapter follows a fail-closed policy: any network error or ambiguous response returns a rejected `ExecutionResponse` rather than risking duplicate execution requests.
|
||||
|
||||
4. **Persistence and audit trail.** The `persist_order()` function writes the execution request to the `orders` table with the full request and response details, risk evaluation results, and the recommendation ID for traceability. When the execution request is filled, the fill details (value, quantity) are recorded. Execution request events are published to the analytical lakehouse via MinIO for downstream analysis. The Redis idempotency marker is set after successful persistence to prevent reprocessing.
|
||||
|
||||
The result is a complete chain of custody: from the original document that produced a signal (Pages [1](01-data-ingestion-and-preparation.md)–[2](02-ai-agent-processing-and-extraction.md)), through signal scoring ([Page 3](03-signal-scoring-and-weighted-signals.md)) and trend aggregation ([Page 4](04-trend-aggregation-and-accumulating-signals.md)), to the recommendation ([Page 5](05-recommendation-generation.md)), the execution decision, the risk evaluation, and the execution response — every step is persisted and linked by foreign keys. The `execution_decisions` table links to `recommendations` via `recommendation_id`, the `orders` table links back to both, and the `commitments` and `pool_snapshots` tables capture the resource pool impact over time.
|
||||
|
||||
For additional reference on the decision execution engine's configuration, queue topology, and database tables, see [docs/services.md](../services.md).
|
||||
|
||||
---
|
||||
|
||||
## Conclusion: From Raw Data to Decision Execution
|
||||
|
||||
This six-page series has traced the full intelligence-to-decision pipeline, from the moment raw data enters the system to the moment an execution request reaches the external execution API.
|
||||
|
||||
It began with [Page 1](01-data-ingestion-and-preparation.md), where the scheduler orchestrates ingestion cycles across four data sources — external news, regulatory filings, external data feeds, and macro news APIs — and the parser normalizes raw content into structured documents ready for AI processing. [Page 2](02-ai-agent-processing-and-extraction.md) described how the Document Intelligence Extractor and Global Event Classifier agents use LLM inference to produce structured JSON intelligence, with hot-swappable model configurations and a robust JSON repair pipeline. [Page 3](03-signal-scoring-and-weighted-signals.md) explained how raw extraction output is transformed into `WeightedSignal` objects through a composite formula that balances recency, credibility, novelty, and environmental context across three independent signal layers. [Page 4](04-trend-aggregation-and-accumulating-signals.md) showed how the aggregation engine merges these signals across five time windows, detecting contradictions, ranking evidence, and computing trend projections — with consecutive same-direction signals accumulating to escalate the system's response from neutral through observe and monitor to act or defer. [Page 5](05-recommendation-generation.md) covered the translation of trend assessments into actionable recommendations through data quality suppression, eligibility evaluation, commitment sizing, thesis generation, and risk classification.
|
||||
|
||||
And here in Page 6, the pipeline reached its terminus: the decision execution engine's decision loop polling those recommendations, subjecting each to circuit breaker checks, confidence gates, deduplication, pool health assessments, and a multi-step commitment sizer — then submitting approved execution requests through the execution adapter to the external execution API, with every decision recorded in a fully auditable trail from signal to execution.
|
||||
|
||||
The pipeline is designed to be conservative by default and transparent throughout. Every stage applies its own safety checks — deduplication at ingestion, confidence gates at extraction, contradiction detection at aggregation, suppression at recommendation, and circuit breakers at execution. The system can be tuned through runtime configuration (risk tier parameters, suppression thresholds, signal layer toggles in `risk_configs`) without code changes or restarts. And the complete audit trail — from `documents` through `document_intelligence`, `document_impact_records`, `trend_windows`, `recommendations`, `execution_decisions`, and `orders` — means that any decision can be traced back to the specific documents, signals, and evaluations that produced it.
|
||||
@@ -0,0 +1 @@
|
||||
|
||||
@@ -0,0 +1,94 @@
|
||||
# Decision Execution Engine Loop
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
subgraph ENGINE["Decision Execution Engine\nservices/trading/engine.py"]
|
||||
direction TB
|
||||
TASKS["5 Concurrent Async Tasks"]
|
||||
T1["_decision_loop()\n60s polling interval"]
|
||||
T2["_risk_threshold_monitor()"]
|
||||
T3["_performance_loop()"]
|
||||
T4["_risk_tier_scheduler()"]
|
||||
T5["_rebalance_scheduler()"]
|
||||
TASKS --> T1 & T2 & T3 & T4 & T5
|
||||
end
|
||||
|
||||
T1 --> POLL["Poll recommendations table\naction IN (act, defer)\nmode IN (simulation_eligible, production_eligible)\ngenerated_at > NOW() − 2h"]
|
||||
|
||||
POLL --> EVAL["evaluate_recommendation()"]
|
||||
|
||||
EVAL --> CHK_A
|
||||
|
||||
subgraph PRETRADE["Pre-Execution Check Sequence\n(first failure short-circuits)"]
|
||||
direction TB
|
||||
CHK_A["a. Circuit Breaker active?\nservices/trading/circuit_breaker.py\nTriggers: daily_loss, single_commitment, volatility"]
|
||||
CHK_B["b. Execution Window?\nis_within_execution_window()"]
|
||||
CHK_C["c. Confidence Gate\nconfidence ≥ risk_tier.min_confidence"]
|
||||
CHK_D["d. Deduplication\nRec ID in processed set?\nRedis: app:dedupe:execution:*"]
|
||||
CHK_E["e. Declining Commitments\n> 50% commitments down > 2%"]
|
||||
CHK_F["f. Max Open Commitments\nopen_count ≥ max (default 10)"]
|
||||
|
||||
CHK_A -->|"pass"| CHK_B
|
||||
CHK_B -->|"pass"| CHK_C
|
||||
CHK_C -->|"pass"| CHK_D
|
||||
CHK_D -->|"pass"| CHK_E
|
||||
CHK_E -->|"pass"| CHK_F
|
||||
end
|
||||
|
||||
CHK_A & CHK_B & CHK_C & CHK_D & CHK_E & CHK_F -->|"fail"| SKIP["ExecutionDecision\ndecision = skip\n+ skip_reason"]
|
||||
|
||||
CHK_F -->|"pass"| SIZER
|
||||
|
||||
subgraph SIZER["Commitment Sizing\nservices/trading/position_sizer.py"]
|
||||
direction TB
|
||||
SZ1["Base sizing\nrisk_tier.max_commitment_pct × 0.5\n× (confidence / min_confidence)"]
|
||||
SZ2["Correlation reduction\nweighted avg corr > 0.8 → reject\n> 0.5 → proportional reduction"]
|
||||
SZ3["Sector exposure\ncap at risk_tier.max_sector_pct"]
|
||||
SZ4["Diversification bonus\n1.2× for new sector (< 3 sectors)"]
|
||||
SZ5["Event proximity\n≤ 1 day → reject\n≤ 3 days → 50% reduction"]
|
||||
SZ6["Absolute commitment cap"]
|
||||
SZ7["Pool exposure check\nmax_pool_exposure × active_pool"]
|
||||
SZ8["Share rounding\nfloor(dollar / price)"]
|
||||
|
||||
SZ1 --> SZ2 --> SZ3 --> SZ4 --> SZ5 --> SZ6 --> SZ7 --> SZ8
|
||||
end
|
||||
|
||||
SIZER -->|"rejected"| SKIP
|
||||
SIZER -->|"approved"| ACT["ExecutionDecision\ndecision = act\nshares, dollar amount"]
|
||||
|
||||
ACT --> PERSIST_TD["Persist to\nexecution_decisions"]
|
||||
|
||||
ACT --> ORDER["Build execution request\n{entity, action, side,\nquantity, request_type}"]
|
||||
|
||||
ORDER -->|"rpush"| Q_BROKER["app:queue:execution_orders"]
|
||||
|
||||
Q_BROKER --> BROKER["Execution Adapter\nexternal execution API (simulation)\nservices/adapters/broker_adapter.py"]
|
||||
|
||||
BROKER --> AUDIT
|
||||
|
||||
subgraph AUDIT["Audit Trail — PostgreSQL"]
|
||||
AU1["execution_requests"]
|
||||
AU2["commitments"]
|
||||
AU3["pool_snapshots"]
|
||||
end
|
||||
|
||||
subgraph CB_DETAIL["Circuit Breaker Detail\nservices/trading/circuit_breaker.py"]
|
||||
CB1["daily_loss\npool loss > 5%\ncooldown: volatility_pause_hours"]
|
||||
CB2["single_commitment\ncommitment loss > 15%\ncooldown: entity_cooldown_hours (48h)"]
|
||||
CB3["volatility\n≥ 3 risk thresholds in 30min\ncooldown: volatility_pause_hours (2h)"]
|
||||
CB4["Redis state\napp:execution:circuit_breaker:*"]
|
||||
end
|
||||
|
||||
subgraph RESERVE["Reserve Pool\nservices/trading/reserve_pool.py"]
|
||||
RP1["Profit siphoning: 20%"]
|
||||
RP2["High-water rebalance: 30%"]
|
||||
RP3["Emergency liquidation"]
|
||||
RP4["reserve_pool_ledger"]
|
||||
end
|
||||
|
||||
subgraph RISK_TIER["Risk Tier Auto-Adjustment\nservices/trading/risk_tier_controller.py"]
|
||||
RT1["Evaluate: risk-adjusted return ratio,\npeak-to-trough decline, success rate"]
|
||||
RT2["conservative → moderate → aggressive"]
|
||||
RT3["risk_tier_history"]
|
||||
end
|
||||
```
|
||||
@@ -0,0 +1,81 @@
|
||||
# Ingestion-to-Extraction Flow
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
subgraph Scheduler["Scheduler\nservices/scheduler/app.py"]
|
||||
S1["schedule_cycle()"]
|
||||
S2["Cadence check\nmarket_api: 300s\nnews_api: 300s\nfilings_api: 3600s\nmacro_news: 600s"]
|
||||
S3["Rate limit check\ncheck_rate_limit()"]
|
||||
S1 --> S2 --> S3
|
||||
end
|
||||
|
||||
S3 -->|"rpush"| Q_ING["app:queue:ingestion"]
|
||||
|
||||
Q_ING -->|"lpop"| ING
|
||||
|
||||
subgraph ING["Ingestion Worker\nservices/ingestion/worker.py"]
|
||||
direction TB
|
||||
AD["Adapter Dispatch\nprocess_job()"]
|
||||
AD --> PA["ExternalDataAdapter\nservices/adapters/market_adapter.py"]
|
||||
AD --> PB["ExternalNewsAdapter\nservices/adapters/news_adapter.py"]
|
||||
AD --> PC["RegulatoryFilingsAdapter\nservices/adapters/filings_adapter.py"]
|
||||
AD --> PD["MacroNewsAdapter\nservices/adapters/macro_news_adapter.py"]
|
||||
AD --> PE["WebScrapeAdapter\nservices/adapters/web_scrape_adapter.py"]
|
||||
end
|
||||
|
||||
ING -->|"Content hash check\napp:dedupe:*\nTTL 24h"| REDIS_DEDUPE[("Redis\nDedupe Markers")]
|
||||
|
||||
ING -->|"upload_raw_artifact()"| MINIO_RAW
|
||||
|
||||
subgraph MINIO_RAW["MinIO Raw Storage"]
|
||||
B1["app-raw-data"]
|
||||
B2["app-raw-content"]
|
||||
B3["app-raw-filings"]
|
||||
end
|
||||
|
||||
ING -->|"persist_ingestion_items()"| PG_ING
|
||||
|
||||
subgraph PG_ING["PostgreSQL"]
|
||||
T1["documents"]
|
||||
T2["ingestion_runs"]
|
||||
T3["document_company_mentions"]
|
||||
end
|
||||
|
||||
ING -->|"rpush new doc IDs"| Q_PARSE["app:queue:parsing"]
|
||||
|
||||
Q_PARSE -->|"lpop"| PARSER
|
||||
|
||||
subgraph PARSER["Parser Worker\nservices/parser/worker.py"]
|
||||
P1["fetch_html() → parse_html()"]
|
||||
P2["Quality scoring\nconfidence: high / medium / low"]
|
||||
P3["Company mention detection\ndetect_company_mentions()"]
|
||||
P4["Routing decision"]
|
||||
P1 --> P2 --> P3 --> P4
|
||||
end
|
||||
|
||||
PARSER -->|"upload_normalized_text()\nupload_parser_output()"| MINIO_NORM["MinIO\napp-normalized"]
|
||||
PARSER -->|"update_document_parse_results()"| PG_ING
|
||||
|
||||
P4 -->|"doc_type = macro_event"| Q_MACRO["app:queue:macro_classification"]
|
||||
P4 -->|"doc_type ≠ macro_event"| Q_EXT["app:queue:extraction"]
|
||||
|
||||
Q_EXT -->|"lpop"| EXT
|
||||
Q_MACRO -->|"lpop"| EXT
|
||||
|
||||
subgraph EXT["Extractor Worker\nservices/extractor/main.py"]
|
||||
E1["Document Intelligence\nExtractor agent\nslug: document-extractor"]
|
||||
E2["Global Event Classifier\nslug: event-classifier\nservices/extractor/event_classifier.py"]
|
||||
E3["persist_extraction()\nservices/extractor/worker.py"]
|
||||
end
|
||||
|
||||
EXT -->|"persist to"| PG_EXT
|
||||
|
||||
subgraph PG_EXT["PostgreSQL"]
|
||||
T4["document_intelligence"]
|
||||
T5["document_impact_records"]
|
||||
T6["global_events"]
|
||||
T7["macro_impact_records"]
|
||||
end
|
||||
|
||||
EXT -->|"rpush"| Q_AGG["app:queue:aggregation"]
|
||||
```
|
||||
@@ -0,0 +1,80 @@
|
||||
# Recommendation Generation Flow
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Q_REC["app:queue:recommendation"] -->|"lpop"| WORKER["Recommendation Worker\nservices/recommendation/main.py"]
|
||||
|
||||
WORKER --> FETCH["Fetch TrendSummary\nfrom trend_windows\nfor entity + window"]
|
||||
|
||||
FETCH --> SUPP
|
||||
|
||||
subgraph SUPP["Data Quality Suppression\nservices/recommendation/suppression.py"]
|
||||
S1["extraction confidence < 0.40?"]
|
||||
S2["evidence staleness > 168h?"]
|
||||
S3["source diversity < 1 type?"]
|
||||
S4["extraction failure rate > 50%?"]
|
||||
S5["valid documents < 2?"]
|
||||
S6["data quality score < 0.30?"]
|
||||
S7["Macro-only signal?\nevaluate_macro_only_suppression()"]
|
||||
S8["Pattern-only signal?\nevaluate_pattern_only_suppression()"]
|
||||
end
|
||||
|
||||
SUPP -->|"Any check fails:\nsuppressed = true\nmode → informational"| ELIG
|
||||
SUPP -->|"All checks pass"| ELIG
|
||||
|
||||
subgraph ELIG["Eligibility Evaluation\nservices/recommendation/eligibility.py"]
|
||||
direction TB
|
||||
G["Gate Checks"]
|
||||
G1["confidence ≥ 0.35"]
|
||||
G2["strength ≥ 0.10"]
|
||||
G3["contradiction ≤ 0.60"]
|
||||
G4["evidence ≥ 2"]
|
||||
G5["direction ≠ neutral"]
|
||||
G --> G1 & G2 & G3 & G4 & G5
|
||||
|
||||
G1 & G2 & G3 & G4 & G5 --> ACT["Action Mapping"]
|
||||
ACT --> A1["ACT: positive + strength ≥ 0.25"]
|
||||
ACT --> A2["DEFER: negative + strength ≥ 0.25"]
|
||||
ACT --> A3["MONITOR: directional + confidence ≥ 0.50"]
|
||||
ACT --> A4["OBSERVE: otherwise"]
|
||||
|
||||
A1 & A2 & A3 & A4 --> MODE["Mode Escalation"]
|
||||
MODE --> M1["informational\n(default for MONITOR/OBSERVE)"]
|
||||
MODE --> M2["simulation_eligible\nconfidence ≥ 0.50"]
|
||||
MODE --> M3["production_eligible\nconfidence ≥ 0.70\ncontradiction ≤ 0.25\nevidence ≥ 5"]
|
||||
end
|
||||
|
||||
ELIG --> SIZING
|
||||
|
||||
subgraph SIZING["Commitment Sizing\nservices/recommendation/eligibility.py"]
|
||||
PS1["base = 1% allocation pool"]
|
||||
PS2["scale by confidence × strength\nup to 10% max"]
|
||||
PS3["contradiction penalty\n−0.5 × contradiction_score"]
|
||||
PS4["evidence count penalty\n< 3 docs → ×0.5\n< 5 docs → ×0.75"]
|
||||
end
|
||||
|
||||
SIZING --> THESIS
|
||||
|
||||
subgraph THESIS["Thesis Generation"]
|
||||
TH1["Deterministic thesis\nassembled from trend data"]
|
||||
TH2["Optional LLM rewrite\nthesis-rewriter agent\nservices/recommendation/thesis_llm.py"]
|
||||
TH1 --> TH2
|
||||
end
|
||||
|
||||
THESIS --> RISK
|
||||
|
||||
subgraph RISK["Risk Classification"]
|
||||
RC1["low"]
|
||||
RC2["moderate"]
|
||||
RC3["high"]
|
||||
RC4["very_high"]
|
||||
end
|
||||
|
||||
RISK --> PERSIST
|
||||
|
||||
subgraph PERSIST["Persistence — PostgreSQL"]
|
||||
P1["recommendations"]
|
||||
P2["recommendation_evidence"]
|
||||
P3["risk_evaluations"]
|
||||
end
|
||||
```
|
||||
@@ -0,0 +1,52 @@
|
||||
# Three-Layer Signal Merging
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
subgraph Layer1["Layer 1 — Entity Signals"]
|
||||
DIR["document_impact_records\n(per-entity extraction output)"]
|
||||
DIR -->|"build_weighted_signals()"| WS1["WeightedSignal[]\nweight = 1.0 (full)"]
|
||||
end
|
||||
|
||||
subgraph Layer2["Layer 2 — Macro Signals"]
|
||||
MIR["macro_impact_records\n(global event interpolation)"]
|
||||
MIR -->|"build_macro_weighted_signals()"| WS2["WeightedSignal[]\nimpact × MACRO_SIGNAL_WEIGHT\n(0.3)"]
|
||||
TOGGLE_M{"macro_enabled\nin risk_configs?"}
|
||||
TOGGLE_M -->|"true"| MIR
|
||||
TOGGLE_M -->|"false"| SKIP_M["Layer skipped\ngraceful degradation"]
|
||||
end
|
||||
|
||||
subgraph Layer3["Layer 3 — Competitive Signals"]
|
||||
CSR["competitive_signal_records\n(pattern mining + propagation)"]
|
||||
CSR -->|"build_pattern_weighted_signals()\nservices/aggregation/signal_propagation.py"| WS3["WeightedSignal[]\nimpact × COMPETITIVE_SIGNAL_WEIGHT\n(0.2)"]
|
||||
TOGGLE_C{"competitive_enabled\nin risk_configs?"}
|
||||
TOGGLE_C -->|"true"| CSR
|
||||
TOGGLE_C -->|"false"| SKIP_C["Layer skipped\ngraceful degradation"]
|
||||
end
|
||||
|
||||
WS1 --> MERGE["Concatenate all WeightedSignal lists"]
|
||||
WS2 --> MERGE
|
||||
WS3 --> MERGE
|
||||
|
||||
MERGE --> AGG
|
||||
|
||||
subgraph AGG["Aggregation Engine\nservices/aggregation/worker.py"]
|
||||
A1["weighted_sentiment_average()"]
|
||||
A2["detect_contradictions()\nservices/aggregation/contradiction.py"]
|
||||
A3["derive_trend_direction()"]
|
||||
A4["compute_trend_confidence()"]
|
||||
A5["rank_evidence()"]
|
||||
A1 --> A2 --> A3 --> A4 --> A5
|
||||
end
|
||||
|
||||
AGG -->|"assemble_trend_summary()"| TS["TrendSummary\nservices/shared/schemas.py"]
|
||||
|
||||
TS -->|"persist_trend_summary()"| PG_TREND
|
||||
|
||||
subgraph PG_TREND["PostgreSQL"]
|
||||
TW["trend_windows\n(upserted each cycle)"]
|
||||
TH["trend_history\n(time-series snapshots)"]
|
||||
TE["trend_evidence\n(per-document rankings)"]
|
||||
end
|
||||
|
||||
AGG -->|"rpush"| Q_REC["app:queue:recommendation"]
|
||||
```
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user