book: scaffold + ch00 (execution trail as spine) — evidence exp 8-31, round 3
This commit is contained in:
+163
@@ -0,0 +1,163 @@
|
||||
# =========================================================
|
||||
# OS / Editor
|
||||
# =========================================================
|
||||
.DS_Store
|
||||
Thumbs.db
|
||||
|
||||
.vscode/
|
||||
.idea/
|
||||
*.swp
|
||||
*.swo
|
||||
*~
|
||||
|
||||
# =========================================================
|
||||
# Environment & Secrets
|
||||
# =========================================================
|
||||
.env
|
||||
.env.*
|
||||
!.env.example
|
||||
|
||||
*.pem
|
||||
*.key
|
||||
*.crt
|
||||
secrets/
|
||||
|
||||
# =========================================================
|
||||
# Logs
|
||||
# =========================================================
|
||||
logs/
|
||||
*.log
|
||||
npm-debug.log*
|
||||
yarn-debug.log*
|
||||
yarn-error.log*
|
||||
pnpm-debug.log*
|
||||
|
||||
# =========================================================
|
||||
# Go
|
||||
# =========================================================
|
||||
# Go build outputs
|
||||
bin/
|
||||
dist/
|
||||
build/
|
||||
|
||||
# Test artifacts
|
||||
*.test
|
||||
coverage.out
|
||||
coverage.html
|
||||
|
||||
# Go workspace
|
||||
go.work.sum
|
||||
|
||||
# =========================================================
|
||||
# Python
|
||||
# =========================================================
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
*$py.class
|
||||
|
||||
# Virtual environments
|
||||
.venv/
|
||||
venv/
|
||||
env/
|
||||
ENV/
|
||||
|
||||
# Packaging
|
||||
build/
|
||||
dist/
|
||||
*.egg-info/
|
||||
.eggs/
|
||||
pip-wheel-metadata/
|
||||
|
||||
# Testing
|
||||
.pytest_cache/
|
||||
.coverage
|
||||
.coverage.*
|
||||
htmlcov/
|
||||
mlruns*
|
||||
mlartifacts/
|
||||
backtest_output
|
||||
|
||||
# Scheduled algo-trading runtime artifacts (rd_train/rd_predict/rd_backtest output)
|
||||
tac-algo-output/
|
||||
|
||||
# Type checking
|
||||
.mypy_cache/
|
||||
.pyre/
|
||||
.pytype/
|
||||
|
||||
# Linting
|
||||
.ruff_cache/
|
||||
|
||||
# Jupyter
|
||||
.ipynb_checkpoints/
|
||||
|
||||
# =========================================================
|
||||
# Rust
|
||||
# =========================================================
|
||||
target/
|
||||
|
||||
# Keep Cargo.lock for applications.
|
||||
# Uncomment for libraries:
|
||||
# Cargo.lock
|
||||
|
||||
# =========================================================
|
||||
# Node.js / Next.js
|
||||
# =========================================================
|
||||
node_modules/
|
||||
|
||||
.next/
|
||||
out/
|
||||
.vercel/
|
||||
|
||||
# Package manager caches
|
||||
.npm/
|
||||
.pnpm-store/
|
||||
|
||||
.yarn/
|
||||
.yarn/cache/
|
||||
.yarn/unplugged/
|
||||
.yarn/build-state.yml
|
||||
.yarn/install-state.gz
|
||||
|
||||
# Next.js build artifacts
|
||||
next-env.d.ts
|
||||
|
||||
# Turborepo
|
||||
.turbo/
|
||||
|
||||
# =========================================================
|
||||
# Coverage / Reports
|
||||
# =========================================================
|
||||
coverage/
|
||||
coverage-final.json
|
||||
lcov.info
|
||||
|
||||
# =========================================================
|
||||
# Temporary files
|
||||
# =========================================================
|
||||
tmp/
|
||||
temp/
|
||||
.cache/
|
||||
.tmp/
|
||||
|
||||
# =========================================================
|
||||
# Experiment lineage repo (runtime clone of $GIT_REPO_URL, never committed —
|
||||
# the URL differs per environment and .gitmodules has no env-var expansion)
|
||||
# =========================================================
|
||||
tac-exp-dev/
|
||||
experiments/
|
||||
|
||||
# =========================================================
|
||||
# Docker
|
||||
# =========================================================
|
||||
.docker/
|
||||
docker-compose.override.yml
|
||||
|
||||
# =========================================================
|
||||
# Terraform (if used)
|
||||
# =========================================================
|
||||
.terraform/
|
||||
*.tfstate
|
||||
*.tfstate.*
|
||||
|
||||
.env*
|
||||
@@ -0,0 +1,64 @@
|
||||
# TradeAC Quant Trading Guide — Agent Working Agreement
|
||||
|
||||
You are writing a practitioner's guide to quantitative trading in a way that is **real**: every number, result and claim must be traceable to evidence produced on the TradeAC stack (this repo's lake + R&D server + live execution trail) or to a cited external source. This file is the contract for that work.
|
||||
|
||||
## Mission
|
||||
|
||||
A book that a quant-desk reader can act on: signal generation → strategy → sizing → execution → risk → reconciliation, grounded in what TradeAC actually ran and actually traded. Where TradeAC has *not* proved something, the book says so and marks it a hypothesis.
|
||||
|
||||
## Truth rules (non-negotiable)
|
||||
|
||||
1. **Never fabricate.** No invented backtests, metrics, fill prices, papers, or quotes. If we did not run it or cannot cite it, we do not state it.
|
||||
2. **Classify every quantitative claim** with an inline evidence tag:
|
||||
- `PROVEN` — reproduced from a recorded experiment run or a reconciled live round. Cite `experiment_id`/`run_id`/branch or `round_id`.
|
||||
- `HYPOTHESIS` — plausible but untested (or tested once, un-reproduced). Always labeled as such; never stated as fact.
|
||||
- `REFERENCED` — industry/academic practice. Cite the external source (websearch/HITL), never from memory.
|
||||
3. **Backtests are historical, not promises.** Anywhere a backtest metric is quoted, say so and note the universe + date window + whether the hypothesis was pre-registered before the run (TradeAC has 31+ experiments — be explicit about post-hoc cherry-picking risk).
|
||||
4. **Live beats backtest.** A claim about trading performance must trace to the tac-rd-book execution trail (round_id, decisions, fills, reconcile: slippage bps, cost), not just to a backtest.
|
||||
5. **Every quoted number lands in the evidence ledger** (`book/EVIDENCE.md`) with a link to where it was produced.
|
||||
|
||||
## Evidence sources (use in this order of trust)
|
||||
|
||||
1. **Traced experiments** — the `experiments/` git repo, per-experiment branches (`exp/7`…`exp/31`), and MLflow runs via `tac-qlib-rd`: `rd_exp_list`, `rd_exp_get_run`, `rd_exp_result`, `rd_exp_model`, `rd_exp_input`, `rd_exp_get_notes`, `rd_exp_lineage`, `rd_trace_search`. Skill: `tradeac-rd-explain` (how to read runs), `tac-qlib-custom` (how experiments are wired).
|
||||
2. **Live execution trail** — `tac-rd-book`: `round_list`, `round_get`, `round_metrics`, `trail_query`, `book_reconcile`. This is ground truth for execution cost, slippage and whether the funnel (targets → decisions → fills) holds up.
|
||||
3. **Lake + market data** — `tac-engine`: `get_lake_bars/_coverage/_features`, `get_account`, `list_positions`, `list_orders`, `get_portfolio_history`, `get_news`. Skill: `tradeac-lake`, `tradeac-alpaca`.
|
||||
4. **Ad-hoc scripting** — a scripted validation is allowed to *confirm or extend* an experiment, but its inputs, code and outputs must be persisted under `book/data/` and referenced from the ledger. It is evidence, not gospel.
|
||||
5. **External references** — use websearch/HITL for academic foundations, market microstructure facts, regulation, industry practice. Always cite.
|
||||
|
||||
## Workflow for each chapter
|
||||
|
||||
1. Draft the chapter outline and a **claim inventory** — each claim listed with its expected truth status.
|
||||
2. Gather evidence claim-by-claim using the MCP tools + experiments repo (parallelize tool calls; read runs, notes, backtest reports, live rounds).
|
||||
3. Write the chapter; embed evidence tags inline: `(EVIDENCE#012 → exp/19)`.
|
||||
4. **HITL review gates** before finalizing anything a reader could act on: live performance numbers, cost/slippage figures, risk-limit advice, size/position formulas, drawdown guidance.
|
||||
5. Update `EVIDENCE.md` and `CLAIMS.md` after each chapter.
|
||||
6. Commit per chapter with a message that names the chapter and the experiments cited.
|
||||
|
||||
## Repository layout (book project)
|
||||
|
||||
```
|
||||
book/
|
||||
README.md # TOC, per-chapter status (drafting/in-review/done), how to read
|
||||
EVIDENCE.md # ledger: id → claim → source (experiment/run/branch, round_id, script, citation) → verified?
|
||||
CLAIMS.md # the proven-vs-hypothesis matrix, updated every chapter
|
||||
chapters/
|
||||
00-intro.md # why a real execution trail matters (tradeac-rd-book as the spine)
|
||||
... # one file per chapter, ordered per README TOC
|
||||
data/ # ad-hoc validation scripts + their outputs
|
||||
references/ # external citations collected during research
|
||||
```
|
||||
|
||||
## Writing conventions
|
||||
|
||||
- **No false precision**: report IC/Rank IC to sensible decimals, always with universe + date window.
|
||||
- **Separate "what TradeAC observed" from "what practice generally does"** in the text.
|
||||
- Hedge hypotheses; avoid absolutes; include standard disclaimers wherever returns or risk are discussed.
|
||||
- Mark open questions as `TODO(evidence-needed: <what would settle this>)`.
|
||||
- Do not add emojis or filler; keep prose desk-grade and direct.
|
||||
|
||||
## Don'ts
|
||||
|
||||
- Don't invent a backtest we never ran, or quote one run as a universal rule.
|
||||
- Don't quote live P&L without a `round_id` + reconcile behind it.
|
||||
- Don't cite a paper/URL from memory — fetch it or ask the user.
|
||||
- Don't claim a "fix" worked if it only shows up in one experiment; demand reproduction or label it hypothesis.
|
||||
@@ -0,0 +1,9 @@
|
||||
[package]
|
||||
name = "tac-workspace"
|
||||
version = "0.0.0"
|
||||
edition = "2021"
|
||||
publish = false
|
||||
|
||||
[workspace]
|
||||
members = ["tac-engine"]
|
||||
resolver = "2"
|
||||
@@ -0,0 +1,64 @@
|
||||
# CLAIMS.md — Proven vs Hypothesis Matrix
|
||||
|
||||
The running scoreboard of every quantitative claim in the book. Updated per chapter after HITL review. Status codes: `PROVEN` (reproduced from recorded run / reconciled round), `HYPOTHESIS` (plausible, tested once or never), `REFUTED` (tested and contradicted), `REFERENCED` (external citation).
|
||||
|
||||
## Signal & features
|
||||
|
||||
| Claim | Status | Evidence |
|
||||
|-------|--------|----------|
|
||||
| Baseline 1-day LGB signal is weak on 2026 OOS (RankIC 0.040, ICIR 0.062) | PROVEN | EVIDENCE#001 → exp 8 |
|
||||
| Costs erase most of the baseline edge (+6.2% gross → +1.6% net) | PROVEN | EVIDENCE#002 → exp 8 |
|
||||
| Dropping model-specific feature families (ou, hmm) improves rank signal (RankIC 0.030→0.064) | PROVEN | EVIDENCE#003 → exp 9 |
|
||||
| Adding moment/volatility families regresses the signal | PROVEN (refuted direction) | EVIDENCE#004 → exp 11 |
|
||||
| Adding OU mean-reversion (sp_ou_zscore) hurts on clean data | PROVEN (refuted direction) | EVIDENCE#014 → exp 25 |
|
||||
| Multi-horizon momentum (M1) degrades the reference | PROVEN (refuted direction) | EVIDENCE#017 → exp 29 |
|
||||
| GARCH(1,1) vol-regime features add no signal | PROVEN (refuted direction) | EVIDENCE#019 → exp 31 |
|
||||
| Risk-adjusted 22d Sharpe drift (M2) improves portfolio metrics | HYPOTHESIS (one run, unreproduced) | EVIDENCE#018 → exp 30 |
|
||||
| More features ≠ better signal on a small (50-name) cross-section | HYPOTHESIS (3 supporting runs, panel-specific) | EVIDENCE#003/004/014/017/019 |
|
||||
| General stochastic features (no TA/HMM/OU) have highest ICIR 0.340 | PROVEN | EVIDENCE#012 → exp 23 |
|
||||
|
||||
## Model
|
||||
|
||||
| Claim | Status | Evidence |
|
||||
|-------|--------|----------|
|
||||
| 5-seed RankIC ensemble raises performance vs single model on ablated set | PROVEN (pre-reset); re-validated post-reset exp 22–24 | EVIDENCE#005/011/013 |
|
||||
| Seed count is load-bearing: 2 seeds < 5 seeds on clean data | PROVEN | EVIDENCE#016 → exp 28 |
|
||||
| n_drop 2→1 flips net excess (−3.21% → +2.13%) with identical signal metrics | PROVEN | EVIDENCE#015 → exp 26 |
|
||||
| Cost drag is the binding constraint, not signal quality | PROVEN | EVIDENCE#015 → exp 26 (IC/RankIC identical across n_drop) |
|
||||
| Fractional-Kelly sizing beats equal-weight top-k net of costs | HYPOTHESIS (exp 15 never finished) | run never completed |
|
||||
|
||||
## Portfolio construction & risk
|
||||
|
||||
| Claim | Status | Evidence |
|
||||
|-------|--------|----------|
|
||||
| TopkDropout beats stochastic-control OptimalStopControl on the ensemble signal | PROVEN | EVIDENCE#006/007 → exp 13/14 |
|
||||
| Stop-control churns and bleeds costs (−11.3pp cost drag) | PROVEN | EVIDENCE#006 → exp 13 |
|
||||
| $5M liquidity floor improves net IR (0.81→0.98) and cuts drawdown (7.9%→5.4%) | PROVEN (pre-clean-lake; not comparable post-reset) | EVIDENCE#008 → exp 18 |
|
||||
| Size/concentration caps hurt by cutting deployed capital | PROVEN (pre-clean-lake) | EVIDENCE#008 → exp 18 |
|
||||
| Entry/risk gates (momentum, HMM) are byte-identical no-ops on the reference signal | PROVEN | EVIDENCE#009 → exp 20 |
|
||||
| Signal quality is the bottleneck, not the execution/risk layer | PROVEN (on the exp-20 reference) | EVIDENCE#009 → exp 20 |
|
||||
|
||||
## Data & reproducibility
|
||||
|
||||
| Claim | Status | Evidence |
|
||||
|-------|--------|----------|
|
||||
| The reference signal did not reproduce on a rebuilt lake (IC 0.035→0.002) | PROVEN | EVIDENCE#010 → exp 21 |
|
||||
| Old-lake data quality inflated the signal and backtest | PROVEN | EVIDENCE#010 → exp 21 |
|
||||
| Signal work must be re-validated after any data rebuild | PROVEN (exp 21) / HYPOTHESIS (generality) | EVIDENCE#010 |
|
||||
| Pre-reset experiment baselines are not comparable to post-reset runs | PROVEN | EVIDENCE#009/010 (exp 20 R0 note, exp 21) |
|
||||
|
||||
## Live execution
|
||||
|
||||
| Claim | Status | Evidence |
|
||||
|-------|--------|----------|
|
||||
| Live funnel held: 10 targets → 10 decided → 10 placed → 9 filled | PROVEN | EVIDENCE#020 → round 3 |
|
||||
| Realized slippage ≈ 4.54 bps, est. cost ≈ $45, turnover 0.74 | PROVEN | EVIDENCE#020 → round 3 metrics |
|
||||
| Execution claims trace to round_id + reconcile, not backtest | PROVEN (methodology, round 3 settled) | EVIDENCE#020 |
|
||||
| 50-ETF panel results generalize to other universes | HYPOTHESIS — TODO(evidence-needed) | — |
|
||||
|
||||
## Open questions (settled by further experiments)
|
||||
|
||||
- exp 30 M2 Sharpe-drift: reproduce on a second window before promoting past HYPOTHESIS.
|
||||
- exp 15 Kelly sizing: re-run on the clean lake.
|
||||
- exp 18 risk-limit spec: re-validate $5M liquidity floor on the post-reset reference signal (exp 26 lineage).
|
||||
- Out-of-universe validation: non-ETF universe for the compact stochastic feature set.
|
||||
@@ -0,0 +1,55 @@
|
||||
# Evidence Ledger
|
||||
|
||||
Every quantitative claim in the book lands here: id → claim → source (experiment/run/branch, round_id, script, citation) → verified?.
|
||||
|
||||
## Key metric-schema note
|
||||
|
||||
Experiments 8–18 record metrics under a legacy schema (`ls_sharpe`, `maxdd_with_cost`, `excess_ann_with_cost`, `excess_ir_with_cost`, `ls_ann_return`). Experiments 21+ use the canonical `IC / ICIR / Rank IC / Rank ICIR / net_IR / net_ann_return / gross_* / Long-Short_Ann_Sharpe / net_max_drawdown`. Do not compare schemas directly; chapter text states which schema a number comes from. Additionally, exp 20's R0 note states the exp-18 baseline is not comparable to post-reset runs due to environment non-determinism, and exp 21 invalidated all pre-clean-lake positive results.
|
||||
|
||||
## Pre-clean-lake period (exp 8–18) — historical, superseded
|
||||
|
||||
| ID | Claim | Source | Verified? |
|
||||
|----|-------|--------|-----------|
|
||||
| EVIDENCE#001 | Baseline 1-day LGB signal weak on 2026 OOS: IC 0.017, ICIR 0.062, RankIC 0.040, RankICIR 0.161 (below 0.2 noise threshold). L/S ann +4.9%. | exp 8, run `e65cf1ec…` (mlflow exp 10), branch `exp/8-baseline-lightgbm-on-the-full-60etf-univ` | yes |
|
||||
| EVIDENCE#002 | Costs erase most of the raw edge on baseline: excess +6.2% ann w/o cost (IR 0.31, MaxDD −20.4%) vs +1.6% ann after costs (IR 0.08). | exp 8 (same run) | yes |
|
||||
| EVIDENCE#003 | Feature-family ablation: generic-only (jump,har,trend,hurst,signature,ret,max_move) beats all-24: RankIC 0.030→0.064, RankICIR 0.146→0.276, L/S Sharpe −0.83→+2.55, net excess −9.4%→+3.1%. | exp 9, run `7b1e7972…` (mlflow exp 11), branch `exp/9-sp5d-feature-family-ablation` | yes |
|
||||
| EVIDENCE#004 | Adding 16 moment/volatility fields regresses every metric (RankIC 0.064→0.047, net excess −16.2% IR −1.57) — same failure mode as ou/hmm. | exp 11, run `a3f7d1d4…` (mlflow exp 12), branch `exp/11-sp5d-momentfeature-extension-after-exten` | yes |
|
||||
| EVIDENCE#005 | 5-seed RankIC ensemble on ablated generic features: RankIC 0.0586, RankICIR 0.224, net excess +7.8% (IR 0.79), L/S Sharpe 3.71, MDD −7.9%. Best pre-clean-lake net result. | exp 12, run `0cea66d9…` (mlflow exp 16), branch `exp/12-isolate-the-multiseed-rankic-ensemble-ef` | yes (superseded by EVIDENCE#011 on clean data) |
|
||||
| EVIDENCE#006 | OptimalStopControl (entry 0.85/exit 0.7/hold 10/sl −0.08) worse than TopkDropout: net excess −2.7% (IR −0.31) vs +7.8%; cost drag −11.3pp. | exp 13, run `4e1f77b4…` (mlflow exp 17), branch `exp/13-portfolioconstruction-variant-of-the-iso` | yes |
|
||||
| EVIDENCE#007 | OptimalStopControlV2 (turnover band/cooldown/cap) also refuted: net −6.9% (IR −0.72) vs TopkDropout +7.8% (IR 0.79). | exp 14, run `83d7e27e…` (mlflow exp 18), branch `exp/14-enhanced-stochasticcontrol-allocation-fo` | yes |
|
||||
| EVIDENCE#008 | Risk-limit A/B: $5M liquidity floor → net IR 0.81→0.98, cumDD 7.93%→5.44%; size cap 15% + conc 60% hurts (IR 0.816, ann 6.11%). | exp 18, run `28c7fa08…` (mlflow exp 21), branch `exp/18-risk-limit-control-on-the-reference-ense` | yes (pre-clean-lake, see note) |
|
||||
| EVIDENCE#009 | Improvement sweep (R1-R5): 4/5 refuted; R2 momentum gate and R3 HMM gate are byte-identical no-ops; R5 MA3/EWMA marginal (IR 0.049). Conclusion: signal quality is the bottleneck, not the execution/risk layer. | exp 20, run `958198a8…` (mlflow exp 21), branch `exp/20-improve-the-risk-limit-reference-signal` | yes |
|
||||
|
||||
## Post-reset period (exp 21–31) — canonical, current
|
||||
|
||||
| ID | Claim | Source | Verified? |
|
||||
|----|-------|--------|-----------|
|
||||
| EVIDENCE#010 | Clean-lake re-execution of the reference collapsed: IC 0.0019 (vs ref 0.0354), RankIC 0.0259, net −20.6% (IR −2.70). Old lake data quality had inflated the signal. | exp 21, run `f1bd3c28…` (mlflow exp 23), branch `exp/21-clean-lake-re-execution-of-the-tac-rd-ra` | yes |
|
||||
| EVIDENCE#011 | Re-run after fixing feature routing: IC 0.0486, RankIC 0.0617, ICIR 0.235, RankICIR 0.243, L/S Sharpe 3.23. | exp 22, run `18db5bc1…` (mlflow exp 24), branch `exp/22-re-run-experiment-16s-5-day-rankic-ensem` | yes |
|
||||
| EVIDENCE#012 | General stochastic features only (no TA/HMM/OU): IC 0.0728, ICIR 0.340, L/S Sharpe 4.56. | exp 23, run `be5cd314…` (mlflow exp 25), branch `exp/23-test-whether-the-5-day-rankic-ensemble-i` | yes |
|
||||
| EVIDENCE#013 | Compact stochastic set (raw OHLCV + sp_ret, jump, RV1/5/22, vol ratios, trend slopes, logp, hurst, signature L1/L2): IC 0.0511, RankIC 0.0663, RankICIR 0.2545, L/S Sharpe 4.54. | exp 24, run `fe469a19…` (mlflow exp 25), branch `exp/24-run-the-rankic-ensemble-in-mlflow-experi` | yes |
|
||||
| EVIDENCE#014 | Adding sp_ou_zscore hurts on clean data: IC 0.0343 vs 0.0511, net −3.76% vs −3.21%. | exp 25, run `57450d1a…` (mlflow exp 25), branch `exp/25-test-the-clean-data-hypothesis-that-addi` | yes |
|
||||
| EVIDENCE#015 | n_drop 2→1 on identical compact stochastic signal: gross +7.02%, net +2.13% (vs −3.21%), MDD −7.69%, IR 0.21. IC/RankIC identical to n_drop 2 — the gain is turnover/cost relief. | exp 26, run `21afc6af…` (mlflow exp 25), branch `exp/26-test-whether-reducing-topkdropout-daily` | yes — best result of the campaign |
|
||||
| EVIDENCE#016 | 2-seed ensemble loses to 5-seed on clean data: RankIC 0.0579 vs 0.0663, net −1.49% (IR −0.14) vs +2.13% (IR 0.21). Seed count is load-bearing. | exp 28, run `c4ab1d01…` (mlflow exp 27), branch `exp/28-isolate-the-seed-count-effect-on-the-ndr` | yes |
|
||||
| EVIDENCE#017 | Multi-horizon momentum bundle refuted: IC 0.0337 vs 0.0511, net −13.35% (IR −1.12) vs +2.13%. | exp 29, run `b4586675…` (mlflow exp 28), branch `exp/29-isolation-run-m1-does-adding-multi-horiz` | yes |
|
||||
| EVIDENCE#018 | Risk-adjusted 22d Sharpe drift: mixed — rank metrics lower (RankIC 0.0576 vs 0.0663) but portfolio strong (net +6.53% IR 0.62 vs +2.13% IR 0.21). Single run, unreproduced. | exp 30, run `d5d775f9…` (mlflow exp 29), branch `exp/30-isolation-run-m2-does-adding-risk-adjust` | yes — mark HYPOTHESIS in text |
|
||||
| EVIDENCE#019 | GARCH(1,1) vol-regime trio refuted: IC 0.0415 vs 0.0511, RankICIR 0.179 vs 0.255, net +1.36% (IR 0.13). | exp 31, run `514cb523…` (mlflow exp 30), branch `exp/31-isolation-run-m3-does-adding-garch11-vol` | yes |
|
||||
|
||||
## Live execution trail
|
||||
|
||||
| ID | Claim | Source | Verified? |
|
||||
|----|-------|--------|-----------|
|
||||
| EVIDENCE#020 | Live round 3 (target 2026-08-17): retrained exp-26 n_drop=1 config on rolling 4y window; Topk10/n_drop1 with risk limits (liq floor $5M dropped 8, size cap 12%, conc 95%, drawdown pause 10%); funnel 10 targets → 10 decided → 10 placed → 9 filled, 1 cancelled, 1 skipped (SLV delta_zero); invested $74,202.85, slippage 4.54 bps, est. cost ~$45. | round 3 (`tac-rd-book`), trace 27, run `721ef257…` (mlflow exp 26), branch `exp/27-scheduled-algo-retrain-on-2026-08-17-tac` | yes — settled, reconcile available |
|
||||
| EVIDENCE#021 | Scheduled retrain on 2026-08-14 (pre-reset reference): 10 buys + 6 sells placed, 0 cancelled by sentiment gate; sized on live equity $99,999.93. | trace 16, run `3b858b2b…` (mlflow exp 13), branch `exp/16-scheduled-algo-retrain-on-20260814-tacrd` | yes — historical, pre-reset signal |
|
||||
|
||||
## Ad-hoc scripts (book/data/)
|
||||
|
||||
| ID | Claim | Source | Verified? |
|
||||
|----|-------|--------|-----------|
|
||||
| (none yet) | — | — | — |
|
||||
|
||||
## External references (book/references/)
|
||||
|
||||
| ID | Claim | Source | Verified? |
|
||||
|----|-------|--------|-----------|
|
||||
| (none yet) | — | — | — |
|
||||
+142
@@ -0,0 +1,142 @@
|
||||
# TradeAC Quant Trading Guide — Table of Contents & Status
|
||||
|
||||
A practitioner's guide to quantitative trading written the only way it is worth reading: grounded in a real research loop and a real execution trail. Every number in this book was either reproduced from a recorded TradeAC experiment (MLflow run + traced git branch) or a reconciled live round, or it is explicitly labeled a hypothesis. See `AGENTS.md` (repo root) for the truth contract; `EVIDENCE.md` for the ledger; `CLAIMS.md` for the proven-vs-hypothesis matrix.
|
||||
|
||||
## What this book is for
|
||||
|
||||
A quant-desk reader should be able to act on this book: replicate a signal pipeline, size a book, gate it with risk limits, execute it, and reconcile what actually happened. The book's spine is **how performance improved with research-proved truth** — the actual arc of TradeAC's campaign from a baseline that barely cleared costs to a live, reconciled round.
|
||||
|
||||
## How to read evidence tags
|
||||
|
||||
- `PROVEN` — reproduced from a recorded run or reconciled round. Citation is an `experiment_id`/`run_id` or `round_id`.
|
||||
- `HYPOTHESIS` — plausible but not yet reproduced; never stated as fact.
|
||||
- `REFERENCED` — industry/academic practice; citation is an external source.
|
||||
- `TODO(evidence-needed: …)` — an open question the desk should settle.
|
||||
|
||||
## Table of contents
|
||||
|
||||
| # | Chapter | Status | Core experiments cited | Core lesson |
|
||||
|---|---------|--------|------------------------|-------------|
|
||||
| 00 | Why a real execution trail matters | drafting | round 3 | A book claims nothing it cannot reconcile |
|
||||
| 01 | The research loop: lake → experiment → live | drafting | exp 8–31 | Traceability is the methodology |
|
||||
| 02 | Baseline and the cost reality | drafting | exp 8 | A signal that dies after 5bp/15bp is not a signal |
|
||||
| 03 | Prune, don't add: feature-family ablation | drafting | exp 9, 10, 11, 25 | On a 50-name panel, generic beats model-specific |
|
||||
| 04 | Ensembles and the seed-count effect | drafting | exp 12, 28 | Averaging raises ICIR; seed count is load-bearing |
|
||||
| 05 | The clean-lake reset: data quality as first-order risk | drafting | exp 21–24 | If it doesn't reproduce on clean data, it was noise |
|
||||
| 06 | Isolation runs: single-variable discipline | drafting | exp 26, 29–31 | Most additions fail; the discipline is the value |
|
||||
| 07 | Portfolio construction: dropout vs optimal stop | drafting | exp 13, 14, 15 | Turnover-sensitive construction bleeds the edge |
|
||||
| 08 | The cost/turnover frontier | drafting | exp 26 | n_drop 2→1: hold the dropped name, keep the edge |
|
||||
| 09 | Risk limits that work | drafting | exp 18, 20 | Liquidity floor > concentration caps; gates are no-ops when signal is the bottleneck |
|
||||
| 10 | Live execution and reconciliation | drafting | exp 27, round 3 | 4.54 bps slippage realized; funnel 10→10→10→9 |
|
||||
| 11 | Synthesis: how proved truth compounds | drafting | all | The scoreboard of what moved performance and why |
|
||||
|
||||
Status legend: `drafting` → `in-review` → `done`.
|
||||
|
||||
## Per-chapter claim inventory (expected truth status)
|
||||
|
||||
Each chapter opens with its claims. The inventory below is the working contract: what the chapter asserts, and what evidence tier it must land in. It is updated as chapters pass their HITL review gate.
|
||||
|
||||
### 00 — Why a real execution trail matters
|
||||
| Claim | Expected status |
|
||||
|-------|-----------------|
|
||||
| A book's claims must be reconcilable to a real trail (targets→decisions→fills) | `PROVEN` — round 3 funnel |
|
||||
| Backtest claims without live reconciliation are hypotheses about execution | `HYPOTHESIS` → settled by round 3 |
|
||||
| The funnel (targets→decided→placed→filled) is the minimal honesty structure | `REFERENCED` (industry ops practice) + `PROVEN` via tac-rd-book schema |
|
||||
|
||||
### 01 — The research loop
|
||||
| Claim | Expected status |
|
||||
|-------|-----------------|
|
||||
| Experiments must be traced: branch + MLflow run + notes (hypothesis before run) | `PROVEN` — traceability loop used on exp 8–31 |
|
||||
| Pre-registration protects against post-hoc cherry-picking | `REFERENCED` (research practice; see CLAIMS for multiple-testing note) |
|
||||
| The lake is the single source of bar/feature truth | `PROVEN` — exp 21 showed dirty-lake risk |
|
||||
|
||||
### 02 — Baseline and the cost reality
|
||||
| Claim | Expected status |
|
||||
|-------|-----------------|
|
||||
| Baseline 1-day LGB signal is weak on 2026 OOS (RankIC ≈ 0.04, below the 0.2 ICIR noise threshold) | `PROVEN` — exp 8 |
|
||||
| Costs erase most of the raw edge: +6.2% ann gross → +1.6% net | `PROVEN` — exp 8 |
|
||||
| A viable signal must clear realistic execution costs | `PROVEN` (exp 8, exp 26) + `REFERENCED` |
|
||||
|
||||
### 03 — Prune, don't add
|
||||
| Claim | Expected status |
|
||||
|-------|-----------------|
|
||||
| Dropping model-specific feature families (ou, hmm) improves the rank signal (RankIC 0.030→0.064) | `PROVEN` — exp 9 |
|
||||
| Adding moment/volatility families regresses the signal (exp 11), same failure mode as ou/hmm | `PROVEN` — exp 11 |
|
||||
| Adding OU mean-reversion (sp_ou_zscore) hurts on clean data | `PROVEN` — exp 25 |
|
||||
| More features ≠ better signal on a small cross-section | `HYPOTHESIS` (supported by 3 runs, still panel-specific) |
|
||||
|
||||
### 04 — Ensembles
|
||||
| Claim | Expected status |
|
||||
|-------|-----------------|
|
||||
| 5-seed RankIC ensemble raises net-of-cost performance vs single model on the ablated set | `PROVEN` — exp 12 (pre-clean-lake), re-validated exp 22–24 |
|
||||
| Seed count is load-bearing: 2 seeds lose to 5 seeds on clean data | `PROVEN` — exp 28 |
|
||||
| Ensemble averaging's benefit is separable from feature expansion | `PROVEN` — exp 12 isolation design |
|
||||
|
||||
### 05 — Clean-lake reset
|
||||
| Claim | Expected status |
|
||||
|-------|-----------------|
|
||||
| The reference signal did not reproduce on a rebuilt lake (IC 0.035→0.002) | `PROVEN` — exp 21 |
|
||||
| Data-quality problems had inflated earlier results; post-reset signal is the only valid one | `PROVEN` — exp 21 + exp 22–24 reproduction |
|
||||
| Signal work must be re-validated after any data rebuild | `PROVEN` (exp 21) + `HYPOTHESIS` for generality |
|
||||
|
||||
### 06 — Isolation runs
|
||||
| Claim | Expected status |
|
||||
|-------|-----------------|
|
||||
| Single-variable changes isolate what moved performance | `PROVEN` — exp 26→29/30/31 design |
|
||||
| Multi-horizon momentum degrades the reference (net IR 0.21→-1.12) | `PROVEN` — exp 29 |
|
||||
| Risk-adjusted 22d Sharpe drift is promising on portfolio metrics, mixed on rank | `HYPOTHESIS` — exp 30 single run, unreproduced |
|
||||
| GARCH(1,1) vol-regime features add no signal | `PROVEN` — exp 31 |
|
||||
|
||||
### 07 — Portfolio construction
|
||||
| Claim | Expected status |
|
||||
|-------|-----------------|
|
||||
| TopkDropout beats stochastic-control OptimalStopControl on the ensemble signal | `PROVEN` — exp 13, 14 |
|
||||
| Stop-control constructions churn and bleed costs (cost drag ≈ −11.3pp) | `PROVEN` — exp 13 |
|
||||
| Fractional-Kelly sizing (exp 15) is unverified | `HYPOTHESIS` — run never finished |
|
||||
|
||||
### 08 — Cost/turnover frontier
|
||||
| Claim | Expected status |
|
||||
|-------|-----------------|
|
||||
| n_drop 2→1 flips net excess from −3.21% to +2.13% with identical signal metrics | `PROVEN` — exp 26 |
|
||||
| Cost drag is the binding constraint, not signal quality | `PROVEN` — exp 26 (IC/RankIC identical between n_drop variants) |
|
||||
|
||||
### 09 — Risk limits
|
||||
| Claim | Expected status |
|
||||
|-------|-----------------|
|
||||
| $5M liquidity floor improves net IR 0.81→0.98 and cuts drawdown 7.9%→5.4% | `PROVEN` — exp 18 (pre-clean-lake; see note in chapter) |
|
||||
| Size/concentration caps hurt by cutting deployed capital | `PROVEN` — exp 18 |
|
||||
| Entry/risk gates are no-ops when the signal is the bottleneck | `PROVEN` — exp 20 (R2/R3 byte-identical) |
|
||||
| Exp-18 numbers are not comparable to post-reset runs due to env non-determinism | `PROVEN` — exp 20 R0 note |
|
||||
|
||||
### 10 — Live execution and reconciliation
|
||||
| Claim | Expected status |
|
||||
|-------|-----------------|
|
||||
| Live funnel held: 10 targets → 10 decided → 10 placed → 9 filled, 1 cancelled, 1 skipped | `PROVEN` — round 3 |
|
||||
| Realized slippage ≈ 4.54 bps, estimated cost ≈ $45, turnover 0.74 | `PROVEN` — round 3 metrics |
|
||||
| Live beats backtest: execution claims trace to round_id, not to backtest | `PROVEN` — methodology |
|
||||
|
||||
### 11 — Synthesis
|
||||
| Claim | Expected status |
|
||||
|-------|-----------------|
|
||||
| The largest performance deltas came from data quality, cost/turnover relief, feature pruning, and risk limits — not from adding features | `PROVEN` — composite of exp 9, 18, 21, 26 |
|
||||
| The campaign's refuted runs (exp 11, 13, 14, 20, 25, 29, 31) were as valuable as wins | `REFERENCED` + `PROVEN` (they stopped wrong directions) |
|
||||
| Generalizability of the 50-ETF panel results is an open question | `HYPOTHESIS` — TODO(evidence-needed: out-of-panel universe) |
|
||||
|
||||
## Repository layout
|
||||
|
||||
```
|
||||
book/
|
||||
README.md # this file
|
||||
EVIDENCE.md # ledger: id → claim → source → verified?
|
||||
CLAIMS.md # proven-vs-hypothesis matrix, updated every chapter
|
||||
chapters/00-intro.md ... # one file per chapter
|
||||
data/ # ad-hoc validation scripts + outputs
|
||||
references/ # external citations
|
||||
```
|
||||
|
||||
## Open questions for the desk
|
||||
|
||||
- `TODO(evidence-needed: reproduction of exp 30 M2 Sharpe-drift run on a second window)`
|
||||
- `TODO(evidence-needed: exp 15 Kelly sizing — run never finished; re-run on the clean lake)`
|
||||
- `TODO(evidence-needed: out-of-universe (non-ETF) validation of the compact stochastic feature set)`
|
||||
- `TODO(evidence-needed: reconciliation of exp 18 risk-limit spec on the post-reset reference signal)`
|
||||
@@ -0,0 +1,56 @@
|
||||
# Chapter 00 — Why a Real Execution Trail Matters
|
||||
|
||||
Status: drafting. Claim inventory: see `README.md` ch. 00.
|
||||
|
||||
Most quant books are written backwards: the author knows the answer, then builds a narrative to fit it. Backtests are quoted as if they were the outcome, the fill price is assumed to be the signal price, and cost is a footnote. This book is written the other way: every claim that could survive contact with a trading desk must survive contact with a *trail* — a record of what was intended, what was decided, what was placed, and what actually filled, at what price.
|
||||
|
||||
This chapter sets the spine: the `tac-rd-book` execution trail, which records every live round end-to-end.
|
||||
|
||||
## The funnel is the minimal honesty structure
|
||||
|
||||
A live trading round on the TradeAC stack is a chain of five checkpoints:
|
||||
|
||||
```
|
||||
targets (intent) → decided → placed (order) → filled → reconciled
|
||||
```
|
||||
|
||||
The trail records each step as first-class evidence. A round is only settled when the funnel has been reconciled — targets versus decisions versus fills, with per-symbol residuals and roll-ups for cash/buying-power impact, slippage in basis points, and cost as a fraction of gross traded notional. `PROVEN` — this is the schema of `tac-rd-book` (`trail_query`, `book_reconcile`), the same tooling used for the live rounds this book cites.
|
||||
|
||||
Why this structure and not a spreadsheet of P&L? Because P&L is the *last* place problems show up. By the time net return is wrong, you no longer know whether the intent was wrong (bad signal), the decision was wrong (bad gating), the fill was wrong (bad execution), or the book was wrong (bad risk). A funnel isolates the four.
|
||||
|
||||
## What the trail proved that a backtest could not
|
||||
|
||||
The book's live ground truth is round 3 (target date 2026-08-17), the first fully reconciled round of the post-reset signal: `EVIDENCE#020 → round 3`.
|
||||
|
||||
- The funnel held under live conditions: **10 targets → 10 decided → 10 placed → 9 filled**. One order was cancelled, and one target (SLV) was skipped because its requested delta was zero.
|
||||
- Realized slippage was **4.54 bps**; estimated cost **≈ $45** on $74,202.85 invested; turnover **0.74**.
|
||||
- The intent included risk limits from the research campaign: a $5M liquidity floor that dropped 8 of the 50 names, a 12% size cap, 95% concentration cap, and a 10% drawdown pause. `EVIDENCE#020`.
|
||||
|
||||
None of these numbers — slippage in bps, cost as a fraction of gross, the ratio of filled to placed — exists in a backtest. A backtest assumes a cost model (on this stack, 5 bp open / 15 bp close / $5 minimum) and a fill at the close price. The trail records what the market actually charged. That is the difference between a research claim and a trading claim.
|
||||
|
||||
`TODO(evidence-needed: the round-3 reconcile's realized-cost-vs-model comparison once the position window closes)`
|
||||
|
||||
## Backtests are historical, not promises
|
||||
|
||||
Throughout this book, backtest metrics carry a warning label, not a hiding place: universe, date window, and whether the hypothesis was pre-registered before the run. This matters because TradeAC ran 31+ experiments; with that many draws, some positive results will be luck. The book is explicit about which runs were pre-registered (e.g. isolation runs exp 28–31) and which were exploratory. `REFERENCED` — multiple-testing/cherry-picking risk is standard research practice; see `references/` as it accrues.
|
||||
|
||||
The most important proof of this discipline is the clean-lake reset, which this book treats as a turning point rather than a footnote: the pre-reset reference signal did **not** reproduce on a rebuilt lake (`EVIDENCE#010 → exp 21`). Had the book quoted the pre-reset backtest as fact, it would have shipped a lie. The trail and the traceability loop are what allowed the desk to catch it. Chapter 05 tells that story in full.
|
||||
|
||||
## How to read this book
|
||||
|
||||
- Every claim is tagged `PROVEN` (traced experiment/round), `HYPOTHESIS` (unreproduced), or `REFERENCED` (external source). `EVIDENCE.md` maps each tag to the run, branch, and round behind it.
|
||||
- Chapters 02–09 follow the research arc: what was tested, what was proved, what was refuted, and what moved performance. Refuted runs are cited as evidence too — knowing what *doesn't* work is how the desk avoided paying for it twice.
|
||||
- Chapter 10 is the reality check: live execution against the research claims.
|
||||
- Chapter 11 is the synthesis: the scoreboard of what actually improved performance and why.
|
||||
|
||||
## Open questions
|
||||
|
||||
- `TODO(evidence-needed: a second live round beyond round 3, to confirm slippage and funnel hold under a different market regime)`
|
||||
- `TODO(evidence-needed: reconcile realized cost against the 5bp/15bp/$5 backtest model over a full position window)`
|
||||
|
||||
## Evidence cited in this chapter
|
||||
|
||||
| Tag | Source |
|
||||
|-----|--------|
|
||||
| `EVIDENCE#020` | round 3, `tac-rd-book`, trace 27 (branch `exp/27-scheduled-algo-retrain-on-2026-08-17-tac`) |
|
||||
| `EVIDENCE#010` | exp 21, run `f1bd3c28…`, branch `exp/21-clean-lake-re-execution-of-the-tac-rd-ra` |
|
||||
Executable
+110
@@ -0,0 +1,110 @@
|
||||
#!/bin/sh
|
||||
set -e
|
||||
|
||||
# ---- OpenCode agent server (background, best-effort) ----
|
||||
# Start `opencode serve` inside the same container so the deployed tac-app can
|
||||
# reach it on :4096, in the SAME working directory (/app) — sharing
|
||||
# opencode.json, the skill library and .opencode/. The browser talks to it via
|
||||
# the app's OPENCODE_BASE_URL; --cors must allow the app's own origin.
|
||||
#
|
||||
# opencode MUST NOT gate app startup. It used to: the entrypoint blocked on an
|
||||
# unbounded probe, so when opencode's HTTP layer accepted the TCP connection
|
||||
# but never answered (slow MCP cold-start) the curl hung forever, the container
|
||||
# never listened on :3000, Coolify's healthcheck failed, and the site stayed
|
||||
# down until `next start` was started manually in the terminal.
|
||||
OPENCODE_PORT="${OPENCODE_PORT:-4096}"
|
||||
OPENCODE_HOSTNAME="${OPENCODE_HOSTNAME:-0.0.0.0}"
|
||||
|
||||
cors_origins=""
|
||||
cors_args=""
|
||||
add_cors() {
|
||||
for existing in $cors_origins; do
|
||||
[ "$existing" = "$1" ] && return
|
||||
done
|
||||
cors_origins="$cors_origins $1"
|
||||
cors_args="$cors_args --cors $1"
|
||||
}
|
||||
if [ -n "${OPENCODE_CORS:-}" ]; then
|
||||
for origin in $(echo "$OPENCODE_CORS" | tr ',' ' '); do
|
||||
[ -n "$origin" ] && add_cors "$origin"
|
||||
done
|
||||
else
|
||||
for origin in "http://localhost:3000" "https://localhost:3000" \
|
||||
"${BETTER_AUTH_URL:-}" "${APP_URL:-}"; do
|
||||
[ -n "$origin" ] && add_cors "$origin"
|
||||
done
|
||||
fi
|
||||
|
||||
echo "> Starting opencode serve on :$OPENCODE_PORT (auto-restart; log: /tmp/opencode-serve.log) ..."
|
||||
(
|
||||
while :; do
|
||||
# shellcheck disable=SC2086 # intentional word splitting for --cors flags
|
||||
opencode serve --hostname "$OPENCODE_HOSTNAME" --port "$OPENCODE_PORT" $cors_args \
|
||||
|| echo "> opencode serve exited ($?) — restarting in 2s ..."
|
||||
sleep 2
|
||||
done
|
||||
) >/tmp/opencode-serve.log 2>&1 &
|
||||
OPENCODE_PID=$!
|
||||
|
||||
# Best-effort readiness probe: bounded (10s) and every curl capped with
|
||||
# --max-time, so a half-open listen can never stall the container again. If
|
||||
# opencode is slow or down, the app still starts — agent features just degrade.
|
||||
i=0
|
||||
while [ "$i" -lt 10 ]; do
|
||||
if curl -sS --max-time 2 -o /dev/null "http://127.0.0.1:$OPENCODE_PORT/"; then
|
||||
echo "> opencode serve ready on :$OPENCODE_PORT (pid $OPENCODE_PID)"
|
||||
break
|
||||
fi
|
||||
i=$((i + 1))
|
||||
sleep 1
|
||||
done
|
||||
if [ "$i" -ge 10 ]; then
|
||||
echo "> WARNING: opencode serve not ready after 10s — continuing anyway (tail -f /tmp/opencode-serve.log)"
|
||||
fi
|
||||
|
||||
# ---- Experiments submodule (git lineage) ----
|
||||
# The `experiments` submodule lives in the ephemeral container layer — it is
|
||||
# re-created at runtime by `trace.sh init` and is wiped on every redeploy. Ensure
|
||||
# it exists on each boot so /rd/graph and trace.sh work right after a deploy.
|
||||
# Idempotent (validates/creates against $GIT_REPO_URL) and best-effort: never
|
||||
# gate app startup.
|
||||
if [ -n "${GIT_REPO_URL:-}" ] && [ -n "${GIT_USER:-}" ]; then
|
||||
if [ -f /app/tac-qlib/skills/tac-qlib-custom/lib/git_exp.sh ]; then
|
||||
(
|
||||
cd /app
|
||||
GIT_REPO_URL="$GIT_REPO_URL" GIT_USER="$GIT_USER" GIT_PASS="${GIT_PASS:-}" \
|
||||
bash tac-qlib/skills/tac-qlib-custom/lib/git_exp.sh ensure_repo >/dev/null 2>&1 \
|
||||
&& GIT_REPO_URL="$GIT_REPO_URL" GIT_USER="$GIT_USER" GIT_PASS="${GIT_PASS:-}" \
|
||||
bash tac-qlib/skills/tac-qlib-custom/lib/git_exp.sh ensure_base main >/dev/null 2>&1
|
||||
) || echo "> WARNING: could not ensure experiments submodule — run trace.sh init in the container"
|
||||
fi
|
||||
fi
|
||||
|
||||
# Default: serve HTTP with `next start` (production mode).
|
||||
if [ "${SERVER_TLS:-false}" != "true" ]; then
|
||||
exec node node_modules/next/dist/bin/next start tac-app
|
||||
fi
|
||||
|
||||
# ---- HTTPS mode (self-signed certificate) ----
|
||||
# Set SERVER_TLS=true to serve the app over HTTPS. A self-signed cert is
|
||||
# generated on first start and kept under TLS_DIR; override TLS_KEY / TLS_CERT
|
||||
# to mount your own certificates.
|
||||
TLS_HOST="${TLS_HOST:-localhost}"
|
||||
TLS_DIR="${TLS_DIR:-/tmp/tls}"
|
||||
TLS_KEY="${TLS_KEY:-$TLS_DIR/key.pem}"
|
||||
TLS_CERT="${TLS_CERT:-$TLS_DIR/cert.pem}"
|
||||
|
||||
if [ ! -s "$TLS_KEY" ] || [ ! -s "$TLS_CERT" ]; then
|
||||
echo "> Generating self-signed certificate for $TLS_HOST ..."
|
||||
echo "> (Browsers will warn ERR_CERT_AUTHORITY_INVALID. For a trusted cert, generate one with"
|
||||
echo "> mkcert on the host and mount it via TLS_KEY/TLS_CERT.)"
|
||||
mkdir -p "$TLS_DIR"
|
||||
openssl req -x509 -newkey rsa:2048 -nodes \
|
||||
-keyout "$TLS_KEY" -out "$TLS_CERT" -days 825 \
|
||||
-subj "/CN=$TLS_HOST" \
|
||||
-addext "subjectAltName=DNS:localhost,DNS:$TLS_HOST,IP:127.0.0.1" \
|
||||
>/dev/null 2>&1
|
||||
fi
|
||||
|
||||
export TLS_KEY TLS_CERT TLS_HOST
|
||||
exec node /app/tls-server.cjs
|
||||
@@ -0,0 +1,34 @@
|
||||
{
|
||||
"$schema": "https://opencode.ai/config.json",
|
||||
"permission": {},
|
||||
"skills": {
|
||||
"paths": ["tac-engine/skills", "tac-qlib/skills"]
|
||||
},
|
||||
"mcp": {
|
||||
"tac-engine": {
|
||||
"type": "local",
|
||||
"command": ["./tac-engine/target/release/tac-engine"],
|
||||
"enabled": true
|
||||
},
|
||||
"tac-qlib-rd": {
|
||||
"type": "local",
|
||||
"command": [".venv/bin/python", "-m", "tac_qlib.rd_server"],
|
||||
"enabled": true,
|
||||
"environment": {
|
||||
"TAC_LAKE_DIR": "{env:TAC_LAKE_DIR}",
|
||||
"DATABASE_URL": "{env:DATABASE_URL}"
|
||||
}
|
||||
},
|
||||
"tac-rd-book": {
|
||||
"type": "local",
|
||||
"command": [".venv/bin/python", "-m", "tac_qlib.book_server"],
|
||||
"enabled": true,
|
||||
"environment": {
|
||||
"DATABASE_URL": "{env:DATABASE_URL}",
|
||||
"APCA_API_KEY_ID": "{env:APCA_API_KEY_ID}",
|
||||
"APCA_API_SECRET_KEY": "{env:APCA_API_SECRET_KEY}",
|
||||
"APCA_API_BASE_URL": "{env:APCA_API_BASE_URL}"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,10 @@
|
||||
packages:
|
||||
- "tac-app"
|
||||
|
||||
onlyBuiltDependencies:
|
||||
- sharp
|
||||
- esbuild
|
||||
|
||||
allowBuilds:
|
||||
sharp: true
|
||||
esbuild: true
|
||||
@@ -0,0 +1,23 @@
|
||||
CREATE TABLE "rd_experiments" (
|
||||
"id" bigserial PRIMARY KEY NOT NULL,
|
||||
"experiment_name" text,
|
||||
"rational" text NOT NULL,
|
||||
"rational_embedding" vector(384),
|
||||
"details" text,
|
||||
"details_embedding" vector(384),
|
||||
"evaluation" text,
|
||||
"metrics" jsonb,
|
||||
"evolved_from" bigint,
|
||||
"start_ts" timestamp with time zone DEFAULT now() NOT NULL,
|
||||
"end_ts" timestamp with time zone,
|
||||
"git_branch" text NOT NULL,
|
||||
"experiment_ref_id" text,
|
||||
"mlruns_dir" text,
|
||||
"status" text DEFAULT 'starting' NOT NULL,
|
||||
"created_at" timestamp with time zone DEFAULT now() NOT NULL,
|
||||
"updated_at" timestamp with time zone DEFAULT now() NOT NULL
|
||||
);
|
||||
--> statement-breakpoint
|
||||
ALTER TABLE "rd_experiments" ADD CONSTRAINT "rd_experiments_evolved_from_rd_experiments_id_fk" FOREIGN KEY ("evolved_from") REFERENCES "public"."rd_experiments"("id") ON DELETE no action ON UPDATE no action;--> statement-breakpoint
|
||||
CREATE INDEX "rd_experiments_ref_id_idx" ON "rd_experiments" USING btree ("experiment_ref_id");--> statement-breakpoint
|
||||
CREATE INDEX "rd_experiments_evolved_from_idx" ON "rd_experiments" USING btree ("evolved_from");
|
||||
@@ -0,0 +1,54 @@
|
||||
CREATE TABLE "rd_models" (
|
||||
"id" bigserial PRIMARY KEY NOT NULL,
|
||||
"name" text NOT NULL,
|
||||
"description" text,
|
||||
"experiment_name" text NOT NULL,
|
||||
"run_id" text NOT NULL,
|
||||
"model_path" text,
|
||||
"universe" text,
|
||||
"label" text,
|
||||
"default_strategy" text,
|
||||
"metrics" jsonb,
|
||||
"status" text DEFAULT 'active' NOT NULL,
|
||||
"created_at" timestamp with time zone DEFAULT now() NOT NULL,
|
||||
"updated_at" timestamp with time zone DEFAULT now() NOT NULL,
|
||||
CONSTRAINT "rd_models_name_unique" UNIQUE("name")
|
||||
);
|
||||
--> statement-breakpoint
|
||||
CREATE TABLE "scheduler_jobs" (
|
||||
"id" bigserial PRIMARY KEY NOT NULL,
|
||||
"name" text,
|
||||
"city" text DEFAULT 'new-york' NOT NULL,
|
||||
"timezone" text DEFAULT 'America/New_York' NOT NULL,
|
||||
"time" text NOT NULL,
|
||||
"model_id" bigint NOT NULL,
|
||||
"strategy" text DEFAULT 'workflow_lgb_sp5d_rankic.yaml' NOT NULL,
|
||||
"enabled" boolean DEFAULT true NOT NULL,
|
||||
"last_run_at" timestamp with time zone,
|
||||
"last_status" text,
|
||||
"last_error" text,
|
||||
"created_at" timestamp with time zone DEFAULT now() NOT NULL,
|
||||
"updated_at" timestamp with time zone DEFAULT now() NOT NULL
|
||||
);
|
||||
--> statement-breakpoint
|
||||
CREATE TABLE "scheduler_runs" (
|
||||
"id" bigserial PRIMARY KEY NOT NULL,
|
||||
"job_id" bigint,
|
||||
"city" text NOT NULL,
|
||||
"model_id" bigint NOT NULL,
|
||||
"strategy" text NOT NULL,
|
||||
"title" text NOT NULL,
|
||||
"session_id" text,
|
||||
"status" text DEFAULT 'pending' NOT NULL,
|
||||
"error" text,
|
||||
"triggered_at" timestamp with time zone DEFAULT now() NOT NULL,
|
||||
"created_at" timestamp with time zone DEFAULT now() NOT NULL
|
||||
);
|
||||
--> statement-breakpoint
|
||||
ALTER TABLE "scheduler_jobs" ADD CONSTRAINT "scheduler_jobs_model_id_rd_models_id_fk" FOREIGN KEY ("model_id") REFERENCES "public"."rd_models"("id") ON DELETE no action ON UPDATE no action;--> statement-breakpoint
|
||||
CREATE INDEX "rd_models_run_id_idx" ON "rd_models" USING btree ("run_id");--> statement-breakpoint
|
||||
CREATE INDEX "rd_models_name_idx" ON "rd_models" USING btree ("name");--> statement-breakpoint
|
||||
CREATE INDEX "scheduler_jobs_model_id_idx" ON "scheduler_jobs" USING btree ("model_id");--> statement-breakpoint
|
||||
CREATE INDEX "scheduler_jobs_enabled_idx" ON "scheduler_jobs" USING btree ("enabled");--> statement-breakpoint
|
||||
CREATE INDEX "scheduler_runs_job_id_idx" ON "scheduler_runs" USING btree ("job_id");--> statement-breakpoint
|
||||
CREATE INDEX "scheduler_runs_session_id_idx" ON "scheduler_runs" USING btree ("session_id");
|
||||
@@ -0,0 +1,5 @@
|
||||
ALTER TABLE "scheduler_jobs" ALTER COLUMN "model_id" DROP NOT NULL;--> statement-breakpoint
|
||||
ALTER TABLE "scheduler_runs" ALTER COLUMN "model_id" DROP NOT NULL;--> statement-breakpoint
|
||||
ALTER TABLE "scheduler_jobs" ADD COLUMN "days" text DEFAULT '1,2,3,4,5' NOT NULL;--> statement-breakpoint
|
||||
ALTER TABLE "scheduler_jobs" ADD COLUMN "experiment_name" text;--> statement-breakpoint
|
||||
ALTER TABLE "scheduler_jobs" ADD COLUMN "run_id" text;
|
||||
@@ -0,0 +1,3 @@
|
||||
ALTER TABLE "scheduler_jobs" ALTER COLUMN "strategy" DROP DEFAULT;--> statement-breakpoint
|
||||
ALTER TABLE "scheduler_jobs" ALTER COLUMN "strategy" DROP NOT NULL;--> statement-breakpoint
|
||||
ALTER TABLE "scheduler_runs" ALTER COLUMN "strategy" DROP NOT NULL;
|
||||
@@ -0,0 +1 @@
|
||||
ALTER TABLE "scheduler_runs" ADD COLUMN "source" text DEFAULT 'scheduled' NOT NULL;
|
||||
@@ -0,0 +1,98 @@
|
||||
CREATE TABLE "fact_events" (
|
||||
"id" bigserial PRIMARY KEY NOT NULL,
|
||||
"round_id" bigint NOT NULL,
|
||||
"kind" text NOT NULL,
|
||||
"symbol" text,
|
||||
"payload" jsonb,
|
||||
"source" text,
|
||||
"at" timestamp with time zone DEFAULT now() NOT NULL
|
||||
);
|
||||
--> statement-breakpoint
|
||||
CREATE TABLE "round_decisions" (
|
||||
"id" bigserial PRIMARY KEY NOT NULL,
|
||||
"round_id" bigint NOT NULL,
|
||||
"intent_id" bigint,
|
||||
"symbol" text NOT NULL,
|
||||
"side" text NOT NULL,
|
||||
"qty" numeric,
|
||||
"order_type" text,
|
||||
"expected_price" numeric,
|
||||
"status" text DEFAULT 'intended' NOT NULL,
|
||||
"reason" text,
|
||||
"reason_detail" text,
|
||||
"superseded_by_decision_id" bigint,
|
||||
"created_at" timestamp with time zone DEFAULT now() NOT NULL,
|
||||
"updated_at" timestamp with time zone DEFAULT now() NOT NULL
|
||||
);
|
||||
--> statement-breakpoint
|
||||
CREATE TABLE "round_intents" (
|
||||
"id" bigserial PRIMARY KEY NOT NULL,
|
||||
"round_id" bigint NOT NULL,
|
||||
"version" bigint NOT NULL,
|
||||
"supersedes_intent_id" bigint,
|
||||
"target_portfolio" jsonb,
|
||||
"raw_strategy_output" jsonb,
|
||||
"reason" text,
|
||||
"created_at" timestamp with time zone DEFAULT now() NOT NULL
|
||||
);
|
||||
--> statement-breakpoint
|
||||
CREATE TABLE "round_orders" (
|
||||
"id" bigserial PRIMARY KEY NOT NULL,
|
||||
"decision_id" bigint NOT NULL,
|
||||
"round_id" bigint NOT NULL,
|
||||
"alpaca_order_id" text,
|
||||
"client_order_id" text,
|
||||
"qty_intended" numeric,
|
||||
"qty_filled" numeric DEFAULT '0' NOT NULL,
|
||||
"avg_fill_price" numeric,
|
||||
"status" text DEFAULT 'accepted' NOT NULL,
|
||||
"superseded_by_order_id" bigint,
|
||||
"created_at" timestamp with time zone DEFAULT now() NOT NULL,
|
||||
"updated_at" timestamp with time zone DEFAULT now() NOT NULL
|
||||
);
|
||||
--> statement-breakpoint
|
||||
CREATE TABLE "trading_rounds" (
|
||||
"id" bigserial PRIMARY KEY NOT NULL,
|
||||
"source" text DEFAULT 'scheduled' NOT NULL,
|
||||
"target_date" date NOT NULL,
|
||||
"signal_date" date,
|
||||
"scheduler_run_id" bigint,
|
||||
"rd_experiment_id" bigint,
|
||||
"experiment_name" text,
|
||||
"run_id" text,
|
||||
"model_path" text,
|
||||
"strategy_snapshot" jsonb,
|
||||
"account_equity_at_sizing" numeric,
|
||||
"status" text DEFAULT 'open' NOT NULL,
|
||||
"locked_intent_id" bigint,
|
||||
"summary_metrics" jsonb,
|
||||
"feedback_note" text,
|
||||
"created_at" timestamp with time zone DEFAULT now() NOT NULL,
|
||||
"updated_at" timestamp with time zone DEFAULT now() NOT NULL
|
||||
);
|
||||
--> statement-breakpoint
|
||||
ALTER TABLE "fact_events" ADD CONSTRAINT "fact_events_round_id_trading_rounds_id_fk" FOREIGN KEY ("round_id") REFERENCES "public"."trading_rounds"("id") ON DELETE no action ON UPDATE no action;--> statement-breakpoint
|
||||
ALTER TABLE "round_decisions" ADD CONSTRAINT "round_decisions_round_id_trading_rounds_id_fk" FOREIGN KEY ("round_id") REFERENCES "public"."trading_rounds"("id") ON DELETE no action ON UPDATE no action;--> statement-breakpoint
|
||||
ALTER TABLE "round_decisions" ADD CONSTRAINT "round_decisions_intent_id_round_intents_id_fk" FOREIGN KEY ("intent_id") REFERENCES "public"."round_intents"("id") ON DELETE no action ON UPDATE no action;--> statement-breakpoint
|
||||
ALTER TABLE "round_decisions" ADD CONSTRAINT "round_decisions_superseded_by_round_decisions_id_fk" FOREIGN KEY ("superseded_by_decision_id") REFERENCES "public"."round_decisions"("id") ON DELETE no action ON UPDATE no action;--> statement-breakpoint
|
||||
ALTER TABLE "round_intents" ADD CONSTRAINT "round_intents_round_id_trading_rounds_id_fk" FOREIGN KEY ("round_id") REFERENCES "public"."trading_rounds"("id") ON DELETE no action ON UPDATE no action;--> statement-breakpoint
|
||||
ALTER TABLE "round_intents" ADD CONSTRAINT "round_intents_supersedes_intent_id_round_intents_id_fk" FOREIGN KEY ("supersedes_intent_id") REFERENCES "public"."round_intents"("id") ON DELETE no action ON UPDATE no action;--> statement-breakpoint
|
||||
ALTER TABLE "round_orders" ADD CONSTRAINT "round_orders_round_id_trading_rounds_id_fk" FOREIGN KEY ("round_id") REFERENCES "public"."trading_rounds"("id") ON DELETE no action ON UPDATE no action;--> statement-breakpoint
|
||||
ALTER TABLE "round_orders" ADD CONSTRAINT "round_orders_decision_id_round_decisions_id_fk" FOREIGN KEY ("decision_id") REFERENCES "public"."round_decisions"("id") ON DELETE no action ON UPDATE no action;--> statement-breakpoint
|
||||
ALTER TABLE "round_orders" ADD CONSTRAINT "round_orders_superseded_by_round_orders_id_fk" FOREIGN KEY ("superseded_by_order_id") REFERENCES "public"."round_orders"("id") ON DELETE no action ON UPDATE no action;--> statement-breakpoint
|
||||
ALTER TABLE "trading_rounds" ADD CONSTRAINT "trading_rounds_scheduler_run_id_scheduler_runs_id_fk" FOREIGN KEY ("scheduler_run_id") REFERENCES "public"."scheduler_runs"("id") ON DELETE no action ON UPDATE no action;--> statement-breakpoint
|
||||
ALTER TABLE "trading_rounds" ADD CONSTRAINT "trading_rounds_rd_experiment_id_rd_experiments_id_fk" FOREIGN KEY ("rd_experiment_id") REFERENCES "public"."rd_experiments"("id") ON DELETE no action ON UPDATE no action;--> statement-breakpoint
|
||||
CREATE INDEX "fact_events_round_id_idx" ON "fact_events" USING btree ("round_id");--> statement-breakpoint
|
||||
CREATE INDEX "fact_events_round_kind_idx" ON "fact_events" USING btree ("round_id","kind");--> statement-breakpoint
|
||||
CREATE INDEX "round_decisions_round_id_idx" ON "round_decisions" USING btree ("round_id");--> statement-breakpoint
|
||||
CREATE INDEX "round_decisions_round_symbol_idx" ON "round_decisions" USING btree ("round_id","symbol");--> statement-breakpoint
|
||||
CREATE INDEX "round_decisions_intent_id_idx" ON "round_decisions" USING btree ("intent_id");--> statement-breakpoint
|
||||
CREATE INDEX "round_intents_round_id_idx" ON "round_intents" USING btree ("round_id");--> statement-breakpoint
|
||||
CREATE INDEX "round_intents_round_version_idx" ON "round_intents" USING btree ("round_id","version");--> statement-breakpoint
|
||||
CREATE INDEX "round_orders_round_id_idx" ON "round_orders" USING btree ("round_id");--> statement-breakpoint
|
||||
CREATE INDEX "round_orders_decision_id_idx" ON "round_orders" USING btree ("decision_id");--> statement-breakpoint
|
||||
CREATE INDEX "round_orders_alpaca_order_id_idx" ON "round_orders" USING btree ("alpaca_order_id");--> statement-breakpoint
|
||||
CREATE INDEX "trading_rounds_target_date_idx" ON "trading_rounds" USING btree ("target_date");--> statement-breakpoint
|
||||
CREATE INDEX "trading_rounds_scheduler_run_id_idx" ON "trading_rounds" USING btree ("scheduler_run_id");--> statement-breakpoint
|
||||
CREATE INDEX "trading_rounds_rd_experiment_id_idx" ON "trading_rounds" USING btree ("rd_experiment_id");--> statement-breakpoint
|
||||
CREATE INDEX "trading_rounds_locked_intent_id_idx" ON "trading_rounds" USING btree ("locked_intent_id");
|
||||
@@ -0,0 +1 @@
|
||||
ALTER TABLE "rd_experiments" ADD COLUMN IF NOT EXISTS "session_id" text;
|
||||
@@ -0,0 +1,183 @@
|
||||
{
|
||||
"id": "07d19100-254c-4adf-8b83-480fa6ffc00e",
|
||||
"prevId": "00000000-0000-0000-0000-000000000000",
|
||||
"version": "7",
|
||||
"dialect": "postgresql",
|
||||
"tables": {
|
||||
"public.rd_experiments": {
|
||||
"name": "rd_experiments",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"experiment_name": {
|
||||
"name": "experiment_name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"rational": {
|
||||
"name": "rational",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"rational_embedding": {
|
||||
"name": "rational_embedding",
|
||||
"type": "vector(384)",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"details": {
|
||||
"name": "details",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"details_embedding": {
|
||||
"name": "details_embedding",
|
||||
"type": "vector(384)",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"evaluation": {
|
||||
"name": "evaluation",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"metrics": {
|
||||
"name": "metrics",
|
||||
"type": "jsonb",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"evolved_from": {
|
||||
"name": "evolved_from",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"start_ts": {
|
||||
"name": "start_ts",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"end_ts": {
|
||||
"name": "end_ts",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"git_branch": {
|
||||
"name": "git_branch",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"experiment_ref_id": {
|
||||
"name": "experiment_ref_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"mlruns_dir": {
|
||||
"name": "mlruns_dir",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"status": {
|
||||
"name": "status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'starting'"
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"updated_at": {
|
||||
"name": "updated_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"rd_experiments_ref_id_idx": {
|
||||
"name": "rd_experiments_ref_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "experiment_ref_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"rd_experiments_evolved_from_idx": {
|
||||
"name": "rd_experiments_evolved_from_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "evolved_from",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {
|
||||
"rd_experiments_evolved_from_rd_experiments_id_fk": {
|
||||
"name": "rd_experiments_evolved_from_rd_experiments_id_fk",
|
||||
"tableFrom": "rd_experiments",
|
||||
"tableTo": "rd_experiments",
|
||||
"columnsFrom": [
|
||||
"evolved_from"
|
||||
],
|
||||
"columnsTo": [
|
||||
"id"
|
||||
],
|
||||
"onDelete": "no action",
|
||||
"onUpdate": "no action"
|
||||
}
|
||||
},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
}
|
||||
},
|
||||
"enums": {},
|
||||
"schemas": {},
|
||||
"sequences": {},
|
||||
"roles": {},
|
||||
"policies": {},
|
||||
"views": {},
|
||||
"_meta": {
|
||||
"columns": {},
|
||||
"schemas": {},
|
||||
"tables": {}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,571 @@
|
||||
{
|
||||
"id": "e7722810-0b69-4a2d-a5b0-88df6939faa3",
|
||||
"prevId": "07d19100-254c-4adf-8b83-480fa6ffc00e",
|
||||
"version": "7",
|
||||
"dialect": "postgresql",
|
||||
"tables": {
|
||||
"public.rd_experiments": {
|
||||
"name": "rd_experiments",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"experiment_name": {
|
||||
"name": "experiment_name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"rational": {
|
||||
"name": "rational",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"rational_embedding": {
|
||||
"name": "rational_embedding",
|
||||
"type": "vector(384)",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"details": {
|
||||
"name": "details",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"details_embedding": {
|
||||
"name": "details_embedding",
|
||||
"type": "vector(384)",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"evaluation": {
|
||||
"name": "evaluation",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"metrics": {
|
||||
"name": "metrics",
|
||||
"type": "jsonb",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"evolved_from": {
|
||||
"name": "evolved_from",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"start_ts": {
|
||||
"name": "start_ts",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"end_ts": {
|
||||
"name": "end_ts",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"git_branch": {
|
||||
"name": "git_branch",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"experiment_ref_id": {
|
||||
"name": "experiment_ref_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"mlruns_dir": {
|
||||
"name": "mlruns_dir",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"status": {
|
||||
"name": "status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'starting'"
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"updated_at": {
|
||||
"name": "updated_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"rd_experiments_ref_id_idx": {
|
||||
"name": "rd_experiments_ref_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "experiment_ref_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"rd_experiments_evolved_from_idx": {
|
||||
"name": "rd_experiments_evolved_from_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "evolved_from",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {
|
||||
"rd_experiments_evolved_from_rd_experiments_id_fk": {
|
||||
"name": "rd_experiments_evolved_from_rd_experiments_id_fk",
|
||||
"tableFrom": "rd_experiments",
|
||||
"tableTo": "rd_experiments",
|
||||
"columnsFrom": [
|
||||
"evolved_from"
|
||||
],
|
||||
"columnsTo": [
|
||||
"id"
|
||||
],
|
||||
"onDelete": "no action",
|
||||
"onUpdate": "no action"
|
||||
}
|
||||
},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
},
|
||||
"public.rd_models": {
|
||||
"name": "rd_models",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"name": {
|
||||
"name": "name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"description": {
|
||||
"name": "description",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"experiment_name": {
|
||||
"name": "experiment_name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"run_id": {
|
||||
"name": "run_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"model_path": {
|
||||
"name": "model_path",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"universe": {
|
||||
"name": "universe",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"label": {
|
||||
"name": "label",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"default_strategy": {
|
||||
"name": "default_strategy",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"metrics": {
|
||||
"name": "metrics",
|
||||
"type": "jsonb",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"status": {
|
||||
"name": "status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'active'"
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"updated_at": {
|
||||
"name": "updated_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"rd_models_run_id_idx": {
|
||||
"name": "rd_models_run_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "run_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"rd_models_name_idx": {
|
||||
"name": "rd_models_name_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "name",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {
|
||||
"rd_models_name_unique": {
|
||||
"name": "rd_models_name_unique",
|
||||
"nullsNotDistinct": false,
|
||||
"columns": [
|
||||
"name"
|
||||
]
|
||||
}
|
||||
},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
},
|
||||
"public.scheduler_jobs": {
|
||||
"name": "scheduler_jobs",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"name": {
|
||||
"name": "name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"city": {
|
||||
"name": "city",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'new-york'"
|
||||
},
|
||||
"timezone": {
|
||||
"name": "timezone",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'America/New_York'"
|
||||
},
|
||||
"time": {
|
||||
"name": "time",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"model_id": {
|
||||
"name": "model_id",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"strategy": {
|
||||
"name": "strategy",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'workflow_lgb_sp5d_rankic.yaml'"
|
||||
},
|
||||
"enabled": {
|
||||
"name": "enabled",
|
||||
"type": "boolean",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": true
|
||||
},
|
||||
"last_run_at": {
|
||||
"name": "last_run_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"last_status": {
|
||||
"name": "last_status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"last_error": {
|
||||
"name": "last_error",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"updated_at": {
|
||||
"name": "updated_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"scheduler_jobs_model_id_idx": {
|
||||
"name": "scheduler_jobs_model_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "model_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"scheduler_jobs_enabled_idx": {
|
||||
"name": "scheduler_jobs_enabled_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "enabled",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {
|
||||
"scheduler_jobs_model_id_rd_models_id_fk": {
|
||||
"name": "scheduler_jobs_model_id_rd_models_id_fk",
|
||||
"tableFrom": "scheduler_jobs",
|
||||
"tableTo": "rd_models",
|
||||
"columnsFrom": [
|
||||
"model_id"
|
||||
],
|
||||
"columnsTo": [
|
||||
"id"
|
||||
],
|
||||
"onDelete": "no action",
|
||||
"onUpdate": "no action"
|
||||
}
|
||||
},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
},
|
||||
"public.scheduler_runs": {
|
||||
"name": "scheduler_runs",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"job_id": {
|
||||
"name": "job_id",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"city": {
|
||||
"name": "city",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"model_id": {
|
||||
"name": "model_id",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"strategy": {
|
||||
"name": "strategy",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"title": {
|
||||
"name": "title",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"session_id": {
|
||||
"name": "session_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"status": {
|
||||
"name": "status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'pending'"
|
||||
},
|
||||
"error": {
|
||||
"name": "error",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"triggered_at": {
|
||||
"name": "triggered_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"scheduler_runs_job_id_idx": {
|
||||
"name": "scheduler_runs_job_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "job_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"scheduler_runs_session_id_idx": {
|
||||
"name": "scheduler_runs_session_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "session_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
}
|
||||
},
|
||||
"enums": {},
|
||||
"schemas": {},
|
||||
"sequences": {},
|
||||
"roles": {},
|
||||
"policies": {},
|
||||
"views": {},
|
||||
"_meta": {
|
||||
"columns": {},
|
||||
"schemas": {},
|
||||
"tables": {}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,590 @@
|
||||
{
|
||||
"id": "1123264a-7a95-4474-9fab-68662142abf7",
|
||||
"prevId": "e7722810-0b69-4a2d-a5b0-88df6939faa3",
|
||||
"version": "7",
|
||||
"dialect": "postgresql",
|
||||
"tables": {
|
||||
"public.rd_experiments": {
|
||||
"name": "rd_experiments",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"experiment_name": {
|
||||
"name": "experiment_name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"rational": {
|
||||
"name": "rational",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"rational_embedding": {
|
||||
"name": "rational_embedding",
|
||||
"type": "vector(384)",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"details": {
|
||||
"name": "details",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"details_embedding": {
|
||||
"name": "details_embedding",
|
||||
"type": "vector(384)",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"evaluation": {
|
||||
"name": "evaluation",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"metrics": {
|
||||
"name": "metrics",
|
||||
"type": "jsonb",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"evolved_from": {
|
||||
"name": "evolved_from",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"start_ts": {
|
||||
"name": "start_ts",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"end_ts": {
|
||||
"name": "end_ts",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"git_branch": {
|
||||
"name": "git_branch",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"experiment_ref_id": {
|
||||
"name": "experiment_ref_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"mlruns_dir": {
|
||||
"name": "mlruns_dir",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"status": {
|
||||
"name": "status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'starting'"
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"updated_at": {
|
||||
"name": "updated_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"rd_experiments_ref_id_idx": {
|
||||
"name": "rd_experiments_ref_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "experiment_ref_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"rd_experiments_evolved_from_idx": {
|
||||
"name": "rd_experiments_evolved_from_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "evolved_from",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {
|
||||
"rd_experiments_evolved_from_rd_experiments_id_fk": {
|
||||
"name": "rd_experiments_evolved_from_rd_experiments_id_fk",
|
||||
"tableFrom": "rd_experiments",
|
||||
"tableTo": "rd_experiments",
|
||||
"columnsFrom": [
|
||||
"evolved_from"
|
||||
],
|
||||
"columnsTo": [
|
||||
"id"
|
||||
],
|
||||
"onDelete": "no action",
|
||||
"onUpdate": "no action"
|
||||
}
|
||||
},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
},
|
||||
"public.rd_models": {
|
||||
"name": "rd_models",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"name": {
|
||||
"name": "name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"description": {
|
||||
"name": "description",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"experiment_name": {
|
||||
"name": "experiment_name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"run_id": {
|
||||
"name": "run_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"model_path": {
|
||||
"name": "model_path",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"universe": {
|
||||
"name": "universe",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"label": {
|
||||
"name": "label",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"default_strategy": {
|
||||
"name": "default_strategy",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"metrics": {
|
||||
"name": "metrics",
|
||||
"type": "jsonb",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"status": {
|
||||
"name": "status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'active'"
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"updated_at": {
|
||||
"name": "updated_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"rd_models_run_id_idx": {
|
||||
"name": "rd_models_run_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "run_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"rd_models_name_idx": {
|
||||
"name": "rd_models_name_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "name",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {
|
||||
"rd_models_name_unique": {
|
||||
"name": "rd_models_name_unique",
|
||||
"nullsNotDistinct": false,
|
||||
"columns": [
|
||||
"name"
|
||||
]
|
||||
}
|
||||
},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
},
|
||||
"public.scheduler_jobs": {
|
||||
"name": "scheduler_jobs",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"name": {
|
||||
"name": "name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"city": {
|
||||
"name": "city",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'new-york'"
|
||||
},
|
||||
"timezone": {
|
||||
"name": "timezone",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'America/New_York'"
|
||||
},
|
||||
"time": {
|
||||
"name": "time",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"days": {
|
||||
"name": "days",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'1,2,3,4,5'"
|
||||
},
|
||||
"experiment_name": {
|
||||
"name": "experiment_name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"run_id": {
|
||||
"name": "run_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"model_id": {
|
||||
"name": "model_id",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"strategy": {
|
||||
"name": "strategy",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'workflow_lgb_sp5d_rankic.yaml'"
|
||||
},
|
||||
"enabled": {
|
||||
"name": "enabled",
|
||||
"type": "boolean",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": true
|
||||
},
|
||||
"last_run_at": {
|
||||
"name": "last_run_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"last_status": {
|
||||
"name": "last_status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"last_error": {
|
||||
"name": "last_error",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"updated_at": {
|
||||
"name": "updated_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"scheduler_jobs_model_id_idx": {
|
||||
"name": "scheduler_jobs_model_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "model_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"scheduler_jobs_enabled_idx": {
|
||||
"name": "scheduler_jobs_enabled_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "enabled",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {
|
||||
"scheduler_jobs_model_id_rd_models_id_fk": {
|
||||
"name": "scheduler_jobs_model_id_rd_models_id_fk",
|
||||
"tableFrom": "scheduler_jobs",
|
||||
"tableTo": "rd_models",
|
||||
"columnsFrom": [
|
||||
"model_id"
|
||||
],
|
||||
"columnsTo": [
|
||||
"id"
|
||||
],
|
||||
"onDelete": "no action",
|
||||
"onUpdate": "no action"
|
||||
}
|
||||
},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
},
|
||||
"public.scheduler_runs": {
|
||||
"name": "scheduler_runs",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"job_id": {
|
||||
"name": "job_id",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"city": {
|
||||
"name": "city",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"model_id": {
|
||||
"name": "model_id",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"strategy": {
|
||||
"name": "strategy",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"title": {
|
||||
"name": "title",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"session_id": {
|
||||
"name": "session_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"status": {
|
||||
"name": "status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'pending'"
|
||||
},
|
||||
"error": {
|
||||
"name": "error",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"triggered_at": {
|
||||
"name": "triggered_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"scheduler_runs_job_id_idx": {
|
||||
"name": "scheduler_runs_job_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "job_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"scheduler_runs_session_id_idx": {
|
||||
"name": "scheduler_runs_session_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "session_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
}
|
||||
},
|
||||
"enums": {},
|
||||
"schemas": {},
|
||||
"sequences": {},
|
||||
"roles": {},
|
||||
"policies": {},
|
||||
"views": {},
|
||||
"_meta": {
|
||||
"columns": {},
|
||||
"schemas": {},
|
||||
"tables": {}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,589 @@
|
||||
{
|
||||
"id": "ef6bf221-a986-47ea-9dee-a7678df84502",
|
||||
"prevId": "1123264a-7a95-4474-9fab-68662142abf7",
|
||||
"version": "7",
|
||||
"dialect": "postgresql",
|
||||
"tables": {
|
||||
"public.rd_experiments": {
|
||||
"name": "rd_experiments",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"experiment_name": {
|
||||
"name": "experiment_name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"rational": {
|
||||
"name": "rational",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"rational_embedding": {
|
||||
"name": "rational_embedding",
|
||||
"type": "vector(384)",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"details": {
|
||||
"name": "details",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"details_embedding": {
|
||||
"name": "details_embedding",
|
||||
"type": "vector(384)",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"evaluation": {
|
||||
"name": "evaluation",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"metrics": {
|
||||
"name": "metrics",
|
||||
"type": "jsonb",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"evolved_from": {
|
||||
"name": "evolved_from",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"start_ts": {
|
||||
"name": "start_ts",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"end_ts": {
|
||||
"name": "end_ts",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"git_branch": {
|
||||
"name": "git_branch",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"experiment_ref_id": {
|
||||
"name": "experiment_ref_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"mlruns_dir": {
|
||||
"name": "mlruns_dir",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"status": {
|
||||
"name": "status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'starting'"
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"updated_at": {
|
||||
"name": "updated_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"rd_experiments_ref_id_idx": {
|
||||
"name": "rd_experiments_ref_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "experiment_ref_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"rd_experiments_evolved_from_idx": {
|
||||
"name": "rd_experiments_evolved_from_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "evolved_from",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {
|
||||
"rd_experiments_evolved_from_rd_experiments_id_fk": {
|
||||
"name": "rd_experiments_evolved_from_rd_experiments_id_fk",
|
||||
"tableFrom": "rd_experiments",
|
||||
"tableTo": "rd_experiments",
|
||||
"columnsFrom": [
|
||||
"evolved_from"
|
||||
],
|
||||
"columnsTo": [
|
||||
"id"
|
||||
],
|
||||
"onDelete": "no action",
|
||||
"onUpdate": "no action"
|
||||
}
|
||||
},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
},
|
||||
"public.rd_models": {
|
||||
"name": "rd_models",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"name": {
|
||||
"name": "name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"description": {
|
||||
"name": "description",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"experiment_name": {
|
||||
"name": "experiment_name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"run_id": {
|
||||
"name": "run_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"model_path": {
|
||||
"name": "model_path",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"universe": {
|
||||
"name": "universe",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"label": {
|
||||
"name": "label",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"default_strategy": {
|
||||
"name": "default_strategy",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"metrics": {
|
||||
"name": "metrics",
|
||||
"type": "jsonb",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"status": {
|
||||
"name": "status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'active'"
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"updated_at": {
|
||||
"name": "updated_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"rd_models_run_id_idx": {
|
||||
"name": "rd_models_run_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "run_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"rd_models_name_idx": {
|
||||
"name": "rd_models_name_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "name",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {
|
||||
"rd_models_name_unique": {
|
||||
"name": "rd_models_name_unique",
|
||||
"nullsNotDistinct": false,
|
||||
"columns": [
|
||||
"name"
|
||||
]
|
||||
}
|
||||
},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
},
|
||||
"public.scheduler_jobs": {
|
||||
"name": "scheduler_jobs",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"name": {
|
||||
"name": "name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"city": {
|
||||
"name": "city",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'new-york'"
|
||||
},
|
||||
"timezone": {
|
||||
"name": "timezone",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'America/New_York'"
|
||||
},
|
||||
"time": {
|
||||
"name": "time",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"days": {
|
||||
"name": "days",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'1,2,3,4,5'"
|
||||
},
|
||||
"experiment_name": {
|
||||
"name": "experiment_name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"run_id": {
|
||||
"name": "run_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"model_id": {
|
||||
"name": "model_id",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"strategy": {
|
||||
"name": "strategy",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"enabled": {
|
||||
"name": "enabled",
|
||||
"type": "boolean",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": true
|
||||
},
|
||||
"last_run_at": {
|
||||
"name": "last_run_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"last_status": {
|
||||
"name": "last_status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"last_error": {
|
||||
"name": "last_error",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"updated_at": {
|
||||
"name": "updated_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"scheduler_jobs_model_id_idx": {
|
||||
"name": "scheduler_jobs_model_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "model_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"scheduler_jobs_enabled_idx": {
|
||||
"name": "scheduler_jobs_enabled_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "enabled",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {
|
||||
"scheduler_jobs_model_id_rd_models_id_fk": {
|
||||
"name": "scheduler_jobs_model_id_rd_models_id_fk",
|
||||
"tableFrom": "scheduler_jobs",
|
||||
"tableTo": "rd_models",
|
||||
"columnsFrom": [
|
||||
"model_id"
|
||||
],
|
||||
"columnsTo": [
|
||||
"id"
|
||||
],
|
||||
"onDelete": "no action",
|
||||
"onUpdate": "no action"
|
||||
}
|
||||
},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
},
|
||||
"public.scheduler_runs": {
|
||||
"name": "scheduler_runs",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"job_id": {
|
||||
"name": "job_id",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"city": {
|
||||
"name": "city",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"model_id": {
|
||||
"name": "model_id",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"strategy": {
|
||||
"name": "strategy",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"title": {
|
||||
"name": "title",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"session_id": {
|
||||
"name": "session_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"status": {
|
||||
"name": "status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'pending'"
|
||||
},
|
||||
"error": {
|
||||
"name": "error",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"triggered_at": {
|
||||
"name": "triggered_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"scheduler_runs_job_id_idx": {
|
||||
"name": "scheduler_runs_job_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "job_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"scheduler_runs_session_id_idx": {
|
||||
"name": "scheduler_runs_session_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "session_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
}
|
||||
},
|
||||
"enums": {},
|
||||
"schemas": {},
|
||||
"sequences": {},
|
||||
"roles": {},
|
||||
"policies": {},
|
||||
"views": {},
|
||||
"_meta": {
|
||||
"columns": {},
|
||||
"schemas": {},
|
||||
"tables": {}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,596 @@
|
||||
{
|
||||
"id": "ac54b5f9-a7b6-4ffa-8da1-adab67393980",
|
||||
"prevId": "ef6bf221-a986-47ea-9dee-a7678df84502",
|
||||
"version": "7",
|
||||
"dialect": "postgresql",
|
||||
"tables": {
|
||||
"public.rd_experiments": {
|
||||
"name": "rd_experiments",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"experiment_name": {
|
||||
"name": "experiment_name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"rational": {
|
||||
"name": "rational",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"rational_embedding": {
|
||||
"name": "rational_embedding",
|
||||
"type": "vector(384)",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"details": {
|
||||
"name": "details",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"details_embedding": {
|
||||
"name": "details_embedding",
|
||||
"type": "vector(384)",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"evaluation": {
|
||||
"name": "evaluation",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"metrics": {
|
||||
"name": "metrics",
|
||||
"type": "jsonb",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"evolved_from": {
|
||||
"name": "evolved_from",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"start_ts": {
|
||||
"name": "start_ts",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"end_ts": {
|
||||
"name": "end_ts",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"git_branch": {
|
||||
"name": "git_branch",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"experiment_ref_id": {
|
||||
"name": "experiment_ref_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"mlruns_dir": {
|
||||
"name": "mlruns_dir",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"status": {
|
||||
"name": "status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'starting'"
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"updated_at": {
|
||||
"name": "updated_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"rd_experiments_ref_id_idx": {
|
||||
"name": "rd_experiments_ref_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "experiment_ref_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"rd_experiments_evolved_from_idx": {
|
||||
"name": "rd_experiments_evolved_from_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "evolved_from",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {
|
||||
"rd_experiments_evolved_from_rd_experiments_id_fk": {
|
||||
"name": "rd_experiments_evolved_from_rd_experiments_id_fk",
|
||||
"tableFrom": "rd_experiments",
|
||||
"tableTo": "rd_experiments",
|
||||
"columnsFrom": [
|
||||
"evolved_from"
|
||||
],
|
||||
"columnsTo": [
|
||||
"id"
|
||||
],
|
||||
"onDelete": "no action",
|
||||
"onUpdate": "no action"
|
||||
}
|
||||
},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
},
|
||||
"public.rd_models": {
|
||||
"name": "rd_models",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"name": {
|
||||
"name": "name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"description": {
|
||||
"name": "description",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"experiment_name": {
|
||||
"name": "experiment_name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"run_id": {
|
||||
"name": "run_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"model_path": {
|
||||
"name": "model_path",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"universe": {
|
||||
"name": "universe",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"label": {
|
||||
"name": "label",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"default_strategy": {
|
||||
"name": "default_strategy",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"metrics": {
|
||||
"name": "metrics",
|
||||
"type": "jsonb",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"status": {
|
||||
"name": "status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'active'"
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"updated_at": {
|
||||
"name": "updated_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"rd_models_run_id_idx": {
|
||||
"name": "rd_models_run_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "run_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"rd_models_name_idx": {
|
||||
"name": "rd_models_name_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "name",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {
|
||||
"rd_models_name_unique": {
|
||||
"name": "rd_models_name_unique",
|
||||
"nullsNotDistinct": false,
|
||||
"columns": [
|
||||
"name"
|
||||
]
|
||||
}
|
||||
},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
},
|
||||
"public.scheduler_jobs": {
|
||||
"name": "scheduler_jobs",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"name": {
|
||||
"name": "name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"city": {
|
||||
"name": "city",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'new-york'"
|
||||
},
|
||||
"timezone": {
|
||||
"name": "timezone",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'America/New_York'"
|
||||
},
|
||||
"time": {
|
||||
"name": "time",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"days": {
|
||||
"name": "days",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'1,2,3,4,5'"
|
||||
},
|
||||
"experiment_name": {
|
||||
"name": "experiment_name",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"run_id": {
|
||||
"name": "run_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"model_id": {
|
||||
"name": "model_id",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"strategy": {
|
||||
"name": "strategy",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"enabled": {
|
||||
"name": "enabled",
|
||||
"type": "boolean",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": true
|
||||
},
|
||||
"last_run_at": {
|
||||
"name": "last_run_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"last_status": {
|
||||
"name": "last_status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"last_error": {
|
||||
"name": "last_error",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"updated_at": {
|
||||
"name": "updated_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"scheduler_jobs_model_id_idx": {
|
||||
"name": "scheduler_jobs_model_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "model_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"scheduler_jobs_enabled_idx": {
|
||||
"name": "scheduler_jobs_enabled_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "enabled",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {
|
||||
"scheduler_jobs_model_id_rd_models_id_fk": {
|
||||
"name": "scheduler_jobs_model_id_rd_models_id_fk",
|
||||
"tableFrom": "scheduler_jobs",
|
||||
"tableTo": "rd_models",
|
||||
"columnsFrom": [
|
||||
"model_id"
|
||||
],
|
||||
"columnsTo": [
|
||||
"id"
|
||||
],
|
||||
"onDelete": "no action",
|
||||
"onUpdate": "no action"
|
||||
}
|
||||
},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
},
|
||||
"public.scheduler_runs": {
|
||||
"name": "scheduler_runs",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "bigserial",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"job_id": {
|
||||
"name": "job_id",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"city": {
|
||||
"name": "city",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"model_id": {
|
||||
"name": "model_id",
|
||||
"type": "bigint",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"strategy": {
|
||||
"name": "strategy",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"title": {
|
||||
"name": "title",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"session_id": {
|
||||
"name": "session_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"status": {
|
||||
"name": "status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'pending'"
|
||||
},
|
||||
"error": {
|
||||
"name": "error",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"source": {
|
||||
"name": "source",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'scheduled'"
|
||||
},
|
||||
"triggered_at": {
|
||||
"name": "triggered_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"scheduler_runs_job_id_idx": {
|
||||
"name": "scheduler_runs_job_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "job_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
},
|
||||
"scheduler_runs_session_id_idx": {
|
||||
"name": "scheduler_runs_session_id_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "session_id",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
}
|
||||
},
|
||||
"enums": {},
|
||||
"schemas": {},
|
||||
"sequences": {},
|
||||
"roles": {},
|
||||
"policies": {},
|
||||
"views": {},
|
||||
"_meta": {
|
||||
"columns": {},
|
||||
"schemas": {},
|
||||
"tables": {}
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,55 @@
|
||||
{
|
||||
"version": "7",
|
||||
"dialect": "postgresql",
|
||||
"entries": [
|
||||
{
|
||||
"idx": 0,
|
||||
"version": "7",
|
||||
"when": 1786550810655,
|
||||
"tag": "0000_dizzy_mister_fear",
|
||||
"breakpoints": true
|
||||
},
|
||||
{
|
||||
"idx": 1,
|
||||
"version": "7",
|
||||
"when": 1786598670922,
|
||||
"tag": "0001_models_and_scheduler",
|
||||
"breakpoints": true
|
||||
},
|
||||
{
|
||||
"idx": 2,
|
||||
"version": "7",
|
||||
"when": 1786608356247,
|
||||
"tag": "0002_even_mister_sinister",
|
||||
"breakpoints": true
|
||||
},
|
||||
{
|
||||
"idx": 3,
|
||||
"version": "7",
|
||||
"when": 1786609840569,
|
||||
"tag": "0003_large_lifeguard",
|
||||
"breakpoints": true
|
||||
},
|
||||
{
|
||||
"idx": 4,
|
||||
"version": "7",
|
||||
"when": 1786691576848,
|
||||
"tag": "0004_mature_pepper_potts",
|
||||
"breakpoints": true
|
||||
},
|
||||
{
|
||||
"idx": 5,
|
||||
"version": "7",
|
||||
"when": 1786792501818,
|
||||
"tag": "0005_trading-round-book",
|
||||
"breakpoints": true
|
||||
},
|
||||
{
|
||||
"idx": 6,
|
||||
"version": "7",
|
||||
"when": 1786881100668,
|
||||
"tag": "0006_rd_experiments_session_id",
|
||||
"breakpoints": true
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
import { existsSync, readFileSync } from "node:fs";
|
||||
import path from "node:path";
|
||||
import { fileURLToPath } from "node:url";
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const workspaceRoot = path.resolve(__dirname, "..");
|
||||
|
||||
// Load the single repo-root `.env` into process.env for dev/build/start.
|
||||
//
|
||||
// This must live in next.config (not `node --env-file=...`): Next forwards the
|
||||
// original node CLI flags to its worker processes via NODE_OPTIONS, where
|
||||
// `--env-file`/`--env-file-if-exists` are rejected. process.env set here is
|
||||
// inherited by the server workers instead.
|
||||
//
|
||||
// Existing process env always wins (e.g. Coolify or `docker --env-file`), and
|
||||
// unknown keys (e.g. SERVER_TLS) are loaded too so the runtime entrypoint and
|
||||
// the rest of the app can read them.
|
||||
const envPath = path.join(workspaceRoot, ".env");
|
||||
if (existsSync(envPath)) {
|
||||
for (const raw of readFileSync(envPath, "utf8").split("\n")) {
|
||||
const line = raw.trim();
|
||||
if (!line || line.startsWith("#")) continue;
|
||||
const eq = line.indexOf("=");
|
||||
if (eq === -1) continue;
|
||||
const key = line.slice(0, eq).trim();
|
||||
let value = line.slice(eq + 1).trim();
|
||||
if (
|
||||
(value.startsWith('"') && value.endsWith('"')) ||
|
||||
(value.startsWith("'") && value.endsWith("'"))
|
||||
) {
|
||||
value = value.slice(1, -1);
|
||||
}
|
||||
if (key && !(key in process.env)) {
|
||||
process.env[key] = value;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** @type {import('next').NextConfig} */
|
||||
const nextConfig = {
|
||||
reactCompiler: true,
|
||||
// Keep the pure-JS Postgres driver external so Turbopack doesn't re-bundle it.
|
||||
serverExternalPackages: ["pg"],
|
||||
compiler: {
|
||||
removeConsole: process.env.NODE_ENV === "production",
|
||||
},
|
||||
// Allow LAN / custom host access (e.g. http://h.lizhao.net:3000) in `next dev`.
|
||||
allowedDevOrigins: ["h.lizhao.net", "tradeac-dev.h.lizhao.net"],
|
||||
experimental: {
|
||||
serverActions: {
|
||||
allowedOrigins: ["h.lizhao.net", "tradeac-dev.h.lizhao.net", "localhost", "127.0.0.1"],
|
||||
},
|
||||
},
|
||||
turbopack: {
|
||||
root: workspaceRoot,
|
||||
},
|
||||
};
|
||||
|
||||
export default nextConfig;
|
||||
@@ -0,0 +1,97 @@
|
||||
{
|
||||
"name": "studio-admin",
|
||||
"version": "2.2.0",
|
||||
"private": true,
|
||||
"scripts": {
|
||||
"build:engine": "cargo build --release --locked --manifest-path ../tac-engine/Cargo.toml --target-dir ../tac-engine/target",
|
||||
"build:engine:dev": "cargo build --release --locked --manifest-path ../tac-engine/Cargo.toml --target-dir ../tac-engine/target",
|
||||
"engine:ensure": "node scripts/ensure-engine.mjs",
|
||||
"dev": "pnpm engine:ensure && next dev --experimental-https",
|
||||
"build": "pnpm build:engine && next build",
|
||||
"start": "next start",
|
||||
"lint": "biome lint",
|
||||
"format": "biome format --write",
|
||||
"check": "biome check",
|
||||
"check:fix": "biome check --write",
|
||||
"prepare": "husky",
|
||||
"generate:presets": "ts-node -P tsconfig.scripts.json src/scripts/generate-theme-presets.ts"
|
||||
},
|
||||
"lint-staged": {
|
||||
"*.{js,ts,jsx,tsx}": [
|
||||
"biome check --write --no-errors-on-unmatched"
|
||||
]
|
||||
},
|
||||
"dependencies": {
|
||||
"@assistant-ui/react": "^0.15.4",
|
||||
"@assistant-ui/react-markdown": "^0.14.8",
|
||||
"@assistant-ui/react-opencode": "^0.2.17",
|
||||
"@base-ui/react": "^1.6.0",
|
||||
"@dnd-kit/core": "^6.3.1",
|
||||
"@dnd-kit/modifiers": "^9.0.0",
|
||||
"@dnd-kit/sortable": "^10.0.0",
|
||||
"@fullcalendar/react": "^7.0.2",
|
||||
"@gitgraph/react": "^1.6.0",
|
||||
"@hookform/resolvers": "^5.7.1",
|
||||
"@opencode-ai/sdk": "^1.18.14",
|
||||
"@shadcn/react": "^0.1.0",
|
||||
"@tanstack/react-table": "^8.21.3",
|
||||
"@vercel/analytics": "^2.0.1",
|
||||
"@xyflow/react": "^12.11.3",
|
||||
"better-auth": "^1.6.25",
|
||||
"class-variance-authority": "^0.7.1",
|
||||
"clsx": "^2.1.1",
|
||||
"cmdk": "^1.1.1",
|
||||
"d3-geo": "^3.1.1",
|
||||
"date-fns": "^4.4.0",
|
||||
"drizzle-orm": "^0.45.2",
|
||||
"echarts": "^6.1.0",
|
||||
"echarts-for-react": "^3.0.6",
|
||||
"embla-carousel-react": "^8.6.0",
|
||||
"geist": "^1.7.2",
|
||||
"input-otp": "^1.4.2",
|
||||
"kysely": "^0.29.4",
|
||||
"lucide-react": "^1.28.0",
|
||||
"next": "^16.2.12",
|
||||
"next-themes": "^0.4.6",
|
||||
"pg": "^8.22.0",
|
||||
"radix-ui": "^1.6.7",
|
||||
"react": "^19.2.8",
|
||||
"react-day-picker": "^10.0.1",
|
||||
"react-dom": "^19.2.8",
|
||||
"react-hook-form": "^7.84.0",
|
||||
"react-markdown": "^10.1.0",
|
||||
"react-resizable-panels": "^4.12.2",
|
||||
"recharts": "^3.8.0",
|
||||
"remark-gfm": "^4.0.1",
|
||||
"shadcn": "^4.16.1",
|
||||
"simple-icons": "^16.28.0",
|
||||
"sonner": "^2.0.7",
|
||||
"tailwind-merge": "^3.6.0",
|
||||
"temporal-polyfill": "^1.0.3",
|
||||
"topojson-client": "^3.1.0",
|
||||
"tw-shimmer": "^0.4.12",
|
||||
"vaul": "^1.1.2",
|
||||
"yaml": "^2.9.0",
|
||||
"zod": "^4.4.3",
|
||||
"zustand": "^5.0.14"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@biomejs/biome": "^2.5.6",
|
||||
"@tailwindcss/postcss": "^4.3.3",
|
||||
"@types/d3-geo": "^3.1.1",
|
||||
"@types/node": "^22.20.1",
|
||||
"@types/pg": "^8.20.3",
|
||||
"@types/react": "^19.2.18",
|
||||
"@types/react-dom": "^19.2.4",
|
||||
"@types/topojson-client": "^3.1.5",
|
||||
"babel-plugin-react-compiler": "^1.0.0",
|
||||
"drizzle-kit": "^0.31.10",
|
||||
"husky": "^9.1.7",
|
||||
"lint-staged": "^16.4.0",
|
||||
"postcss": "^8.5.25",
|
||||
"tailwindcss": "^4.1.5",
|
||||
"ts-node": "^10.9.2",
|
||||
"tw-animate-css": "^1.4.0",
|
||||
"typescript": "^5.9.3"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,161 @@
|
||||
---
|
||||
name: tradeac-alpaca
|
||||
description: Guide agents to call tac-engine MCP tools for Alpaca trading and market data (news, corporate actions, screener, FX, stocks, options, realtime streams) via MCP Inspector, Cursor, or other clients.
|
||||
---
|
||||
|
||||
# tradeac-alpaca
|
||||
|
||||
Use **tac-engine** MCP tools — not raw Alpaca REST — for brokerage ops and market data. Same tool surface will back TradeAC’s Next.js UI later.
|
||||
|
||||
## MCP-first policy
|
||||
|
||||
- **Prefer the MCP tools registered in this session** (`tac-engine` server, tools listed below) over writing scripts that reimplement them. If a tool exists, call it directly — do not reinvent it with curl/bash/python (raw Alpaca REST, hand-rolled pagination, own JSON-RPC clients).
|
||||
- **NEVER script directly against the MCP server** (spawning the binary, talking stdio JSON-RPC, or driving it via bash/curl) unless the MCP tool surface genuinely can't do the job — and in that case **stop and ask the user to confirm first** before writing the script.
|
||||
- If a direct Alpaca call is needed (e.g. an endpoint with no tool), say so and let the user confirm the approach; otherwise keep everything on the MCP surface.
|
||||
|
||||
## Hosts / env
|
||||
|
||||
| Purpose | Env | Default |
|
||||
|---------|-----|---------|
|
||||
| Trading REST | `APCA_BASE_URL` | `https://paper-api.alpaca.markets` |
|
||||
| Market data REST | `APCA_DATA_BASE_URL` | `https://data.alpaca.markets` |
|
||||
| Market data WS | `APCA_STREAM_BASE_URL` | `wss://stream.data.alpaca.markets` |
|
||||
| Auth | `APCA_API_KEY_ID`, `APCA_API_SECRET_KEY` | required |
|
||||
|
||||
```bash
|
||||
cargo build --release
|
||||
./target/release/tac-engine
|
||||
```
|
||||
|
||||
## Secrets policy
|
||||
|
||||
- NEVER write secrets into files: API keys (`APCA_API_KEY_ID`/`APCA_API_SECRET_KEY`), DB passwords, OAuth tokens, or credential-bearing URLs in scripts, configs, notes or committed code.
|
||||
- NEVER read `*.env` / `.env.*` directly (`cat`/`tail`/`grep`/`sed`/`head` on `.env`). That pulls secrets into this session and leaks them to any agent sharing it.
|
||||
- When a tool or command needs an env var, ASK the user to set it in the environment (shell/container env, or the user-owned `.env`) and reference it by name (`$VAR`), never by value. If it's missing, report which variable is required instead of reading it yourself.
|
||||
- If you find a committed secret, flag it, remove it, and replace it with a placeholder.
|
||||
|
||||
## MCP clients
|
||||
|
||||
### MCP Inspector
|
||||
|
||||
```bash
|
||||
npx @modelcontextprotocol/inspector /absolute/path/to/tac-engine/target/release/tac-engine
|
||||
```
|
||||
|
||||
Connect → `tools/list` → `tools/call` with JSON `arguments`.
|
||||
|
||||
### Cursor
|
||||
|
||||
```json
|
||||
{
|
||||
"mcpServers": {
|
||||
"tac-engine": {
|
||||
"command": "/absolute/path/to/tac-engine/target/release/tac-engine",
|
||||
"env": {
|
||||
"APCA_API_KEY_ID": "${APCA_API_KEY_ID}",
|
||||
"APCA_API_SECRET_KEY": "${APCA_API_SECRET_KEY}"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
stdio is NDJSON JSON-RPC; logs on stderr.
|
||||
|
||||
## Safety
|
||||
|
||||
- Paper trading by default (`APCA_BASE_URL`).
|
||||
- Mutating trading tools: `place_order`, `close_*`, `cancel_*`, watchlist writes.
|
||||
- `subscribe_market_stream` opens a **short-lived** WebSocket (samples then disconnects). Most Alpaca plans allow **one** concurrent stream — close other clients first.
|
||||
|
||||
## Tool catalog
|
||||
|
||||
Lake tools (`get_lake_bars`, `get_lake_ta`, `get_lake_sp`, `get_lake_features`, `backfill_lake_calendar`, …) live under **`tradeac-lake`** (see `tac-engine/skills/tradeac-lake/SKILL.md`). When a lake call's purpose is **backfilling/persisting** (not reading the payload), pass `"quiet": true` so the tool returns a summary (`count`/`first_t`/`last_t`/`source`/`columns`) instead of echoing back the full bar/feature rows.
|
||||
|
||||
### Trading (brokerage account)
|
||||
|
||||
Account: `get_account`, `get_portfolio_history`, `list_account_activities`, `get_account_activities_by_type`
|
||||
Assets master: `list_assets`, `get_asset`
|
||||
Watchlists / positions / orders: `list_*`, `get_*`, `create_*`, `place_order`, `close_*`, `cancel_*`
|
||||
|
||||
### Market data — news & corporate actions
|
||||
|
||||
| Tool | Notes |
|
||||
|------|------|
|
||||
| `get_news` | optional `symbols`, `start`/`end`, `limit`, `include_content` |
|
||||
| `get_corporate_actions` | optional `symbols`, `types`, `start`/`end`, `data_quality` |
|
||||
|
||||
### Screener
|
||||
|
||||
| Tool | Notes |
|
||||
|------|------|
|
||||
| `get_most_actives` | optional `by`=`volume`\|`trades`, `top` |
|
||||
| `get_market_movers` | required `market_type`=`stocks`\|`crypto`, optional `top` |
|
||||
|
||||
### FX
|
||||
|
||||
| Tool | Notes |
|
||||
|------|------|
|
||||
| `get_forex_latest_rates` | required `currency_pairs` e.g. `USDJPY,EURUSD` |
|
||||
| `get_forex_rates` | historical; optional `timeframe`, `start`, `end` |
|
||||
|
||||
### Stocks
|
||||
|
||||
| Tool | Notes |
|
||||
|------|------|
|
||||
| `get_stock_bars` / `get_stock_bars_single` | historical; needs `timeframe` |
|
||||
| `get_stock_latest_bars` | latest minute bars |
|
||||
| `get_stock_quotes` / `get_stock_latest_quotes` | quotes |
|
||||
| `get_stock_trades` / `get_stock_latest_trades` | trades |
|
||||
| `get_stock_snapshots` / `get_stock_snapshot` | trade+quote+bars |
|
||||
| `get_stock_auctions` | auctions |
|
||||
|
||||
Multi-symbol tools take comma-separated `symbols`. Optional `feed` (`iex`/`sip`), `limit`, `page_token`, …
|
||||
|
||||
### Options
|
||||
|
||||
| Tool | Notes |
|
||||
|------|------|
|
||||
| `get_option_bars` | historical bars for contract symbols |
|
||||
| `get_option_latest_quotes` / `get_option_latest_trades` | latest |
|
||||
| `get_option_trades` | historical trades |
|
||||
| `get_option_snapshots` | contracts |
|
||||
| `get_option_chain` | underlying + filters (`type`, strikes, expiration) |
|
||||
| `get_option_meta_conditions` / `get_option_meta_exchanges` | code maps |
|
||||
|
||||
### Realtime stream sampling
|
||||
|
||||
| Tool | Notes |
|
||||
|------|------|
|
||||
| `subscribe_market_stream` | `stream`=`stocks`\|`options`\|`news`\|`test`; optional `feed`; channels `trades`/`quotes`/`bars`/`news` as CSV symbols; `duration_secs` (1–30), `max_messages` (1–200) |
|
||||
|
||||
Examples:
|
||||
|
||||
```json
|
||||
{"stream":"test","duration_secs":5,"max_messages":20}
|
||||
```
|
||||
|
||||
```json
|
||||
{"stream":"stocks","feed":"iex","quotes":"AAPL,MSFT","duration_secs":5}
|
||||
```
|
||||
|
||||
```json
|
||||
{"stream":"news","news":"*","duration_secs":8,"max_messages":30}
|
||||
```
|
||||
|
||||
```json
|
||||
{"stream":"options","feed":"indicative","quotes":"AAPL250117C00200000","duration_secs":5}
|
||||
```
|
||||
|
||||
## Example workflows
|
||||
|
||||
1. **Dashboard:** `get_account` → `list_positions` → `get_stock_snapshots` (`symbols` from positions)
|
||||
2. **Research:** `get_news` → `get_corporate_actions` → `get_stock_bars`
|
||||
3. **Screener → trade (paper):** `get_most_actives` → `get_stock_snapshot` → `place_order`
|
||||
4. **Options:** `get_option_chain` (`underlying_symbol=AAPL`) → `get_option_latest_quotes`
|
||||
5. **Live sample:** `subscribe_market_stream` with `stream=test` first, then stocks/news
|
||||
|
||||
## Protocol
|
||||
|
||||
- rmcp **3.1** / MCP **2026-07-28**, stdio
|
||||
- Prefer these MCP tool names/args over calling Alpaca hosts directly from agents/UI
|
||||
@@ -0,0 +1,417 @@
|
||||
---
|
||||
name: tradeac-lake
|
||||
description: Guide agents to build and query the TradeAC parquet+DuckDB data lake on the local filesystem — hive-partitioned bar store (market/timeframe/symbol) plus symbols, watchlist, calendar, features and coverage metadata — with lazy backfill from the tac-engine MCP get_stock_bars tool (tradeac-alpaca skill).
|
||||
---
|
||||
|
||||
# tradeac-lake
|
||||
|
||||
Local-first market data lake: **Apache Parquet** files on disk, consumed with **DuckDB** (or Apache Arrow). Bar data is the core payload; the lake also keeps small metadata parquet files (symbols, watchlist, calendar, features, coverage) at the lake root.
|
||||
|
||||
Reading is a **cache-first** pattern: if the requested range is already in the lake, serve it directly from parquet; otherwise **lazy-load** the missing window via the tac-engine MCP `get_stock_bars` tool (see `tac-engine/skills/tradeac-alpaca/SKILL.md`), persist it, update metadata, then return.
|
||||
|
||||
## MCP-first policy
|
||||
|
||||
- **Prefer the tac-engine lake MCP tools** (`get_lake_bars`, `get_lake_ta`, `get_lake_sp`, `get_lake_features`, `get_lake_status`, `get_lake_coverage`, `get_lake_calendar`, `backfill_lake_calendar`, …) whenever they cover the need. They handle coverage checks, lazy backfill, feed fallback, metadata updates and pagination for you — do not reimplement that in DuckDB/pyarrow scripts.
|
||||
- **Direct parquet reads are only for verification** (DuckDB CLI / pyarrow snippets below) or when no lake tool covers the query (e.g. an arbitrary ad-hoc SQL join). Keep hand-rolled lake *writes* off the happy path — the write path is what the MCP tools automate.
|
||||
- **NEVER script directly against the MCP server** (spawning the engine binary, stdio JSON-RPC, bash/curl) unless a tool genuinely can't do the job — then **stop and ask the user to confirm first**.
|
||||
- The engine bundles its own DuckDB; direct verification only needs the `duckdb` CLI or a Python venv with `duckdb` + `pyarrow` (see dependencies below).
|
||||
|
||||
## Env / root
|
||||
|
||||
| Var | Default | Purpose |
|
||||
|-----|---------|---------|
|
||||
| `TAC_LAKE_DIR` | **required** (no default) | lake root on the local filesystem. Local dev: absolute path (e.g. `/home/data/lake`). |
|
||||
|
||||
```bash
|
||||
export TAC_LAKE_DIR=/path/to/lake
|
||||
mkdir -p "$TAC_LAKE_DIR"
|
||||
```
|
||||
|
||||
## Secrets policy
|
||||
|
||||
- NEVER write secrets into files: API keys, DB passwords, OAuth tokens, or credential-bearing URLs (`DATABASE_URL`, `APCA_*`) in scripts, configs, notes or committed code.
|
||||
- NEVER read `*.env` / `.env.*` directly (`cat`/`tail`/`grep`/`sed`/`head` on `.env`). That pulls secrets into this session and leaks them to any agent sharing it.
|
||||
- When a tool or command needs an env var, ASK the user to set it in the environment (shell/container env, or the user-owned `.env`) and reference it by name (`$VAR`), never by value. If it's missing, report which variable is required instead of reading it yourself.
|
||||
- If you find a committed secret, flag it, remove it, and replace it with a placeholder.
|
||||
|
||||
## Lake layout
|
||||
|
||||
Hive partition convention, partitioned by `market`, `timeframe`, `symbol`. Metadata parquet files live alongside the partition dirs at the lake root.
|
||||
|
||||
```
|
||||
$TAC_LAKE_DIR/
|
||||
├── market=US/
|
||||
│ └── timeframe=1d/
|
||||
│ ├── symbol=AAPL.parquet
|
||||
│ ├── symbol=MSFT.parquet
|
||||
│ └── ...
|
||||
│ └── timeframe=10m/
|
||||
│ └── symbol=AAPL.parquet
|
||||
├── market=CRYPTO/... # optional: BTC/USD etc.
|
||||
├── features/ # TA + SP indicators, hive-partitioned with family tier
|
||||
│ └── market=US/
|
||||
│ └── timeframe=1d/
|
||||
│ ├── family=ta/
|
||||
│ │ └── symbol=AAPL.parquet # TA indicators (sma, rsi, macd, ...)
|
||||
│ └── family=sp/
|
||||
│ └── symbol=AAPL.parquet # Stochastic-process features (ou, hmm, har, ...)
|
||||
├── symbols.parquet # asset master seen/known to the lake
|
||||
├── watchlist.parquet # watchlists
|
||||
├── calendar.parquet # trading days per market (coverage ground truth)
|
||||
├── coverage.parquet # per (market,timeframe,symbol) loaded window
|
||||
└── manifest.yaml # lake config: feed, adjustment, timezone
|
||||
```
|
||||
|
||||
Convention: **one parquet file per symbol per timeframe per family** under the partition dirs. Bar upserts **merge by canonical timestamp** (read existing file → overlay new bars → write the full set atomically); feature persistence is a **fresh write** per family (Appender-based, no merge with existing).
|
||||
|
||||
## Lake MCP tools (tac-engine)
|
||||
|
||||
The tac-engine MCP server exposes a **lake tools** category that wraps the read/write paths below. `symbols`/`market` are uppercased, `timeframe` is normalized to lake spelling, and `start`/`end` accept `YYYY-MM-DD` or RFC-3339 (default `end=now`, `start=end-30d`).
|
||||
|
||||
| Tool | Purpose |
|
||||
|------|---------|
|
||||
| `get_lake_bars` | Cache-first bars: `{market?, symbols, timeframe, start?, end?, feed?, adjustment?, lazy?, quiet?}`. Feed defaults to `iex`; SIP is never used (requires license). For daily bars, Yahoo Finance fills gaps before Alpaca's earliest available date. `lazy=true` (default) backfills missing windows via Alpaca and persists; `lazy=false` reads the lake only. Returns `{request, source, bars: {SYM: [{t,o,h,l,c,v,n,vw}]}}`; `source` is `lake` (complete hit), `partial` (present but missing windows and `lazy=false`), or `fetched` (gaps backfilled). With `quiet=true` returns `{request, source, summary: {SYM: {count, first_t, last_t}}}` instead of the bar rows — use for backfill-to-lake jobs. Bar writes **merge by timestamp** (read existing file, overlay new bars, write the full set atomically) — safe for both tail appends and leading-gap backfills. Coverage/symbols/calendar metadata are reconciled against the actual file on each write. |
|
||||
| `get_lake_ta` | Compute + optionally persist indicators: `{market?, symbol, timeframe, start?, end?, indicators?, persist?, quiet?}`. `indicators` is comma-separated, default all: `sma_5,sma_20,ema_12,ema_26,rsi_14,macd,bb,atr_14,adx_14`. Lookback is pulled automatically; returned rows cover `[start,end]`. When `persist=true`, features are written to `features/market=*/timeframe=*/family=ta/symbol=*.parquet`. With `quiet=true` returns `{count, columns, persisted}` instead of the feature rows — use when the goal is persisting indicators. |
|
||||
| `get_lake_sp` | Compute + optionally persist **stochastic-process features** (`sp_*` columns) from lake bars: `{market?, symbol, timeframe, start?, end?, fit_end?, families?, persist?, quiet?}`. Rust port of `sp_features.py` on the stochastic-rs stack. `families` is comma-separated, default all: `ou,hmm,jump,har,trend,hurst,signature,moments`. In addition to `sp_rv*`/`sp_vol_ratio_*` the `har` family also emits `sp_rv_ac1` (RV lag-1 autocorr) + `sp_rv_cv_22` (RV coefficient of variation); `jump` also emits `sp_max_up`/`sp_max_down` (signed max-move asymmetry); `signature` also emits the lag-5 level-2 cross terms `sp_sig_level2_{lead_lag,lag_lead}_5`; and `moments` emits the scale-free realized skewness/kurtosis (`sp_rskew_{5,22}`, `sp_rkurt_{5,22}`; a 1-day window is undefined) and downside semi-variance (`sp_dsv_{1,5,22}`, `sp_dsv_ratio_{1,5,22}`) via stochastic-rs `realized`. `fit_end` limits the 2-state Gaussian-HMM fit window (no lookahead; posteriors still cover the whole window). On `persist=true`, features are written to `features/market=*/timeframe=*/family=sp/symbol=*.parquet`. With `quiet=true` returns `{count, sp_columns, persisted}` instead of the feature rows. Deferred (not in stochastic-rs): `garch`, `entropy`, `catch22`. |
|
||||
| `get_lake_features` | Read persisted TA + SP features (hive-partitioned `features/` dir, families `ta` and `sp` merged by timestamp): `{market?, symbol?, timeframe?, start?, end?, quiet?}`. With `quiet=true` returns `{count, columns}` instead of the full feature rows. |
|
||||
| `get_lake_symbols` | Read `symbols.parquet` asset master; optional `{symbol?}` filter. |
|
||||
| `get_lake_watchlist` | Read `watchlist.parquet`. |
|
||||
| `get_lake_calendar` | Read `calendar.parquet` trading days: `{market?, start?, end?}`. |
|
||||
| `get_lake_coverage` | Read `coverage.parquet` cache index: `{market?, timeframe?, symbol?}`. |
|
||||
| `get_lake_status` | Lake root, `manifest.yaml`, bar partition inventory and metadata file sizes. |
|
||||
| `rebuild_lake_symbol` | Delete bar + feature parquet files and re-fetch from `TAC_LAKE_START_DATE` (default `2000-01-03`) for a single symbol: `{market?, symbol, timeframe, feed?, adjustment?}`. Resets coverage so the next `get_lake_bars` call re-downloads the full history. Use after changing `TAC_LAKE_START_DATE` or to fix stale/corrupt data. |
|
||||
| `load_lake_symbols` | **Bulk-load + persist** bars + TA + SP features for a comma-separated list of symbols. Runs in a background thread and returns immediately with a `job_id`: `{market?, symbols, timeframe, start?, end?, feed?, adjustment?, indicators?, families?}`. Per-symbol start is computed automatically from lake coverage: if the lake has no data or `first_t > TAC_LAKE_START_DATE`, fetches from `TAC_LAKE_START_DATE` (default 2000-01-03); if `first_t <= TAC_LAKE_START_DATE`, fetches only from `last_t` (tail refresh). TA/SP features are always computed over the full `TAC_LAKE_START_DATE` to `end` range. Poll `load_lake_status` with the returned `job_id` to track progress. |
|
||||
| `load_lake_status` | Query the status of a background bulk-load job: `{job_id}`. Returns `{job_id, status, total_symbols, processed, results, error, started_at, completed_at}` where `status` is `running`, `completed`, or `failed`, and `results` contains per-symbol bar counts, TA/SP column counts, and any errors. |
|
||||
| `backfill_lake_calendar` | **Gap-fill tool**: seed/enrich `calendar.parquet` from Alpaca historical auctions (feed=iex; records exist only on trading days): `{market?, symbols, start?, end?}`. Returns `{market, symbols, start, end, calendar_days_added}`. Call this before lazy bar loads so the `1d` completeness check knows the expected trading-day set. |
|
||||
| `validate_lake_dataset` | **Pre-workflow quality gate**: `{market?, timeframe?, symbols?, start?, end?}`. Scans every symbol in coverage (or a comma-separated `symbols` subset) and reports `verdict: OK/WARNINGS/ERRORS` plus per-symbol issues. Catches the failure modes qlib silently tolerates: **all-NaN feature columns** (would be dropped by `DropAllNaN` — the model trains on fewer features without notice), **missing TA/SP feature files**, **hollow coverage / stale date ranges** (coverage claims a wide span but the bar file is empty/truncated/sparse), **stale coverage** (first/last/num_bars vs the actual file), and **partition misalignment** (flat-layout feature orphans the family=ta|sp consumers can't see). Pass `start`/`end` to also check feature-vs-bar row alignment and per-column all-NaN status in that window. Call before `rd_run_workflow` / `rd_train` to fail fast instead of training on silent data holes. |
|
||||
|
||||
Example:
|
||||
```json
|
||||
{"symbols": "AAPL,MSFT", "timeframe": "1d", "start": "2026-06-06", "lazy": true}
|
||||
```
|
||||
→ `{"request": {...}, "source": {"AAPL": "fetched", "MSFT": "lake"}, "bars": {"AAPL": [{...}], "MSFT": [{...}]}}`
|
||||
|
||||
## Quiet mode
|
||||
|
||||
`get_lake_bars`, `get_lake_ta`, `get_lake_sp` and `get_lake_features` accept `"quiet": true`. When the point of the call is **writing to the lake** (backfill/fetch bars, compute + persist indicators or `sp_*` features), use `quiet: true` — the tool still performs the full backfill / computation / persist, but returns a **summary** instead of echoing back the potentially huge payload (thousands of bar rows / feature rows). Full-row output (`bars` / `features`) is the default, so requests that *need* the data to read it must leave `quiet` unset/false.
|
||||
|
||||
| Tool | `quiet: true` response |
|
||||
|------|------------------------|
|
||||
| `get_lake_bars` | `{request, source: {SYM: lake\|partial\|fetched}, summary: {SYM: {count, first_t, last_t}}}` |
|
||||
| `get_lake_ta` | `{market, symbol, timeframe, start, end, count, columns, persisted}` |
|
||||
| `get_lake_sp` | `{market, symbol, timeframe, start, end, fit_end, count, sp_columns, persisted}` |
|
||||
| `get_lake_features` | `{count, columns}` |
|
||||
|
||||
Backfill-to-lake job (no payload echoed):
|
||||
```json
|
||||
{"symbols": "AAPL,MSFT", "timeframe": "1d", "start": "2026-06-06", "lazy": true, "quiet": true}
|
||||
```
|
||||
→ `{"request": {...}, "source": {"AAPL": "fetched", "MSFT": "lake"}, "summary": {"AAPL": {"count": 44, "first_t": "2026-06-06T04:00:00Z", "last_t": "2026-08-05T04:00:00Z"}, "MSFT": {...}}}`
|
||||
|
||||
Persist indicators to the lake (summary only):
|
||||
```json
|
||||
{"symbol": "AAPL", "timeframe": "1d", "indicators": "sma_5,sma_20,rsi_14", "persist": true, "quiet": true}
|
||||
```
|
||||
→ `{"market": "US", "symbol": "AAPL", "timeframe": "1d", "count": 44, "columns": ["sma_5","sma_20","rsi_14"], "persisted": true}`
|
||||
|
||||
```json
|
||||
{"symbols": "AAPL,MSFT", "start": "2026-06-06"}
|
||||
```
|
||||
→ `{"market": "US", "symbols": ["AAPL","MSFT"], "start": ..., "end": ..., "calendar_days_added": 44}`
|
||||
|
||||
## Conventions
|
||||
|
||||
- `market`: `US` (equities), `CRYPTO`, `FOREX`. Uppercase.
|
||||
- `timeframe`: normalized lake name — lowercase, `1m 5m 10m 15m 30m 1h 2h 4h 1d 1w 1M`. The MCP tool spells them differently; always map:
|
||||
| Lake | MCP `timeframe` | Lake | MCP `timeframe` |
|
||||
|------|-----------------|------|-----------------|
|
||||
| `1m` | `1Min` | `2h` | `2Hour` |
|
||||
| `5m` | `5Min` | `4h` | `4Hour` |
|
||||
| `10m` | `10Min` | `1d` | `1Day` |
|
||||
| `15m` | `15Min` | `1w` | `1Week` |
|
||||
| `30m` | `30Min` | `1M` | `1Month` |
|
||||
| `1h` | `1Hour` | | |
|
||||
- `symbol`: uppercase, e.g. `AAPL`. Hyphens/`.` in special symbols (e.g. `BRK-B`, `SPY`) are valid filenames; avoid `/` and spaces.
|
||||
- All timestamps stored as **UTC** instants (`TIMESTAMPTZ`). Alpaca returns RFC-3339 UTC; normalize on write.
|
||||
- `1d` bars: `t` is the session date at `04:00Z` (midnight ET — Alpaca stamps daily bars at `04:00:00Z`); also store a `date` column (`CAST(t AS DATE)`, UTC) for calendar joins. A date-only `end` (e.g. `2026-08-05`) is treated as **inclusive of the whole end day**, so the end-day bar is not dropped.
|
||||
|
||||
## Bar parquet schema (`market=…/timeframe=…/symbol=….parquet`)
|
||||
|
||||
| col | type | source field |
|
||||
|-----|------|--------------|
|
||||
| `t` | TIMESTAMPTZ | bar `t` (UTC) |
|
||||
| `o` | DOUBLE | `o` |
|
||||
| `h` | DOUBLE | `h` |
|
||||
| `l` | DOUBLE | `l` |
|
||||
| `c` | DOUBLE | `c` |
|
||||
| `v` | BIGINT | `v` |
|
||||
| `n` | BIGINT | `n` |
|
||||
| `vw` | DOUBLE | `vw` |
|
||||
|
||||
Partition columns `market`/`timeframe`/`symbol` are derived from the path; DuckDB exposes them automatically when reading a hive glob.
|
||||
|
||||
## Metadata parquet files
|
||||
|
||||
All written with DuckDB `COPY … (FORMAT PARQUET)` from in-memory `SELECT`, or `pyarrow.parquet`.
|
||||
|
||||
`symbols.parquet`
|
||||
| col | type | notes |
|
||||
|-----|------|-------|
|
||||
| `symbol` | VARCHAR (pk) |
|
||||
| `name` | VARCHAR |
|
||||
| `asset_class` | VARCHAR |
|
||||
| `exchange` | VARCHAR |
|
||||
| `tradable` | BOOLEAN |
|
||||
| `status` | VARCHAR |
|
||||
| `first_seen` | TIMESTAMPTZ | the symbol's earliest bar in the lake (its first trading date), not the load timestamp |
|
||||
| `updated_at` | TIMESTAMPTZ | |
|
||||
|
||||
`watchlist.parquet`
|
||||
| col | type |
|
||||
|-----|------|
|
||||
| `watchlist_id` | VARCHAR |
|
||||
| `name` | VARCHAR |
|
||||
| `symbol` | VARCHAR |
|
||||
| `added_at` | TIMESTAMPTZ |
|
||||
| `updated_at` | TIMESTAMPTZ |
|
||||
|
||||
`calendar.parquet` — the trading-day ground truth per market (see “Calendar gap” below). Bars only seed which dates are trading days; per-symbol prices/session times are NOT attributed by the bars path (no symbol column, 1d bars all share `t=04:00`).
|
||||
| col | type | notes |
|
||||
|-----|------|-------|
|
||||
| `market` | VARCHAR | pk + `date` |
|
||||
| `date` | DATE | a trading day (UTC) |
|
||||
| `session_open` | TIMESTAMPTZ | from auctions `o[0].t` only (nullable; not set by bars) |
|
||||
| `session_close` | TIMESTAMPTZ | from auctions `c[0].t` only (nullable; not set by bars) |
|
||||
| `open_price` | DOUBLE | opening auction price (nullable) |
|
||||
| `close_price` | DOUBLE | closing auction price (nullable) |
|
||||
| `source` | VARCHAR | `auctions` \| `bars` \| `manual` |
|
||||
| `updated_at` | TIMESTAMPTZ | |
|
||||
|
||||
`features/` — TA + stochastic-process indicators, **wide** format, hive-partitioned with a `family` tier: `features/market=US/timeframe=1d/family=ta/symbol=AAPL.parquet` and `family=sp/symbol=AAPL.parquet`. Each row is one `t`, with one column per indicator. The partition columns (market/symbol/timeframe) come from the directory structure; the file itself stores `t` + indicator columns (e.g. `sma_5`, `sma_20`, `ema_12`, `ema_26`, `rsi_14` for `family=ta`; `sp_ou_halflife`, `sp_hmm_regime`, `sp_har_rv_5` for `family=sp`), all DOUBLE. Writes are Appender-based fresh writes per family (no read-merge-write cycle).
|
||||
| col | type |
|
||||
|-----|------|
|
||||
| `t` | TIMESTAMPTZ |
|
||||
| `sma_5`, `sma_20`, `ema_12`, `ema_26` | DOUBLE |
|
||||
| `rsi_14` | DOUBLE |
|
||||
| `macd`, `macd_signal`, `macd_hist` | DOUBLE |
|
||||
| `bb_upper`, `bb_middle`, `bb_lower` | DOUBLE |
|
||||
| `atr_14`, `adx_14`, `stoch_k`, `stoch_d` | DOUBLE |
|
||||
| `_feature_<name>` | DOUBLE |
|
||||
|
||||
`coverage.parquet` — **the cache index**: the exact loaded window per bar set. This is what makes direct hits fast.
|
||||
| col | type |
|
||||
|-----|------|
|
||||
| `market` | VARCHAR |
|
||||
| `timeframe` | VARCHAR |
|
||||
| `symbol` | VARCHAR |
|
||||
| `first_t` | TIMESTAMPTZ |
|
||||
| `last_t` | TIMESTAMPTZ |
|
||||
| `num_bars` | BIGINT |
|
||||
| `feed` | VARCHAR |
|
||||
| `adjustment` | VARCHAR |
|
||||
| `loaded_at` | TIMESTAMPTZ |
|
||||
| `updated_at` | TIMESTAMPTZ |
|
||||
|
||||
`manifest.yaml` (plain text, not parquet) — lake config so reads/writes stay consistent:
|
||||
```yaml
|
||||
lake_version: 1
|
||||
default_market: US
|
||||
default_feed: iex # iex is the default; SIP is never used (requires license)
|
||||
default_adjustment: raw # raw|split|dividend|all — pick once per lake
|
||||
timezone: UTC
|
||||
features_lib: ta-lib
|
||||
```
|
||||
|
||||
## Read path (cache-first)
|
||||
|
||||
### DuckDB
|
||||
|
||||
```bash
|
||||
duckdb :memory:
|
||||
```
|
||||
|
||||
```sql
|
||||
-- hive glob adds market/timeframe/symbol columns automatically
|
||||
SELECT * FROM read_parquet('$TAC_LAKE_DIR/market=*/timeframe=*/symbol=*.parquet');
|
||||
```
|
||||
|
||||
Canonical queries:
|
||||
```sql
|
||||
-- past 2 months, 1d bars
|
||||
SELECT symbol, date, o, h, l, c, v, n, vw
|
||||
FROM read_parquet('$TAC_LAKE_DIR/market=US/timeframe=1d/symbol=*.parquet')
|
||||
WHERE symbol = 'AAPL'
|
||||
AND t >= now() - INTERVAL 2 MONTH
|
||||
ORDER BY t;
|
||||
|
||||
-- past 2 days, 10m bars
|
||||
SELECT * FROM read_parquet('$TAC_LAKE_DIR/market=US/timeframe=10m/symbol=*.parquet')
|
||||
WHERE symbol = 'AAPL' AND t >= now() - INTERVAL 2 DAY ORDER BY t;
|
||||
|
||||
-- past 2 hours, 1m bars
|
||||
SELECT * FROM read_parquet('$TAC_LAKE_DIR/market=US/timeframe=1m/symbol=*.parquet')
|
||||
WHERE symbol = 'AAPL' AND t >= now() - INTERVAL 2 HOUR ORDER BY t;
|
||||
```
|
||||
|
||||
Join with features (hive-partitioned, family=ta):
|
||||
```sql
|
||||
SELECT b.t, b.c, f.sma_20, f.rsi_14
|
||||
FROM read_parquet('$TAC_LAKE_DIR/market=US/timeframe=1d/symbol=AAPL.parquet') b
|
||||
LEFT JOIN read_parquet('$TAC_LAKE_DIR/features/market=US/timeframe=1d/family=ta/symbol=AAPL.parquet') f
|
||||
ON f.t=b.t
|
||||
WHERE b.t >= now() - INTERVAL 2 MONTH;
|
||||
```
|
||||
|
||||
Join with SP features (family=sp):
|
||||
```sql
|
||||
SELECT b.t, b.c, sp.sp_ou_halflife, sp.sp_hmm_regime
|
||||
FROM read_parquet('$TAC_LAKE_DIR/market=US/timeframe=1d/symbol=AAPL.parquet') b
|
||||
LEFT JOIN read_parquet('$TAC_LAKE_DIR/features/market=US/timeframe=1d/family=sp/symbol=AAPL.parquet') sp
|
||||
ON sp.t=b.t
|
||||
WHERE b.t >= now() - INTERVAL 2 MONTH;
|
||||
```
|
||||
|
||||
### Apache Arrow / Python
|
||||
|
||||
```python
|
||||
import pyarrow.parquet as pq
|
||||
t = pq.read_table(
|
||||
"$TAC_LAKE_DIR/market=US/timeframe=1d/symbol=*.parquet",
|
||||
filters=[("symbol", "==", "AAPL")],
|
||||
)
|
||||
df = t.to_pandas()
|
||||
```
|
||||
|
||||
## Verify lake data (duckdb CLI)
|
||||
|
||||
Any parquet file in the lake can be inspected directly with the **DuckDB CLI** — no MCP call needed. Handy for confirming a `get_lake_bars`/`backfill_lake_calendar` write landed:
|
||||
|
||||
### Dependencies (duckdb + apache arrow)
|
||||
|
||||
`duckdb` and `pyarrow` are declared in `tac-qlib/pyproject.toml` (installed into the repo `.venv` by `uv`). If the runtime venv lacks them, **lazy-install** rather than falling back to another SQL tool:
|
||||
|
||||
```bash
|
||||
uv pip install --python $VIRTUAL_ENV/bin/python duckdb pyarrow # or: uv pip install -e ./tac-qlib
|
||||
```
|
||||
|
||||
Then re-check with `python -c "import duckdb, pyarrow"`. Only use the DuckDB CLI / pyarrow path when the MCP lake tools can't answer (see MCP-first policy above).
|
||||
|
||||
```bash
|
||||
duckdb :memory: "SELECT * FROM read_parquet('$TAC_LAKE_DIR/market=US/timeframe=1d/symbol=AAPL.parquet') LIMIT 10;"
|
||||
```
|
||||
|
||||
Or interactively:
|
||||
```bash
|
||||
duckdb :memory:
|
||||
SELECT * FROM read_parquet('$TAC_LAKE_DIR/market=US/timeframe=1d/symbol=AAPL.parquet') LIMIT 10;
|
||||
```
|
||||
|
||||
Quick checks:
|
||||
- **Bars written:** `SELECT count(*), min(t), max(t) FROM read_parquet('$TAC_LAKE_DIR/market=US/timeframe=1d/symbol=AAPL.parquet');`
|
||||
- **Coverage index:** `SELECT * FROM read_parquet('$TAC_LAKE_DIR/coverage.parquet') LIMIT 10;`
|
||||
- **Metadata:** `SELECT * FROM read_parquet('$TAC_LAKE_DIR/symbols.parquet') LIMIT 10;`
|
||||
- **Calendar:** `SELECT * FROM read_parquet('$TAC_LAKE_DIR/calendar.parquet') LIMIT 10;`
|
||||
|
||||
> Note: `TAC_LAKE_DIR` is **mandatory** and must be an absolute path — do **not** use `~` or `$HOME`
|
||||
> (no fallback/expansion logic exists; a literal `~` is not expanded by shells/duckdb inside an env var).
|
||||
|
||||
## Lazy-load write path
|
||||
|
||||
The core procedure when the requested range is **not** fully covered. Steps 1–9 are automated by the **`get_lake_bars`** lake tool (`lazy=true`) — the manual walk-through below documents what it does under the hood, and is the pattern to follow if writing the lake directly (DuckDB/pyarrow):
|
||||
|
||||
1. **Normalize the request.** `market`, lake `timeframe` (map back to MCP spelling), `symbols`, `start`, `end`. Decide `feed` and `adjustment` from `manifest.yaml` (or request overrides). Keep them fixed per lake — mixing feeds/adjustments corrupts history.
|
||||
2. **Check coverage** (`coverage.parquet`). See decision table below.
|
||||
3. **Compute the missing window(s).** e.g. request `[S,E]`, lake has `[S,M]` → fetch `(M,E]`; no row → fetch `[S,E]`.
|
||||
4. **Call MCP `get_stock_bars`** (multi-symbol variant; comma-separated `symbols`):
|
||||
|
||||
```json
|
||||
{"symbols":"AAPL,MSFT","timeframe":"1Day","start":"2026-06-06T00:00:00Z","end":"2026-08-06T00:00:00Z","feed":"iex","adjustment":"raw","limit":10000}
|
||||
```
|
||||
|
||||
Response: `{"bars": {"AAPL": [{t,o,h,l,c,v,n,vw}, …], …}, "next_page_token": "…"}`. Bars are sorted symbol-first, so a page may contain only some symbols — **loop with `next_page_token`** until `null`.
|
||||
5. **Parse + normalize.** Keep `t,o,h,l,c,v,n,vw`; convert `t` to UTC `TIMESTAMPTZ`; add `date` for `1d`.
|
||||
6. **Merge into the partition file** `$TAC_LAKE_DIR/market=<m>/timeframe=<tf>/symbol=<s>.parquet`: read the existing file, overlay the fetched bars keyed by canonical timestamp (new wins on duplicate `t`), write the full merged set to a tmp file, then atomically rename over the old one. This is safe for both tail appends and leading-gap backfills (a file that already holds the newest bar still accepts older fetched history).
|
||||
7. **Update `coverage.parquet`**: recompute `first_t`/`last_t`/`num_bars` from the **actual file contents** (not the fetched range) — a fetch that landed nothing must not widen the span into a hollow coverage.
|
||||
8. **Update metadata**: upsert `symbols.parquet` (`first_seen` = the symbol's earliest bar in the lake) and `calendar.parquet` (distinct `date`s observed in bars, `source='bars'`; the bars path records only trading days — no session/prices).
|
||||
9. **Return the requested range** from the lake (the read path above).
|
||||
|
||||
### Coverage decision table
|
||||
|
||||
For a request `(market, timeframe, symbol, S, E)` against `coverage.parquet`:
|
||||
|
||||
| coverage row | action |
|
||||
|--------------|--------|
|
||||
| missing | backfill whole `[S,E]` |
|
||||
| `first_t <= S` and `last_t >= E` | **direct hit** — read from lake, no fetch |
|
||||
| `first_t > S` | fetch `[S, first_t)` prefix, merge |
|
||||
| `last_t < E` | fetch `(last_t, E]` suffix, merge |
|
||||
| (with `calendar`) for `1d`: expected trading days `∈ [S,E]` == bars present | consider complete |
|
||||
|
||||
Use `calendar.parquet` for the `1d` completeness check — a weekend/holiday gap is normal, so “no bar on Saturday” must **not** trigger a refetch. Also treat the in-progress current session carefully: an intraday `end=now` should not trigger a refetch loop on the forming bar.
|
||||
|
||||
### Bulk loading multiple symbols
|
||||
|
||||
For loading bars + TA + SP features for many symbols at once, use **`load_lake_symbols`**. It runs in a background thread and returns immediately with a `job_id`:
|
||||
|
||||
```json
|
||||
{"symbols":"AAPL,MSFT,GOOGL,AMZN","timeframe":"1d","start":"2020-01-03","end":"2026-08-15"}
|
||||
```
|
||||
|
||||
Response:
|
||||
```json
|
||||
{"job_id":"load-20260815-143022","status":"started","symbols":["AAPL","MSFT","GOOGL","AMZN"],"note":"load running in background -- poll load_lake_status with this job_id to track progress"}
|
||||
```
|
||||
|
||||
Poll progress with **`load_lake_status`**:
|
||||
```json
|
||||
{"job_id":"load-20260815-143022"}
|
||||
```
|
||||
|
||||
Response (while running):
|
||||
```json
|
||||
{"job_id":"load-20260815-143022","status":"running","total_symbols":4,"processed":2,"results":[...]}
|
||||
```
|
||||
|
||||
Response (when done):
|
||||
```json
|
||||
{"job_id":"load-20260815-143022","status":"completed","total_symbols":4,"processed":4,"results":[...],"completed_at":"2026-08-15T14:35:00Z"}
|
||||
```
|
||||
|
||||
Each entry in `results` contains per-symbol `fetch_start` (the date the load started from), `bars_count`, `bars_source`, `ta_count`, `ta_columns`, `sp_count`, `sp_columns`, and any `*_error` fields.
|
||||
|
||||
### Calendar gap — why `get_stock_auctions`
|
||||
|
||||
The MCP tool surface has **no calendar endpoint**, but the coverage check needs to know which days are trading days before bars exist. The **auctions** tool fills this gap:
|
||||
|
||||
- `get_stock_auctions` only accepts `feed: "sip"` (SIP is the only valid feed for auctions).
|
||||
- Auction records exist **only on trading days** → the set of distinct dates `d` across symbols is the trading-day set.
|
||||
- Response shape: `{"auctions": {"AAPL": [{"d":"2026-06-09","o":[{t,x,p,c}…],"c":[{t,x,p,c}…]}, …]}, "next_page_token": "…"}` — `o` = opening auctions, `c` = closing auctions.
|
||||
|
||||
```json
|
||||
{"symbols":"AAPL","feed":"sip","start":"2026-06-06","end":"2026-08-06"}
|
||||
```
|
||||
|
||||
Usage: to seed/enrich `calendar.parquet` for a range **before** loading bars, call the **`backfill_lake_calendar`** lake tool (it loops `get_stock_auctions` internally across symbols and pages, inserts one row per distinct `d` with `source='auctions'`, `session_open`/`open_price` from `o[0]`, `session_close`/`close_price` from `c[0]`, and reports `calendar_days_added`). Direct call equivalent:
|
||||
|
||||
```json
|
||||
{"symbols":"AAPL","feed":"sip","start":"2026-06-06","end":"2026-08-06"}
|
||||
```
|
||||
|
||||
Cheap single-day confirmation for “was this a trading day?” and first pass of daily open/close. Intraday bars and per-symbol coverage still come from `get_stock_bars`.
|
||||
|
||||
## Operations notes
|
||||
|
||||
- **Rate limits:** Alpaca data API ~200 req/min. On `429` back off (exponential, start 1s) and retry. Batch symbols in one call, but page through `next_page_token`.
|
||||
- **Atomicity:** write parquet to a `.<name>.tmp` then `rename()`; readers never see partial files. Apply the same pattern to metadata upserts.
|
||||
- **Consistency:** one `feed` + one `adjustment` per lake (record in `manifest.yaml`). Refetching a window with a different feed/adjustment would silently corrupt merged history.
|
||||
- **Feed / history limits (Alpaca):** SIP is never used (requires license). IEX goes back to **2020-07-27** for daily bars. For earlier data, Yahoo Finance fills gaps automatically (daily bars only). The `TAC_LAKE_START_DATE` (default `2000-01-03`) controls the earliest date requested; Yahoo provides data back to ~1970 for most symbols.
|
||||
- **Dedup / merge:** bar upserts read the existing file, merge by canonical timestamp (new wins on duplicate `t`), and write the full set atomically — safe for both tail appends and leading-gap backfills (a file that already holds the newest bar still accepts older fetched history). Feature persistence is a fresh write per family, so no dedup needed.
|
||||
- **Features** are derived from the lake bars (compute after bars are persisted, keyed `(market, symbol, timeframe, t)`), so indicator history stays aligned with bar history.
|
||||
|
||||
## End-to-end example (1d, 2 months, AAPL)
|
||||
|
||||
1. `coverage.parquet` has no `(US,1d,AAPL)` row → backfill.
|
||||
2. Seed calendar: `backfill_lake_calendar` `{"symbols":"AAPL","start":…,"end":…}` (loops `get_stock_auctions`) → `calendar.parquet` trading days.
|
||||
3. `get_lake_bars` `{"symbols":"AAPL","timeframe":"1d","start":…,"end":…,"lazy":true,"quiet":true}` → auto backfills the window, persists bars, updates coverage/symbols/calendar, returns a `{count, first_t, last_t}` summary instead of the bar rows.
|
||||
4. Answer: DuckDB `SELECT * FROM read_parquet('$TAC_LAKE_DIR/market=US/timeframe=1d/symbol=AAPL.parquet') WHERE t >= now() - INTERVAL 2 MONTH` (or `get_lake_bars` again).
|
||||
5. **Next identical request is a direct hit** (`source:"lake"`) from step 1’s decision table — no Alpaca fetch.
|
||||
@@ -0,0 +1,250 @@
|
||||
# tac-qlib
|
||||
|
||||
Run stock [qlib](https://github.com/microsoft/qlib) ML workflows (LightGBM → signal → backtest) directly on the
|
||||
TradeAC parquet lake. No CSV/bin dump, no data conversion: the lake's calendar, instrument master, OHLCV bars and
|
||||
pre-computed ta-lib features plug into qlib as first-class data providers, and a custom `DataHandlerLP`
|
||||
(`TACHandler`) exposes them through the normal qlib dataset/processor pipeline.
|
||||
|
||||
```
|
||||
$ qrun workflows/workflow_lgb_taclake.yaml --experiment_name tac-lake-lgb
|
||||
```
|
||||
|
||||
trains a LightGBM, records predictions/labels, evaluates the signal (IC/RankIC), runs a daily
|
||||
`TopkDropoutStrategy` backtest with cost model and risk analysis, and logs everything to mlflow (sqlite).
|
||||
|
||||
## Layout
|
||||
|
||||
```
|
||||
tac-qlib/
|
||||
├── tac_qlib/
|
||||
│ ├── qlib_init.py # qlib_init() drop-in wired to the lake providers
|
||||
│ ├── data/
|
||||
│ │ ├── config.py # LakeConfig: paths + metadata readers, freq↔timeframe map
|
||||
│ │ └── providers.py # LakeCalendarProvider / LakeInstrumentProvider / LakeFeatureProvider
|
||||
│ └── contrib/data/
|
||||
│ └── handler.py # TACHandler (DataHandlerLP) + DropAllNaN processor
|
||||
├── workflows/
|
||||
│ └── workflow_lgb_taclake.yaml # qrun workflow: train -> signal -> backtest
|
||||
├── examples/
|
||||
│ └── run_backtest.py # same loop as the workflow, plain Python (no yaml)
|
||||
└── tests/
|
||||
└── test_lake_providers.py # plain-assert smoke tests
|
||||
```
|
||||
|
||||
## Requirements / install
|
||||
|
||||
- Python 3.12, `qlib` nightly (`0.1.dev2066` in the repo venv), pandas/pyarrow, lightgbm.
|
||||
- The TradeAC lake (see below). `TAC_LAKE_DIR` is **mandatory** (no default) — set it to the
|
||||
lake root, or pass the `lake_root` kwargs.
|
||||
|
||||
## The lake (data layout)
|
||||
|
||||
```
|
||||
$TAC_LAKE_DIR/
|
||||
├── market=US/
|
||||
│ └── timeframe=1d/
|
||||
│ └── symbol=AAPL.parquet # OHLCV bars: t, date, o, h, l, c, v, n, vw
|
||||
├── features/
|
||||
│ └── market=US/timeframe=1d/
|
||||
│ └── symbol=AAPL.parquet # ta-lib indicators, wide format: t, sma_5, rsi_14, ...
|
||||
├── calendar.parquet # trading days per market
|
||||
├── coverage.parquet # per (market,timeframe,symbol) loaded windows
|
||||
└── symbols.parquet # asset master
|
||||
```
|
||||
|
||||
Field routing (`tac_qlib/data/config.py`):
|
||||
|
||||
- `$open $high $low $close $volume $vwap` → bar parquet columns; `$amount` = `v * vw`, `$avg_amount` = `vw`.
|
||||
- `$factor $change $trade_unit $suspend_flag` → all-NaN (not stored; the backtest Exchange only needs `$close`).
|
||||
- anything else (e.g. `$rsi_14`, `$sma_20`) → a ta-lib column in the features parquet.
|
||||
|
||||
## Step 1 — Prepare data
|
||||
|
||||
The lake is populated and backfilled with the tac-engine MCP lake tools (see
|
||||
`tac-engine/skills/tradeac-lake`). Typical sequence:
|
||||
|
||||
1. Seed the trading calendar from historical auctions (so the 1d completeness check has an expected day set):
|
||||
`backfill_lake_calendar(symbols="AAPL,MSFT,...")`.
|
||||
2. Backfill bars: `get_lake_bars(symbols="AAPL,MSFT,...", timeframe="1d", start="2026-02-09")` (lazy: missing
|
||||
windows are fetched from Alpaca and persisted; `sip`/`iex` auto-fallback on 403).
|
||||
3. Persist features: `get_lake_ta(symbol="AAPL", timeframe="1d", indicators="sma_5,sma_20,rsi_14,macd,bb,atr_14", persist=true)`.
|
||||
Only indicators that exist in *every* features file are auto-loaded by the handler; add columns per symbol by
|
||||
re-running `get_lake_ta`.
|
||||
4. `get_lake_symbols` / `get_lake_coverage` to verify the universe and loaded windows.
|
||||
|
||||
`TACHandler` discovers the feature columns itself (`get_common_feature_fields` = the intersection of columns
|
||||
across all features files), so no config change is needed as the lake grows.
|
||||
|
||||
## Step 2 — Preprocess
|
||||
|
||||
Preprocessing happens in `TACHandler` (a `DataHandlerLP`), composed from standard qlib processors:
|
||||
|
||||
- **infer** (`DEFAULT_INFER_PROCESSORS`), applied to the input features:
|
||||
1. `DropAllNaN` — drops columns that are all-NaN over the fit window (fixes the lake's fully-empty ta-lib
|
||||
columns, e.g. a `stoch_*` output that is NaN from the start). The drop set is fixed in `fit()` and applied
|
||||
identically to train/valid/test so feature columns never diverge.
|
||||
2. `ProcessInf`, `ZScoreNorm` (fit on the fit window), `Fillna`.
|
||||
- **learn** (`DEFAULT_LEARN_PROCESSORS`), applied to the label: `DropnaLabel`, `CSZScoreNorm`.
|
||||
|
||||
Handler kwargs (used by both the workflow yaml and the Python API):
|
||||
|
||||
| kwarg | default | meaning |
|
||||
|---|---|---|
|
||||
| `instruments` | `all` | universe; list, `all`, or a named pool from `markets:` |
|
||||
| `start_time` / `end_time` | – | queried window (must be within the lake calendar) |
|
||||
| `fit_start_time` / `fit_end_time` | start/end | window the fit-able processors (ZScoreNorm, DropAllNaN) fit on |
|
||||
| `freq` | `day` | maps to the lake timeframe (`day`→`1d`, `1min`→`1m`, …) |
|
||||
| `feature_fields` | auto | raw OHLCV + common ta-lib columns; or an explicit list |
|
||||
| `label` | `Ref($close,-2)/Ref($close,-1)-1` | qlib expression for the target |
|
||||
| `lake_root` / `market` | `$TAC_LAKE_DIR` / `US` | lake location (required) / market partition |
|
||||
|
||||
Only daily (`1d`) is currently supported by the calendar provider; intraday freq raises `NotImplementedError`.
|
||||
|
||||
## Step 3 — Train
|
||||
|
||||
Either write the model task in yaml and run qrun (see *Glue with qrun*), or train in Python:
|
||||
|
||||
```python
|
||||
from qlib.data.dataset import DatasetH
|
||||
from tac_qlib.qlib_init import qlib_init
|
||||
from tac_qlib.contrib.data.handler import TACHandler
|
||||
from qlib.contrib.model.gbdt import LGBModel
|
||||
from qlib.workflow import R
|
||||
|
||||
qlib_init(provider_uri=os.environ["TAC_LAKE_DIR"], market="US", freq="day")
|
||||
|
||||
handler = TACHandler(
|
||||
instruments="all",
|
||||
start_time="2026-03-01", end_time="2026-08-06",
|
||||
fit_start_time="2026-03-01", fit_end_time="2026-05-31",
|
||||
freq="day", lake_root=os.environ["TAC_LAKE_DIR"], market="US",
|
||||
)
|
||||
dataset = DatasetH(handler=handler, segments={
|
||||
"train": ("2026-03-01", "2026-05-31"),
|
||||
"valid": ("2026-06-01", "2026-06-30"),
|
||||
"test": ("2026-07-01", "2026-08-06"),
|
||||
})
|
||||
|
||||
model = LGBModel(n_estimators=200, learning_rate=0.05, num_leaves=15, ...)
|
||||
with R.start(experiment_name="tac-lake-demo"):
|
||||
model.fit(dataset) # trains on the train segment
|
||||
```
|
||||
|
||||
## Step 4 — Test / evaluate the signal
|
||||
|
||||
`model.predict(dataset)` returns the prediction on the **test** segment (a `(datetime, instrument)` Series).
|
||||
Evaluate it with qlib's `SigAnaRecord` / `sig_analysis`:
|
||||
|
||||
```python
|
||||
from qlib.workflow.record_temp import SigAnaRecord
|
||||
from qlib.contrib.evaluate import signal_analysis
|
||||
|
||||
pred = model.predict(dataset) # "score" column
|
||||
label = dataset.prepare("test", col_set="label", data_key=DataHandlerLP.DK_I)["LABEL0"]
|
||||
# per-day + overall IC / ICIR / RankIC / RankICIR
|
||||
report = signal_analysis(pred, label)
|
||||
```
|
||||
|
||||
In the workflow this is automatic (`SigAnaRecord`): the run logs IC 0.0072 / ICIR 0.016 /
|
||||
RankIC 0.0138 / RankICIR 0.034 for the default split — weak but the plumbing is verified.
|
||||
|
||||
## Step 5 — Backtesting
|
||||
|
||||
```python
|
||||
from qlib.contrib.evaluate import backtest_daily, risk_analysis
|
||||
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
|
||||
|
||||
strategy = TopkDropoutStrategy(signal=pred, topk=2, n_drop=1, only_tradable=True, risk_degree=0.95)
|
||||
report_normal, positions_normal = backtest_daily(
|
||||
start_time="2026-07-01", end_time="2026-08-06",
|
||||
strategy=strategy, account=1_000_000, benchmark=None, # lake has no index quotes
|
||||
exchange_kwargs={"codes": universe, "deal_price": "$close", "freq": "day",
|
||||
"open_cost": 0.0005, "close_cost": 0.0015, "min_cost": 5.0},
|
||||
)
|
||||
risk = risk_analysis(report_normal["return"], freq="day")
|
||||
```
|
||||
|
||||
- `TopkDropoutStrategy` is the default mapping *prediction → positions* (hold top-k, drop `n_drop` per day).
|
||||
For other sizing frameworks — equal/score-weighting, softmax, z-score, fractional Kelly, mean-variance —
|
||||
subclass `qlib.contrib.strategy.SignalStrategy` and implement `generate_trade_decision` (see
|
||||
`../.tmp/signalTrade.md` for the recipe catalogue).
|
||||
- The Exchange needs `$close`; other fields the backtest probes (`$factor`, `$trade_unit`) are all-NaN and fine.
|
||||
- Benchmark: pick any symbol the lake holds (e.g. `benchmark: AAPL`); null benchmark triggers benign
|
||||
"Mean of empty slice" warnings from the risk analysis.
|
||||
|
||||
## Step 6 — Predict
|
||||
|
||||
`SignalRecord` already saved `pred.pkl` (test segment) during the qrun run. For predictions on arbitrary data:
|
||||
|
||||
```python
|
||||
pred = model.predict(dataset) # predict on the "test" segment
|
||||
pred.to_frame("score").to_pickle("pred.pkl") # (datetime, instrument) x ["score"]
|
||||
```
|
||||
|
||||
To predict a live/rolling window instead of the configured test segment, point a handler's `segments["test"]`
|
||||
at the window of interest, or call `model.predict(dataset, segment="test")` after overriding the segment.
|
||||
|
||||
## Glue everything with qrun
|
||||
|
||||
`workflows/workflow_lgb_taclake.yaml` wires the whole chain (init → train → signal record → signal analysis →
|
||||
backtest + risk analysis) into one qrun invocation:
|
||||
|
||||
```bash
|
||||
cd tac-qlib
|
||||
qrun workflows/workflow_lgb_taclake.yaml --experiment_name tac-lake-lgb
|
||||
# custom lake root:
|
||||
TAC_LAKE_DIR=/path/to/lake qrun workflows/workflow_lgb_taclake.yaml --experiment_name tac-lake-lgb
|
||||
```
|
||||
|
||||
YAML anatomy:
|
||||
|
||||
- `qlib_init` — points `provider_uri` at the lake and installs the lake providers by their full class paths
|
||||
(`tac_qlib.data.providers.Lake*Provider`), plus an `exp_manager` backed by `sqlite:///<lake>/mlruns.db`
|
||||
(avoids mlflow's filesystem-backend maintenance-mode opt-in). The unified R&D store lives under the lake
|
||||
root: `mlruns.db` + `mlruns/<exp>/<run>/`. Override the tracking URI with `MLRUNS_URI`.
|
||||
- `task.model` — `LGBModel` hyperparameters.
|
||||
- `task.dataset` — `DatasetH` over `TACHandler`; `segments.train/valid/test` split the window;
|
||||
`fit_start_time`/`fit_end_time` pin the processor fit window to train.
|
||||
- `task.record` — ordered records:
|
||||
1. `SignalRecord` → writes `pred.pkl` (and `label.pkl`).
|
||||
2. `SigAnaRecord` → `sig_analysis/{ic,ric}.pkl` (IC/ICIR/RankIC/RankICIR).
|
||||
3. `PortAnaRecord` → daily `TopkDropoutStrategy` backtest + `risk_analysis_freq: 1d` →
|
||||
`portfolio_analysis/*.pkl` (report, positions, indicators, risk metrics, benchmark & cost-adjusted excess returns).
|
||||
|
||||
Run artifacts land under the mlflow run: `<lake>/mlruns/<exp>/<run>/artifacts/*.pkl` (metadata in `<lake>/mlruns.db`).
|
||||
|
||||
Template notes:
|
||||
|
||||
- The header uses jinja2 (`{%- set LAKE = TAC_LAKE_DIR %}`) — `TAC_LAKE_DIR` is **required** and names the
|
||||
lake root. Do **not** use `-%}` on the closing tag — it strips the newline and glues
|
||||
`qlib_init:` onto the comment line (YAML parse error).
|
||||
- `qrun` is `qlib.cli.run:run` (fire): positional CONFIG_PATH + `--experiment_name` / `--uri_folder`. No
|
||||
`--config` flag.
|
||||
|
||||
## Manual (no-yaml) path
|
||||
|
||||
`examples/run_backtest.py` runs the identical loop in plain Python (good for parametrizing universe, features,
|
||||
label, topk, costs):
|
||||
|
||||
```bash
|
||||
.venv/bin/python tac-qlib/examples/run_backtest.py
|
||||
.venv/bin/python tac-qlib/examples/run_backtest.py --features '$close,$rsi_14,$sma_5,$macd' \
|
||||
--universe AAPL,MSFT,TSLA,USO,SLV,TLT --topk 2 --n-drop 1 --output ./backtest_out
|
||||
```
|
||||
|
||||
Writes `pred.pkl`, `report_normal.csv`, `positions_normal.csv`, `risk.csv` to the output dir.
|
||||
|
||||
## Reference
|
||||
|
||||
- `tac_qlib/data/providers.py` — the three lake providers; they match qlib's provider interface
|
||||
(`feature()` keyed by calendar position, `list_instruments()` with listing spans, `load_calendar()`), so the
|
||||
expression engine, `DatasetH` and the backtest `Exchange` work unchanged.
|
||||
- `tac_qlib/contrib/data/handler.py` — `TACHandler` (DataHandlerLP over `QlibDataLoader`),
|
||||
`DropAllNaN`, `get_common_feature_fields`, `discover_feature_fields`.
|
||||
- `tac_qlib/data/config.py` — `LakeConfig` path/reader helpers, `FREQ_TO_TIMEFRAME`, `BAR_FIELD_MAP`,
|
||||
`resolve_lake_root` (`$TAC_LAKE_DIR`, required — fails fast if unset).
|
||||
- Tests (no pytest; plain asserts):
|
||||
|
||||
```bash
|
||||
.venv/bin/python tac-qlib/tests/test_lake_providers.py
|
||||
```
|
||||
@@ -0,0 +1,149 @@
|
||||
"""End-to-end example: train a LightGBM on TradeAC lake data and backtest it.
|
||||
|
||||
Reads OHLCV + ta-lib features straight from the TradeAC parquet lake through the
|
||||
tac-qlib providers and the ``TACHandler``, then runs the standard qlib research
|
||||
loop (LightGBM + TopkDropoutStrategy + daily backtest).
|
||||
|
||||
Usage::
|
||||
|
||||
.venv/bin/python tac-qlib/examples/run_backtest.py # defaults
|
||||
.venv/bin/python tac-qlib/examples/run_backtest.py --features '$close,$rsi_14,$sma_5,$macd' \\
|
||||
--universe AAPL,MSFT,TSLA,USO,SLV,TLT --output ./backtest_out
|
||||
|
||||
The lake has ~5 months of 1d bars (2026-02-09 .. 2026-08-06); the default split is
|
||||
train 2026-03-01..2026-05-31 / valid 2026-06-01..2026-06-30 / test 2026-07-01..2026-08-06.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
|
||||
def parse_args():
|
||||
p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
p.add_argument("--lake-root", default=os.environ.get("TAC_LAKE_DIR"))
|
||||
p.add_argument("--market", default="US")
|
||||
p.add_argument("--universe", default="AAPL,MSFT,TSLA,USO,SLV,TLT",
|
||||
help="comma-separated instruments (default: the 1d-bar symbols)")
|
||||
p.add_argument("--features", default="$open,$high,$low,$close,$vwap,$volume,$amount",
|
||||
help="comma-separated feature fields ($-prefixed)")
|
||||
p.add_argument("--label", default="Ref($close,-2)/$close-1")
|
||||
p.add_argument("--train-start", default="2026-03-01")
|
||||
p.add_argument("--train-end", default="2026-05-31")
|
||||
p.add_argument("--valid-end", default="2026-06-30")
|
||||
p.add_argument("--test-end", default="2026-08-06")
|
||||
p.add_argument("--topk", type=int, default=2)
|
||||
p.add_argument("--n-drop", type=int, default=1)
|
||||
p.add_argument("--init-cash", type=float, default=1_000_000.0)
|
||||
p.add_argument("--output", default="backtest_output")
|
||||
return p.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
args = parse_args()
|
||||
logging.basicConfig(level=logging.WARNING)
|
||||
logging.getLogger("lightgbm").setLevel(logging.WARNING)
|
||||
os.environ.setdefault("MLFLOW_ALLOW_FILE_STORE", "true") # qlib's mlflow file store opt-in
|
||||
|
||||
universe = [s.strip().upper() for s in args.universe.split(",") if s.strip()]
|
||||
feature_fields = [f.strip() for f in args.features.split(",") if f.strip()]
|
||||
|
||||
from tac_qlib.qlib_init import qlib_init
|
||||
|
||||
qlib_init(provider_uri=args.lake_root, market=args.market, freq="day")
|
||||
|
||||
from qlib.data.dataset import DatasetH
|
||||
from tac_qlib.contrib.data.handler import TACHandler
|
||||
|
||||
valid_start = str(pd.Timestamp(args.train_end) + pd.Timedelta(days=1)).split()[0]
|
||||
test_start = str(pd.Timestamp(args.valid_end) + pd.Timedelta(days=1)).split()[0]
|
||||
|
||||
# ---- dataset ---------------------------------------------------------
|
||||
handler = TACHandler(
|
||||
instruments=universe,
|
||||
start_time=args.train_start,
|
||||
end_time=args.test_end,
|
||||
freq="day",
|
||||
fit_start_time=args.train_start,
|
||||
fit_end_time=args.train_end,
|
||||
feature_fields=feature_fields,
|
||||
label=args.label,
|
||||
lake_root=args.lake_root,
|
||||
market=args.market,
|
||||
)
|
||||
dataset = DatasetH(
|
||||
handler=handler,
|
||||
segments={
|
||||
"train": (args.train_start, args.train_end),
|
||||
"valid": (valid_start, args.valid_end),
|
||||
"test": (test_start, args.test_end),
|
||||
},
|
||||
)
|
||||
|
||||
# ---- train ------------------------------------------------------------
|
||||
from qlib.contrib.model.gbdt import LGBModel
|
||||
|
||||
model = LGBModel(n_estimators=200, learning_rate=0.05, num_leaves=15, colsample_bytree=0.8,
|
||||
subsample=0.8, subsample_freq=1, reg_alpha=0.01, reg_lambda=0.01)
|
||||
|
||||
t0 = time.time()
|
||||
from qlib.workflow import R
|
||||
|
||||
with R.start(experiment_name="tac-lake-demo"):
|
||||
model.fit(dataset)
|
||||
print(f"[train] fitted LGBModel in {time.time() - t0:.1f}s")
|
||||
|
||||
# ---- predict ----------------------------------------------------------
|
||||
pred = model.predict(dataset) # (datetime, instrument) MultiIndex Series
|
||||
print(f"[predict] {len(pred)} signals on test segment {test_start}..{args.test_end}")
|
||||
print(pred.head(5))
|
||||
|
||||
# ---- backtest ---------------------------------------------------------
|
||||
from qlib.contrib.evaluate import backtest_daily, risk_analysis
|
||||
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
|
||||
|
||||
strategy = TopkDropoutStrategy(signal=pred, topk=args.topk, n_drop=args.n_drop,
|
||||
only_tradable=True, risk_degree=0.95)
|
||||
t0 = time.time()
|
||||
report_normal, positions_normal = backtest_daily(
|
||||
start_time=test_start,
|
||||
end_time=args.test_end,
|
||||
strategy=strategy,
|
||||
account=args.init_cash,
|
||||
benchmark=None, # the lake has no index quotes
|
||||
exchange_kwargs={
|
||||
"codes": universe,
|
||||
"deal_price": "$close",
|
||||
"freq": "day",
|
||||
"open_cost": 0.0005,
|
||||
"close_cost": 0.0015,
|
||||
"min_cost": 5.0,
|
||||
},
|
||||
)
|
||||
print(f"[backtest] ran in {time.time() - t0:.1f}s over {len(report_normal)} trading days")
|
||||
|
||||
risk = risk_analysis(report_normal["return"], freq="day")
|
||||
print("\n=== backtest risk analysis ===")
|
||||
print(risk.round(6).to_string())
|
||||
|
||||
# ---- save -------------------------------------------------------------
|
||||
out = Path(args.output)
|
||||
out.mkdir(parents=True, exist_ok=True)
|
||||
pred.to_frame("score").to_pickle(out / "pred.pkl")
|
||||
report_normal.to_csv(out / "report_normal.csv")
|
||||
pd.DataFrame({ts: pos.get_stock_amount_dict() for ts, pos in positions_normal.items()}).T.to_csv(
|
||||
out / "positions_normal.csv"
|
||||
)
|
||||
risk.to_csv(out / "risk.csv")
|
||||
print(f"\nsaved artifacts to {out}/ (pred.pkl, report_normal.csv, positions_normal.csv, risk.csv)")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,28 @@
|
||||
[build-system]
|
||||
requires = ["setuptools>=61"]
|
||||
build-backend = "setuptools.build_meta"
|
||||
|
||||
[project]
|
||||
name = "tac-qlib"
|
||||
version = "0.1.0"
|
||||
description = "TradeAC qlib integration: read the parquet+DuckDB lake (OHLCV bars + ta-lib features) from within the qlib research workflow"
|
||||
requires-python = ">=3.10"
|
||||
dependencies = [
|
||||
"pyarrow",
|
||||
"duckdb",
|
||||
"pandas>=1.1",
|
||||
"pyqlib",
|
||||
"mcp[cli]",
|
||||
"python-dotenv",
|
||||
"psycopg[binary]",
|
||||
]
|
||||
|
||||
[tool.setuptools]
|
||||
packages = [
|
||||
"tac_qlib",
|
||||
"tac_qlib.data",
|
||||
"tac_qlib.contrib",
|
||||
"tac_qlib.contrib.data",
|
||||
"tac_qlib.contrib.model",
|
||||
"tac_qlib.contrib.strategy",
|
||||
]
|
||||
@@ -0,0 +1,325 @@
|
||||
---
|
||||
name: tac-algo-trade
|
||||
description: Guide agents to run the TradeAC scheduled algo-trading flow end-to-end. First backfill the data lake for all symbols up to the latest completed trading day, then — given a reference MLflow run (experiment_name + run_id) — re-train the same model configuration on a rolling window (4 years up to the latest completed trading day), generate fresh signals, run the selected strategy into a target order list, and place the orders on the Alpaca paper account — chaining every step from the previous one's output. Uses the `tac-engine` lake MCP tools (get_lake_bars, backfill_lake_calendar, get_lake_coverage) for data, the `tac-qlib-rd` MCP tools (rd_train, rd_predict, rd_strategy_targets, rd_exp_*) for the quant side, and the `tac-engine` MCP tools (place_order, list_orders, list_positions, get_news, ...) for execution.
|
||||
---
|
||||
|
||||
# tac-algo-trade
|
||||
|
||||
Scheduled algo trading for the TradeAC paper account. Each scheduled execution (1) backfills the lake so it is current through the latest completed trading day, (2) re-trains the reference run's configuration on the most recent 4 years of lake data, (3) predicts, (4) derives an order list from the strategy, and (5) places orders on Alpaca.
|
||||
|
||||
This is the skill the app's scheduler invokes (`/dashboard/scheduler`). It chains strict step-to-step outputs: **do not skip ahead, do not fabricate outputs — every step consumes the artifact path returned by the previous one.**
|
||||
|
||||
## MCP tools
|
||||
|
||||
- Data side: `tac-engine` lake tools (see `tac-engine/skills/tradeac-lake/SKILL.md`) — `get_lake_coverage`, `backfill_lake_calendar`, `get_lake_bars` (lazy backfill), `get_lake_status`.
|
||||
- Quant side: `tac-qlib-rd` (see `tradeac-rd/SKILL.md`) — `rd_status`, `rd_exp_get_experiment`, `rd_exp_input`, `rd_train`, `rd_predict`, `rd_strategy_targets`, `rd_exp_get_run`.
|
||||
- Execution side: `tac-engine` (see `tac-engine/skills/tradeac-alpaca/SKILL.md`) — `get_account`, `list_positions`, `list_orders`, `place_order`, `get_stock_snapshot`, `get_stock_latest_quotes`, `get_news`.
|
||||
- Round book: `tac-rd-book` — the execution trail (see the "Round book" section below). `round_create` / `round_update`, `fact_record`, `intent_set`, `decision_record`, `round_sync_fills`, `round_update_status`, `book_reconcile`, `trail_funnel`.
|
||||
|
||||
All steps use the MCP tools directly. Never hand-compute scores, read mlruns files directly (`mlruns.db` / pickles — the store is Postgres via `DATABASE_URL` when set), or script the MCP servers yourself. This is a **paper** account — trade normally, size each order per the strategy's target weight × live equity (whole shares), capped by available buying power, and skip anything untradeable.
|
||||
|
||||
## Inputs
|
||||
|
||||
- `experiment_name` — MLflow experiment of the reference run.
|
||||
- `run_id` — the reference run inside that experiment (its saved `config` artifact is the source of truth for the whole pipeline).
|
||||
- `strategy` (optional) — a workflow YAML from `tac-qlib/workflows/*.yaml` defining the strategy sizing (e.g. `topk` / `n_drop` / `risk_degree`, benchmark, costs). Default: the reference run's own backtest config.
|
||||
- Time context: today's date in the scheduling city's timezone.
|
||||
|
||||
## Step 1 — Backfill the lake (data currency)
|
||||
|
||||
The retrain must see every symbol up to the latest available bar — do **not** train on stale data.
|
||||
|
||||
1. `get_lake_coverage` `{"market":"US","timeframe":"1d"}` → read each symbol's loaded window; the **last loaded date** across the universe is your backfill start.
|
||||
2. `backfill_lake_calendar` `{"market":"US","symbols":"all","start":<backfill start>,"end":<today>}` → seed the trading-day set first so the 1d completeness check knows which days to expect.
|
||||
3. `get_lake_bars` `{"market":"US","symbols":"all","timeframe":"1d","start":<backfill start>,"end":<today>,"lazy":true,"quiet":true}` → backfill every symbol's gap (Alpaca historical bars) and persist to the lake. Use `quiet: true` so the tool returns a per-symbol `{count, first_t, last_t}` summary instead of echoing back thousands of bar rows. Alpaca has no bar for today until the session closes, so the latest bar landed is the **latest completed trading day** `D` (for a Monday run this is Friday).
|
||||
|
||||
Confirm with `rd_status` (calendar range + coverage) that the lake is populated through `D`. **Output: `D`, the latest completed trading day.**
|
||||
|
||||
## Step 2 — Inspect the reference run
|
||||
|
||||
`rd_exp_get_experiment` with `experiment_id` (or `rd_exp_input` with `run_id`) → extract from the run's `config` artifact:
|
||||
|
||||
- handler config: `universe` (instruments), `features`, `label`, `freq`
|
||||
- model kwargs: `learning_rate`, `num_leaves`, `n_estimators`, `colsample_bytree`, `subsample`, `subsample_freq`, `reg_alpha`, `reg_lambda`, `seed`
|
||||
- strategy sizing: `topk` / `n_drop` / `risk_degree`, costs, benchmark
|
||||
|
||||
Record these — they define the retrain. **Output: config values above.**
|
||||
|
||||
## Step 3 — Open the traced experiment (git lineage)
|
||||
|
||||
Every scheduled run is a **traced experiment** on the tac-qlib-custom lineage: a row in the
|
||||
`rd_experiments` table plus a per-experiment git branch in the `experiments` submodule,
|
||||
forked from the predecessor's branch. **This is part of the run — do it automatically, do
|
||||
not wait for the user to prompt** (see `tac-qlib/skills/tac-qlib-custom/SKILL.md`,
|
||||
"Experiment traceability", for the full procedure and env vars).
|
||||
|
||||
1. Resolve the predecessor: if the reference run (`experiment_name` / `run_id` from Step 2)
|
||||
is itself traced, reuse its traced id as `evolved_from`; otherwise use `--evolved-from auto`
|
||||
(semantic search over existing rationals).
|
||||
2. Open the trace — this inserts the row, forks the branch from the predecessor and pushes it (via the `rd_trace_*` MCP tools on tac-qlib-rd):
|
||||
|
||||
```
|
||||
rd_trace_init
|
||||
rd_trace_start rational="scheduled algo retrain on <D>: <ref exp>/<ref run> re-trained on 4y -> live paper orders" \
|
||||
details="<universe / features / label / model / strategy sizing from the reference run config>" \
|
||||
experiment_name=<THE RUN'S experiment name — see naming below> \
|
||||
evolved_from=<predecessor id or auto> \
|
||||
session_id="<this chat's opencode session id>"
|
||||
# -> {"experiment_id": N, "branch": "...", "evolved_from": ..., "base_branch": ...}
|
||||
```
|
||||
|
||||
**Experiment naming (unique per run):** every retrain runs into its OWN
|
||||
experiment — `<reference experiment name>-<epoch seconds>` (e.g.
|
||||
`tac-basic-short-1786883261`). The scheduler prompt names the exact
|
||||
experiment for you; use that name for `rd_trace_start experiment_name`,
|
||||
`rd_train`'s `experiment_name`, and the round's `experiment_name`. **Never**
|
||||
reuse the reference experiment name for this run's trace node — reusing it
|
||||
creates duplicate lineage entries with the same name and a wrong parent
|
||||
chain (seen with `tac-basic-short`).
|
||||
|
||||
The tool returns `experiment_id` / `branch` as JSON — record them; every
|
||||
later `rd_trace_*` call uses the id. Commit the run's files (workflow YAML /
|
||||
notes) with `rd_trace_commit experiment_id=<N> message="..."` as you go.
|
||||
|
||||
3. **Round window** — the execution trail for `D`:
|
||||
- **If the scheduler pre-created it** (your instructions name a `ROUND_ID` / `target_date` / `source`) — **skip `round_create`** and use that `ROUND_ID`. If the `D` you computed in Step 1 differs from the given `target_date`, correct it first with `round_update {round_id:<ROUND_ID>, target_date:<D>}` (weekday rule can't see NYSE holidays; the agent reconciles).
|
||||
- Otherwise create it yourself (idempotent: a second scheduled run for the same day reuses the open window):
|
||||
|
||||
```
|
||||
round_create {target_date:<D>, signal_date:<D>, source:"scheduled",
|
||||
rd_experiment_id:<EXPERIMENT_ID>, experiment_name:<the run's unique experiment name>}
|
||||
# -> round_id (record it; every round-book call below uses it)
|
||||
```
|
||||
|
||||
**Output: `EXPERIMENT_ID` (and its branch), `ROUND_ID`.**
|
||||
|
||||
## Step 4 — Re-train with the rolling window
|
||||
|
||||
Call `rd_train` with the **exact same configuration** from Step 2, only the dates change:
|
||||
|
||||
- `train_start` = 4 years before `D` (same day-of-month), `train_end` = `D`
|
||||
- **Validation is optional** — qlib supports omitting it, so omit `valid_start`/`valid_end`/`test_start`/`test_end` (pass them empty). If the tool/your run requires a holdout for sanity, use a short recent `valid` window only; never reserve data the live model needs.
|
||||
- `record_analysis=false` (we only need the model; no SignalRecord/PortAnaRecord on a holdout we don't use)
|
||||
- `wait=false` (recommended) — `rd_train` returns immediately and the fit runs in the background; poll `rd_exp_get_run` (or `rd_exp_list` filtered to the experiment) until the newest run's status is `FINISHED`, then take its `run_id`. With `wait=true` the call blocks until the fit completes — fine when the window is small, but a 4y LightGBM fit can outlive the MCP call timeout, which forced manual recovery in an earlier run.
|
||||
- `out_dir` — the working directory for this run (e.g. `tac-algo-output`)
|
||||
- `experiment_name` — the **run's unique experiment name** (the scheduler prompt names it: `<reference experiment name>-<epoch seconds>`). This is the SAME name used for `rd_trace_start experiment_name` and the round's `experiment_name`. Do not reuse the reference experiment name.
|
||||
|
||||
Keep the same `universe`, `features`, `label`, and every model hyper-parameter. **Output: the new run's `model_path` (and its `run_id`).**
|
||||
|
||||
> If a 4-year window is slower than the schedule allows, use the largest trailing window you can complete and say so in the summary — never silently shrink the horizon.
|
||||
|
||||
Pin the new training run to the round window:
|
||||
|
||||
```
|
||||
round_update {round_id:<ROUND_ID>, run_id:<new run_id>, model_path:<params.pkl path>}
|
||||
```
|
||||
|
||||
## Step 5 — Generate predictions (the signal)
|
||||
|
||||
Call `rd_predict` with `model_path` = the path returned by Step 4 (preferred over `run_id` since it is the freshly-trained artifact):
|
||||
|
||||
- `test_start` = `D`, `test_end` = `D` (the just-completed trading day — this is the signal we trade on)
|
||||
- same `universe` / `features` / `label` as Step 2
|
||||
- `out_dir` = the same working directory
|
||||
|
||||
**Output: `pred_path` (pred.pkl) and the score ranking.** The model's predicted score per instrument IS the alpha signal for day `D` — top-scored names are candidates.
|
||||
|
||||
Record the signal into the round book (one `fact_record` per top-scored name, plus the strategy config and the market snapshot at prediction time):
|
||||
|
||||
```
|
||||
fact_record {round_id:<ROUND_ID>, kind:"signal_score", symbol:<ticker>, payload:{"pred":<score>, "rank":<rank>}, source:"rd_predict"}
|
||||
fact_record {round_id:<ROUND_ID>, kind:"strategy_config", payload:{...strategy sizing...}, source:"reference config"}
|
||||
fact_record {round_id:<ROUND_ID>, kind:"market_snapshot", payload:{<ticker>: {last:<px>, change_pct:<%>, vol:<vol>, updated:<ts>}, ...}, source:"get_stock_snapshots / get_stock_latest_quotes"}
|
||||
```
|
||||
|
||||
`market_snapshot` freezes the market state **when the prediction was made** — the latest price / % change / volume per universe name, so the signal can later be judged against what the market looked like at that moment.
|
||||
|
||||
## Step 6 — Run the configured strategy, derive the target order list
|
||||
|
||||
**First pull the current portfolio — it is an input to the strategy step** (the order list is a delta, not a full rebuild):
|
||||
|
||||
- `get_account` → cash / buying power **and total equity** (equity sizes the positions; buying power caps total buys)
|
||||
- `list_positions` → current holdings and their market value
|
||||
|
||||
Then run the strategy **exactly as it was configured in the reference run** — this works for any model/strategy, not just TopkDropout. The reference run's saved `config` artifact (from `rd_exp_input`, Step 2) carries the strategy configuration from its backtest/record block (e.g. `TopkDropoutStrategy` kwargs: `topk`, `n_drop`, `risk_degree`, or any custom strategy's own kwargs, plus costs, `account`, `benchmark`). **Use those values — not tool defaults.** The model's score is the signal the strategy consumes; the strategy's config decides allocation.
|
||||
|
||||
Call `rd_strategy_targets` with:
|
||||
|
||||
- `pred_path` = the signal from Step 5
|
||||
- the run-configured `topk` / `n_drop` / `risk_degree` (from the reference run config)
|
||||
- `account` = the **live account equity** from `get_account` (a new account is not a $1M book — sizing against `$1M` when equity is far smaller produces oversized orders)
|
||||
- `prices` = a JSON `{symbol: price}` of latest quotes (from `get_stock_latest_quotes`) so the tool floors each order to whole shares (`qty`) and reports `expected_price` / `invested`
|
||||
- `risk_limits` = the round's risk-limit spec JSON (see below) — the SAME spec that `rd_backtest` uses, so live gating is provable against backtest
|
||||
- `equity` / `peak_equity` = live equity and its trailing peak (from `get_portfolio_history`) when `risk_limits.drawdown_pause_pct` is set
|
||||
|
||||
The tool applies the exact TopkDropout selection on day `D`: rank the cross-sectional scores, **drop the top `n_drop`**, take the next `topk` as buys, sized at `account × risk_degree / topk` per name. It then applies `risk_limits` as pre-gates — liquidity floor (drops names with avg daily dollar volume below `liquidity_floor_adv`), per-name `size_cap_pct` of equity, `concentration_cap_pct` of equity on total deployed, and `drawdown_pause_pct` (equity ≤ (1−pause)×peak ⇒ no buys). **Output: the deterministic target buy list** (`symbol`, `rank`, `score`, `side`, `notional`, `qty`), the full `ranking`, and `risk_limits_applied` (which limits cut what — record it). **No manual strategy replication** (an earlier run's hand-rolled sizing silently dropped the n_drop and bought the wrong names).
|
||||
|
||||
> If the strategy in the run/workflow config does not fit TopkDropout's `topk`/`n_drop`/`risk_degree`, apply the strategy's own rules to the Step 5 scores directly to derive the target portfolio, still bounded by `get_account` buying power and today's `list_positions`.
|
||||
|
||||
Then convert the target portfolio into an order list against the current holdings:
|
||||
|
||||
- For each target ticker compute the **delta** vs. what the account already holds: buy the shortfall, sell the excess. Do not blindly re-buy names already held, and do not sell names that are not in the portfolio.
|
||||
- **Fresh account (no positions):** the target portfolio is entirely new buys — emit no sell orders, and size each buy from the tool's `qty` (or `notional` ÷ latest quote), capped by buying power.
|
||||
- Skip any ticker whose delta is ~0 (already at target) so you don't churn held names.
|
||||
- Cap total buy size to available buying power. Drop any ticker with no score in Step 5 or no tradable quote.
|
||||
|
||||
**Output: the explicit order list** (ticker, side, qty, order type).
|
||||
|
||||
**Write the target into the round book** — this is the intent the round reconciles against (versions auto-increment; a second strategy pass for the same round supersedes the first):
|
||||
|
||||
```
|
||||
fact_record {round_id:<ROUND_ID>, kind:"account_state", payload:{"equity":<live equity>, "buying_power":<bp>}, source:"get_account"}
|
||||
fact_record {round_id:<ROUND_ID>, kind:"position_state", symbol:<ticker>, payload:{"shares":<held>}, source:"list_positions"}
|
||||
fact_record {round_id:<ROUND_ID>, kind:"risk_check", payload:{"risk_limits":{...spec...}, "applied":{...risk_limits_applied from the tool...}, "equity":<equity>, "peak_equity":<peak>}, source:"rd_strategy_targets"}
|
||||
round_update {round_id:<ROUND_ID>, account_equity_at_sizing:<live equity>, strategy_snapshot:{topk, n_drop, risk_degree, costs, benchmark, risk_limits:{liquidity_floor_adv?, size_cap_pct?, concentration_cap_pct?, drawdown_pause_pct?}}}
|
||||
intent_set {round_id:<ROUND_ID>, target_portfolio:[{symbol, side, qty, notional, expected_price, score, rank}...],
|
||||
raw_strategy_output:{...the strategy output as computed...}, reason:"topk<N> from <ref run>"}
|
||||
```
|
||||
|
||||
**Risk-limit spec (B)**: the round's `risk_limits` (a JSON map with any of `liquidity_floor_adv`, `size_cap_pct`, `concentration_cap_pct`, `drawdown_pause_pct`) is the single source of truth — **the same spec is passed to `rd_backtest` when calibrating** (B2), folded into `rd_train`'s PortAnaRecord via `risk_degree`, and consulted by `rd_strategy_targets` live. Store it verbatim in `strategy_snapshot.risk_limits`. When the tool's `risk_limits_applied` reports a limit that cut targets (dropped liquidity / capped sizing / drawdown pause), record it — the audit trail proves the limit fired live exactly as the calibration predicted. If `drawdown_pause_pct` fired and produced an empty target list, **settle the round as open→settled with no orders** rather than forcing buys (that is the intended behavior).
|
||||
|
||||
**Record the evidence behind each selected name** — the feature snapshot and the decision rationale, so the fact table can answer *why this symbol was ranked top-K*:
|
||||
|
||||
- `symbol_features` — the model-input feature values that produced the score on day `D` (the top features by `rd_exp_model` importance, plus the handful most relevant for that name — e.g. trend slopes, RSI, volume/vol ratios, MACD):
|
||||
```
|
||||
get_lake_ta {symbol:<ticker>, timeframe:"1d", start:<~60d before D>, end:<D>, persist:true, quiet:true} # (re)compute TA + sp_* columns up to D
|
||||
get_lake_features {symbol:<ticker>, timeframe:"1d", start:<D>, end:<D>} # read the D row; if 0 rows, the persisted features are stale -> persist first as above
|
||||
rd_exp_model {run_id:<new training run_id>, tree_id:0, max_depth:4} # feature_importances + tree nodes
|
||||
fact_record {round_id:<ROUND_ID>, kind:"symbol_features", symbol:<ticker>,
|
||||
payload:{"score":<score>, "rank":<rank>, "features":{<top feature>:<value>, ...}}, source:"get_lake_features / rd_exp_model"}
|
||||
```
|
||||
`get_lake_features` returns 0 rows for day `D` when the persisted feature files were last written before `D` (they are per-symbol parquet files that only extend to the last time they were computed). In that case **first persist** with `get_lake_ta ... persist:true` (and `get_lake_sp` when the model uses `sp_*` columns — the rd_train feature list from Step 2 tells you which), then read `get_lake_features` for `D` again — it must return a row per ticker.
|
||||
- `decision_justification` — **concise** (under 500 words total, aim for 2–4 sentences per name): why the model ranked the symbol top-K. Ground it in the actual data — the `rd_exp_model` tree path (which feature conditions led the row down the high-score branch) and the `symbol_features` values — not generic commentary:
|
||||
```
|
||||
fact_record {round_id:<ROUND_ID>, kind:"decision_justification", symbol:<ticker>,
|
||||
payload:{"score":<score>, "rank":<rank>, "why": "<2-4 sentences, e.g. 'strong 5d trend slope + rising volume ratio put TSLA above $sp_trend_slope_60 threshold, sending it down the high-score branch (leaf value +0.0545); RSI recovering but not overbought.'>"},
|
||||
source:"rd_exp_model tree + feature snapshot"}
|
||||
```
|
||||
|
||||
## Step 7 — Execution context + news sentiment gate
|
||||
|
||||
Before placing anything, per candidate ticker:
|
||||
|
||||
1. `get_account` (buying power), `list_orders` (open orders), `list_positions` (current holdings).
|
||||
2. `get_stock_snapshot` / `get_stock_latest_quotes` → sanity-check each quote: skip tickers with no quote, a stale/illiquid quote (wide spread or near-zero volume), or a halt. Use the latest quote, not just the model score, for sizing and order type.
|
||||
3. `get_news` with `symbols=<ticker>`, `limit=20`, `include_content=true` → assign a sentiment score **−3 (strongly negative) … +3 (strongly positive)**.
|
||||
|
||||
**Sentiment gate:** if sentiment strongly contradicts the signal — a **BUY** with sentiment ≤ −2 or a **SELL** with sentiment ≥ +2 — **cancel** that order and record it as `cancelled: sentiment conflict`. Tickers with no news or neutral sentiment (−1..+1) trade normally.
|
||||
|
||||
Record the evidence per candidate into the round book (so the reconcile step can explain every skip):
|
||||
|
||||
```
|
||||
fact_record {round_id:<ROUND_ID>, kind:"quote", symbol:<ticker>, payload:{bid, ask, last, spread_bps}, source:"get_stock_snapshot"}
|
||||
fact_record {round_id:<ROUND_ID>, kind:"news_sentiment", symbol:<ticker>, payload:{"sentiment":<−3..+3>, "headline":<top headline>}, source:"get_news"}
|
||||
```
|
||||
|
||||
## Step 8 — Place orders on Alpaca
|
||||
|
||||
For each surviving order in the Step 6 list (respecting the gate): call the `tac-engine` `place_order` tool with the ticker, side, qty and order type. Then verify with `list_orders` / `list_positions` that the intended changes went through.
|
||||
|
||||
**Record every decision in the round book** — placed orders AND deliberate skips, each with its reason (this is what the reconcile / funnel view reads):
|
||||
|
||||
```
|
||||
# each placed order (order id from the place_order response):
|
||||
decision_record {round_id:<ROUND_ID>, symbol:<ticker>, side:<buy|sell>, qty:<qty>, order_type:<type>,
|
||||
expected_price:<last quote px>, status:"placed", reason:"placed",
|
||||
intent_id:<intent id from intent_set>, alpaca_order_id:<alpaca order id>, client_order_id:<cl id>}
|
||||
# each gate cancel / skip (delta≈0, no quote, illiquid, halt, bp cap, sentiment conflict, no score, risk limit):
|
||||
decision_record {round_id:<ROUND_ID>, symbol:<ticker>, side:<side>, qty:<qty>, status:"skipped",
|
||||
reason:"sentiment_conflict"|"illiquid"|"no_quote"|"halt"|"delta_zero"|"bp_cap"|"no_score"|"risk_limit",
|
||||
reason_detail:<short why>, intent_id:<intent id>}
|
||||
```
|
||||
|
||||
**Sync fills** — pull Alpaca's order state into the round (pass the `list_orders` output as `orders` so no API call is needed; unmatched orders are reported back):
|
||||
|
||||
```
|
||||
round_sync_fills {round_id:<ROUND_ID>, orders:[{id, client_order_id, symbol, side, qty, filled_qty, filled_avg_price, status}...]}
|
||||
```
|
||||
|
||||
## Step 9 — Evidence check, close the traced experiment + summarize
|
||||
|
||||
**Evidence gate — run this BEFORE committing/closing. Do not skip, do not "summarize only".** Query the round and confirm every evidence kind is present; record anything missing right now, then re-query:
|
||||
|
||||
```
|
||||
fact_query {round_id:<ROUND_ID>} # or per-kind: fact_query {round_id:<ROUND_ID>, kind:"<kind>"}
|
||||
```
|
||||
|
||||
For each of the per-universe kinds (`signal_score`, `market_snapshot`, `quote`, `news_sentiment`, `symbol_features`, `decision_justification`) count that you recorded one per symbol you processed; `strategy_config`, `account_state`, `position_state` once each. If any kind is missing or any target symbol is missing from a kind, **go back and `fact_record` it now** (use the persist→read recipe in Step 6 for `symbol_features`). Only when every kind above is present, proceed:
|
||||
|
||||
1. Commit the run artifacts to the experiment branch: `rd_trace_commit experiment_id=<EXPERIMENT_ID> message="algo run <D>: orders placed"`.
|
||||
2. Close the lineage — re-embeds the rational/details, records metrics/evaluation, commits + pushes:
|
||||
```
|
||||
rd_trace_finish experiment_id=<EXPERIMENT_ID> \
|
||||
ref_id=<new training run_id from Step 4> \
|
||||
evaluation="<outcome of today's trade: target vs placed, cancellations>" \
|
||||
metrics='{"n_buys":N,"n_sells":M,"n_cancelled":K}' \
|
||||
mlruns_dir=<lake>/mlruns/<exp_id>/<run_id>
|
||||
```
|
||||
3. **Settle the round** — reconcile and close the window:
|
||||
```
|
||||
book_reconcile {round_id:<ROUND_ID>} # residual vs target, per-symbol reasons
|
||||
trail_funnel {round_id:<ROUND_ID>} # targets -> decided -> placed -> filled, skips by reason
|
||||
round_update_status {round_id:<ROUND_ID>, status:"settled", summary_metrics:{...funnel + invested...}}
|
||||
```
|
||||
4. **Close the loop** — record the round's execution economics for the next run's tuning:
|
||||
- `round_metrics` → the round's invested notional, turnover, slippage bps, estimated cost, cost-as-% of gross (the `fetchPriorRoundFeedback` in the scheduler injects these into the NEXT run's prompt automatically).
|
||||
- If the round had fills, run `rd_factor_attribution` over the round window (pass the `get_portfolio_history` equity curve as `portfolio_equity`, benchmark e.g. `IVV`, realized slippage+cost bps from `round_metrics`, expected values from the calibration) and record the result:
|
||||
```
|
||||
fact_record {round_id:<ROUND_ID>, kind:"attribution", payload:{beta, alpha_annualized_pct, pnl_beta, pnl_alpha, drift_alarm}, source:"rd_factor_attribution"}
|
||||
```
|
||||
- A `drift_alarm` in the attribution means live execution cost is deviating from the backtest assumption — re-run `rd_risk_calibrate` before the next round and tighten sizing/limits.
|
||||
5. End your reply with the compact summary: date `D`, reference run (`experiment_name` / `run_id`), new training run (`run_id` / `model_path`), window (4y → `D`), number of scores, top names, per-ticker sentiment scores, what was bought/sold, which orders were cancelled by the sentiment gate (and why), and any skipped trades (with reasons).
|
||||
|
||||
## Example
|
||||
|
||||
```
|
||||
experiment_name=tac-rd run_id=<ref-uuid> strategy=tune_run1_wider_5d.yaml
|
||||
1. get_lake_coverage {US,1d} -> last loaded date; backfill_lake_calendar; get_lake_bars lazy -> lake current -> D
|
||||
2. rd_exp_input run_id=<ref-uuid> -> universe=all, features=KR..(ta fields), label=Ref($close,-2)/Ref($close,-1)-1, lr=0.05, leaves=15 ... topk/n_drop from the run's backtest config
|
||||
3. EXP_NEW=<ref exp>-<epoch seconds> # unique per run (scheduler names it)
|
||||
rd_trace_init && rd_trace_start experiment_name=$EXP_NEW evolved_from=auto -> experiment_id / branch
|
||||
# scheduler usually pre-creates the round (ROUND_ID in the instructions) -> skip round_create, use it
|
||||
round_create {target_date:<D>, signal_date:<D>, source:"scheduled", rd_experiment_id:<EXPERIMENT_ID>, experiment_name:$EXP_NEW} -> ROUND_ID
|
||||
4. rd_train experiment_name=$EXP_NEW train_start=<D-4y> train_end=<D> record_analysis=false wait=false out_dir=tac-algo-output
|
||||
# -> returns immediately; poll rd_exp_get_run until status FINISHED -> run_id <new-uuid>, model_path tac-algo-output/params.pkl
|
||||
round_update {round_id:<ROUND_ID>, run_id:<new-uuid>, model_path:"tac-algo-output/params.pkl"}
|
||||
5. rd_predict model_path=tac-algo-output/params.pkl test_start=<D> test_end=<D>
|
||||
# -> pred_path tac-algo-output/pred.pkl, score head ...
|
||||
fact_record {kind:"signal_score", symbol:<ticker>, payload:{pred, rank}} per top name
|
||||
6. get_account + list_positions # current portfolio as strategy input; account=live equity
|
||||
get_stock_latest_quotes -> prices JSON for sizing
|
||||
rd_strategy_targets pred_path=tac-algo-output/pred.pkl signal_date=<D> \
|
||||
topk=<from run config> n_drop=<from run config> risk_degree=<from run config> account=<live equity> prices='{...}' \
|
||||
risk_limits='{"liquidity_floor_adv":5000000,"size_cap_pct":8,"concentration_cap_pct":30}' equity=<equity> peak_equity=<peak>
|
||||
# -> deterministic target buys (symbol/rank/score/notional/qty); delta vs list_positions -> order list (fresh account = all buys)
|
||||
fact_record {kind:"risk_check", payload:{risk_limits:{...}, applied:{...risk_limits_applied...}, equity, peak_equity}}
|
||||
round_update {round_id:<ROUND_ID>, account_equity_at_sizing:<equity>, strategy_snapshot:{topk, n_drop, risk_degree, costs, benchmark, risk_limits:{...}}}
|
||||
intent_set {round_id:<ROUND_ID>, target_portfolio:[{symbol, side, qty, expected_price, score, rank}]} -> intent_id
|
||||
7. get_news per ticker -> sentiment gate; fact_record quote + news_sentiment per ticker
|
||||
8. place_order ... per surviving delta; decision_record per placed + skipped (with reason)
|
||||
round_sync_fills {round_id:<ROUND_ID>, orders:[...list_orders output...]}
|
||||
9. rd_trace_commit experiment_id=<EXPERIMENT_ID> + rd_trace_finish experiment_id=<EXPERIMENT_ID> ref_id=<new-uuid>
|
||||
book_reconcile + trail_funnel; round_update_status {status:"settled", summary_metrics:{...}}; summary
|
||||
```
|
||||
|
||||
## Round book — the execution trail
|
||||
|
||||
Every scheduled run writes its decision→fill trail to Postgres via the `tac-rd-book`
|
||||
tools, mirroring the `/dashboard/rounds` UI. The round is the link between the scheduler
|
||||
run, the traced experiment, and the actual account activity:
|
||||
|
||||
```
|
||||
scheduler_runs ──► ROUND ──► rd_experiments
|
||||
│ fact_events evidence: signal_score / market_snapshot / quote / news_sentiment / account_state / position_state / symbol_features / decision_justification / risk_check
|
||||
│ round_intents versioned target portfolios (new version supersedes old)
|
||||
│ round_decisions per-symbol: placed OR skipped, each with a reason (incl. risk_limit)
|
||||
└──► round_orders execution rows (Alpaca order id + fills), synced via round_sync_fills
|
||||
```
|
||||
|
||||
`book_reconcile` returns the per-symbol residual (target qty − filled qty, with the reason
|
||||
it did not fill) plus cash/BP impact, slippage bps and estimated cost — that is the answer
|
||||
to "why is the account not at the target portfolio". `round_metrics` reports the round
|
||||
roll-ups (invested notional, turnover, slippage bps, estimated cost, cost-as-% of gross);
|
||||
`trail_funnel` gives the counts (targets → decided → placed → filled, skips by reason).
|
||||
All surface unchanged in the UI. When a round fires `drawdown_pause_pct`, its `round_metrics`
|
||||
will show `invested_notional: 0` — that is the pause working, not a broken round.
|
||||
@@ -0,0 +1,529 @@
|
||||
---
|
||||
name: tac-qlib-custom
|
||||
description: "Guide agents to customize and extend Qlib on the TradeAC R&D stack — how to configure workflow YAMLs (qlib_init, model, dataset/handler, processors, records, PortAnaRecord strategies), how to extend Qlib classes wired into those workflows (custom Model, BaseStrategy, DataHandler, Record), and the empirically-tested knobs from this repo (RankIC early-stopping, stochastic-control strategies, stochastic-process features, catch22/GARCH/Hurst/signature). Also encodes the experiment traceability loop: every backtest runs as a workflow-with-recorder, is recorded in the Postgres experiments table (rationale/details/evaluation/metrics with pgvector embeddings, evolution chain) and on a per-experiment git branch that is committed + pushed. Companion to tradeac-rd (MCP run tools) and tradeac-lake (parquet lake)."
|
||||
---
|
||||
|
||||
# tac-qlib-custom
|
||||
|
||||
Customizing and extending Qlib on the TradeAC stack. This skill encodes what was
|
||||
learned from actual experiments in this repo: how a workflow YAML maps to Qlib
|
||||
classes, how to write a custom class that the YAML can load, and which training /
|
||||
strategy / feature knobs measurably moved IC, RankIC and the backtest.
|
||||
|
||||
Read `tac-qlib/skills/tradeac-rd/SKILL.md` for the MCP run/inspect tools and
|
||||
`tac-qlib/README.md` for the package layout. The venv is `/app/.venv`
|
||||
(qlib 0.1.dev2066); `tac_qlib` is installed into the venv's `site-packages`
|
||||
(editable copy under `/opt/venv/.../tac_qlib/`), so **any new module must be
|
||||
copied to `/opt/venv/lib/python3.12/site-packages/tac_qlib/...` too** (or use an
|
||||
editable install) before `rd_run_workflow` can import it.
|
||||
|
||||
## MCP-first policy
|
||||
|
||||
- **Drive every backtest and run through the `tac-qlib-rd` MCP tools** (`rd_run_workflow`,
|
||||
`rd_train`, `rd_predict`, `rd_exp_*`) and the tac-engine lake tools for data prep. Do not
|
||||
reimplement them with ad-hoc scripts (custom qlib glue, own mlruns readers, direct
|
||||
JSON-RPC/stdio clients).
|
||||
- **NEVER script directly against the MCP server** (spawning `tac_qlib.rd_server` /
|
||||
`tac-engine`, bash/curl/stdio) unless a tool genuinely can't do the job — then **stop and
|
||||
ask the user to confirm first**.
|
||||
- The traceability bookkeeping (Postgres `rd_experiments` row + pgvector embeddings +
|
||||
branch-per-experiment git) is exposed as the **`rd_trace_*` MCP tools** on the tac-qlib-rd
|
||||
server — use those, not bash scripts. Data prep, training, evaluation and backtests also go
|
||||
through MCP tools.
|
||||
- If the venv is missing a runtime dep (`duckdb`, `pyarrow`, feature libs), lazy-install it
|
||||
(`uv pip install --python $VIRTUAL_ENV/bin/python <pkg>`) instead of switching tools.
|
||||
|
||||
## Secrets policy
|
||||
|
||||
- NEVER write secrets into files: DB passwords, API keys, OAuth tokens, or
|
||||
credential-bearing URLs (`DATABASE_URL`, `GIT_PASS`, `EMBEDDING_API_KEY`) in
|
||||
workflow YAMLs, scripts, configs, notes or committed code.
|
||||
- NEVER read `*.env` / `.env.*` directly (`cat`/`tail`/`grep`/`sed`/`head` on
|
||||
`.env`). That pulls secrets into this session and leaks them to any agent
|
||||
sharing it.
|
||||
- When a tool or command needs an env var, ASK the user to set it in the
|
||||
environment (shell/container env, or the user-owned `.env`) and reference it
|
||||
by name (`$VAR`), never by value. If it's missing, report which variable is
|
||||
required instead of reading it yourself.
|
||||
- Tracking store: use `uri: "sqlite:///mlruns.db"` (relative) in workflows —
|
||||
`rd_run_workflow` normalizes it to Postgres when `$DATABASE_URL` is set, else
|
||||
the lake sqlite. Never hardcode a `postgres://user:pass@…` URI.
|
||||
- If you find a committed secret, flag it, remove it, and replace it with a
|
||||
placeholder. (The `rd_trace_*` MCP tools' commit guard blocks adding
|
||||
credential-shaped lines.)
|
||||
|
||||
## How a workflow YAML maps to Qlib classes
|
||||
|
||||
A workflow YAML (`tac-qlib/workflows/*.yaml`) is rendered by Jinja (vars like
|
||||
`{{ LAKE }}` from `TAC_LAKE_DIR`) then executed by `qrun` / `rd_run_workflow`.
|
||||
Every block is a Qlib class reference resolved by `module_path` + `class`:
|
||||
|
||||
```yaml
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
calendar_provider: # custom tac-qlib providers read the parquet lake
|
||||
class: LakeCalendarProvider
|
||||
module_path: tac_qlib.data.providers
|
||||
instrument_provider: # ... (markets: {} => lake universe)
|
||||
feature_provider: # LakeFeatureProvider: routes $open..$volume from bars,
|
||||
class: LakeFeatureProvider # $<ta-lib/sp_*> from features parquet, $amount derived
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs: { uri: "sqlite:///{{ LAKE }}/mlruns.db", default_exp_name: "my-exp" }
|
||||
|
||||
task:
|
||||
model: # <MODEL BLOCK> — custom model → new module_path
|
||||
class: RankICLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_gbdt
|
||||
kwargs: { loss: mse, learning_rate: 0.02, num_leaves: 31, ... }
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler: # <HANDLER BLOCK> — feature selection + processors live here
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: "SPY,QQQ,..."
|
||||
start_time: 2015-01-03
|
||||
end_time: 2026-08-10
|
||||
fit_start_time: 2015-01-03 # processors fit on this window
|
||||
fit_end_time: 2025-09-01
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1" # 5d forward return
|
||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_ou_zscore,..."
|
||||
infer_processors: # feature-time transforms, fit on fit_*
|
||||
- { class: DropAllNaN, kwargs: {} }
|
||||
- { class: ProcessInf, kwargs: {} }
|
||||
- { class: CSRankNorm, kwargs: {} } # per-day cross-sectional rank
|
||||
- { class: ZScoreNorm, kwargs: {} }
|
||||
- { class: Fillna, kwargs: {} }
|
||||
segments:
|
||||
train: [2015-01-03, 2025-09-01]
|
||||
valid: [2025-09-03, 2026-01-03]
|
||||
test: [2026-01-04, 2026-08-10]
|
||||
record: # each entry records one artifact type to the run
|
||||
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
|
||||
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp,
|
||||
kwargs: { ana_long_short: true, ann_scaler: 252 } }
|
||||
- { class: PortAnaRecord, module_path: qlib.workflow.record_temp,
|
||||
kwargs: { config: { strategy: <STRATEGY BLOCK>, backtest: {...} }, risk_analysis_freq: 1d } }
|
||||
```
|
||||
|
||||
`rd_run_workflow config_path=<yaml> experiment_name=<exp>` runs it; the MCP call
|
||||
may time out for long runs (RankIC tuning, heavy feature sets) — the run keeps
|
||||
executing; poll via `rd_exp_list` / `rd_exp_get_run` on the returned experiment.
|
||||
|
||||
## Experiment traceability (DB + git + embeddings)
|
||||
|
||||
Every backtest you run as an agent MUST be tracked: it runs as a workflow with the
|
||||
`record` block (SignalRecord/SigAnaRecord/PortAnaRecord → MLflow artifacts on disk
|
||||
under `<lake>/mlruns/<exp_id>/<run_id>`), and a row is written to the Postgres
|
||||
`experiments` table plus a git branch per experiment. The `tac-app` UI owns the
|
||||
schema (Drizzle migrations in `tac-app/drizzle/`); this skill's `lib/` scripts are
|
||||
the executor the agent drives.
|
||||
|
||||
**Trigger the lineage as part of the run — automatically, not on prompt.** Any
|
||||
time you execute a qlib workflow (`rd_run_workflow`) or a train/predict pipeline
|
||||
on this stack, the traceability bookkeeping is part of that run, not a separate
|
||||
step the user must ask for: open the traced experiment with `rd_trace_start`
|
||||
before running, commit intermediates with `rd_trace_commit`, and close it with
|
||||
`rd_trace_finish` after — without waiting to be prompted (see "The
|
||||
per-experiment procedure" below).
|
||||
|
||||
### Env vars
|
||||
|
||||
| Var | Purpose |
|
||||
|-----|---------|
|
||||
| `DATABASE_URL` | Postgres URL for the `rd_experiments` table AND the MLflow tracking store (set in repo `.env`) |
|
||||
| `EMBEDDING_API_BASE_URL` | embedding POST endpoint (e.g. `https://embd.h.lizhao.net/embeddings`) |
|
||||
| `EMBEDDING_API_KEY` | basic-auth credential (`user:pass` form is supported) |
|
||||
| `GIT_USER` / `GIT_PASS` | git remote credentials for push/fetch |
|
||||
| `GIT_REPO_URL` | experiment git repo tracked by the `experiments` submodule (branches are pushed here) |
|
||||
| `TAC_LAKE_DIR` | lake root (mlruns artifact files live under it) |
|
||||
|
||||
The experiment repo is the **`experiments` git submodule** at the workspace root
|
||||
(`<repo-root>/experiments`), always tracking `$GIT_REPO_URL`. `rd_trace_init`
|
||||
creates/validates it; it errors if `experiments/` exists but points at a
|
||||
different URL. There is no `TAC_EXP_GIT_DIR` — the submodule path IS the
|
||||
experiment repo, and ALL experiment/backtest changes (workflow YAMLs, notes,
|
||||
outputs) must live inside it, never in the parent tradeac repo.
|
||||
|
||||
### The `rd_experiments` table
|
||||
|
||||
Owned by tac-app's Drizzle schema (`tac-app/src/db/schema.ts`); `rd_trace_init`
|
||||
can `init` it idempotently. The table is named **`rd_experiments`** (NOT
|
||||
`experiments`) because MLflow's Postgres tracking store creates its own
|
||||
`experiments` table in the same database. Key columns: `id` (PK), `rational` +
|
||||
`rational_embedding` (pgvector `vector(384)`), `details` + `details_embedding`,
|
||||
`evaluation`, `metrics` (jsonb), `evolved_from` (FK → rd_experiments.id),
|
||||
`start_ts`/`end_ts`, `git_branch`, `experiment_ref_id`, `mlruns_dir`, `status`.
|
||||
|
||||
`experiment_ref_id` holds the **mlflow run id** returned by `rd_run_workflow` and
|
||||
is an FK to MLflow's `runs(run_uuid)` (added by `rd_trace_init` after the
|
||||
mlflow store tables exist — MLflow creates `runs` lazily).
|
||||
|
||||
Tracking store: **Postgres `$DATABASE_URL`** (MLflow's own tables) when set,
|
||||
falling back to the unified lake sqlite `sqlite:///<lake>/mlruns.db`. Artifact
|
||||
files always stay on disk under `<lake>/mlruns/<exp_id>/<run_id>/artifacts`.
|
||||
|
||||
Embedding model: `michaelfeil/bge-small-en-v1.5` (384-dim, **512-token context**).
|
||||
Rational/details are written paper-summary style (≤512 tokens) and embedded verbatim —
|
||||
NEVER truncate; if a text is longer, summarize it first (the embed helper rejects
|
||||
over-limit input).
|
||||
|
||||
### Git repo + branch-per-experiment
|
||||
|
||||
The experiment repo is the `experiments` submodule at the workspace root
|
||||
(`<repo-root>/experiments`, tracking `$GIT_REPO_URL`). The `rd_trace_*` MCP
|
||||
tools handle it, and every git operation is scoped to that submodule —
|
||||
experiments NEVER stage or push parent-repo (tradeac) files.
|
||||
|
||||
- `rd_trace_init` creates/validates the submodule and the base branch. If
|
||||
`experiments/` does not exist it runs `git clone $GIT_REPO_URL experiments`;
|
||||
if it exists but tracks a different URL, init errors out.
|
||||
- Base branch: `main` (or `master`). If the submodule is empty, a seed commit is
|
||||
made and pushed so there are commits to fork from.
|
||||
- Every experiment runs on its own branch `exp/<id>-<slug>`.
|
||||
- `evolved_from` resolution (in order):
|
||||
1. If the wizard prompt explicitly says `evolved_from=<id>` (run wizard click on an
|
||||
existing experiment) — use that id directly.
|
||||
2. Otherwise `--evolved-from auto`: the user prompt / rational is embedded and
|
||||
cosine-searched over the `experiments.rational_embedding` column; the top hit
|
||||
above the similarity threshold (0.5) becomes `evolved_from`.
|
||||
3. Otherwise (first experiment, or a new chat with no predecessor) — no evolved_from;
|
||||
fork from `main`'s latest commits.
|
||||
- The new branch is forked from the **evolved-from experiment's branch** (its latest
|
||||
commits), or from `main` when there is no predecessor — so experiment lineages form
|
||||
a git branch chain.
|
||||
- On every finish, and for intermediate steps, changes are committed + pushed.
|
||||
|
||||
### Custom code is part of the lineage (code snapshot)
|
||||
|
||||
Custom contrib modules (`tac_qlib/contrib/model/`, `tac_qlib/contrib/strategy/`,
|
||||
`tac_qlib/contrib/data/`, `tac_qlib/data/providers.py`) live in the **parent**
|
||||
tradeac repo, not in the `experiments/` submodule — so they are normally invisible
|
||||
to the experiment branch and a descendant forking from it would reinvent them.
|
||||
The lineage tooling fixes this: **every experiment branch carries a `code/`
|
||||
snapshot of exactly the qlib extension code that run depended on**, so descendants
|
||||
reuse it instead of re-authoring it.
|
||||
|
||||
- `rd_trace_start` and `rd_trace_finish` automatically snapshot the default paths
|
||||
(`tac-qlib/tac_qlib/contrib`, `tac-qlib/tac_qlib/data`) into
|
||||
`<experiments>/code/<parent-relative-path>` on the experiment branch.
|
||||
- `rd_trace_snapshot` snapshots mid-run (e.g. after writing a
|
||||
new custom model) without waiting for finish.
|
||||
- The snapshot also writes `code/MANIFEST.txt` recording the **parent-repo HEAD
|
||||
commit** and the per-file blob hashes it was taken from — so a run can be traced
|
||||
back to the exact parent commit that produced its custom code.
|
||||
- Descendants: the custom modules your run needs are under `code/tac_qlib/...` on the
|
||||
evolved-from branch. Reuse them (copy/`git show`) instead of writing new ones; check
|
||||
`code/MANIFEST.txt` to see which parent commit they came from and port fixes back.
|
||||
- Guardrail exception: parent-repo changes under `tac_qlib/tac_qlib/contrib` and
|
||||
`tac_qlib/tac_qlib/data` are **expected** (they are the snapshotted code);
|
||||
`parent_changes` reports them as a note, not a violation. Any OTHER parent change
|
||||
is still a guardrail violation.
|
||||
|
||||
Guardrail — experiments must NOT introduce side effects to the parent repo:
|
||||
- Write workflow YAMLs, notes and experiment outputs ONLY inside
|
||||
`<repo-root>/experiments/` (they are committed on the experiment branch).
|
||||
- Never `git add`/commit/stage anything in the parent tradeac repo.
|
||||
- Run `rd_trace_guard` to list any parent
|
||||
changes outside the submodule pointer; `rd_trace_finish` also surfaces them.
|
||||
Revert any accidental parent edits before finishing.
|
||||
- If an experiment reveals a PRODUCT change (workflow template, skill, tac-app),
|
||||
propose it separately for the tradeac repo — do not mix it into the experiment
|
||||
branch.
|
||||
|
||||
The `rd_trace_*` MCP tools perform git operations with the mandated credential
|
||||
helper (from `GIT_USER` / `GIT_PASS`), so you do not need to construct it by hand.
|
||||
|
||||
### The per-experiment procedure
|
||||
|
||||
**Use the `rd_trace_*` MCP tools (tac-qlib-rd)** — they replace the old
|
||||
`trace.sh`/`trace_db.py` scripts. The server is long-lived (psycopg imported
|
||||
once, DB connection reused per call) and every tool returns one JSON object, so
|
||||
no output parsing is needed:
|
||||
|
||||
```text
|
||||
# 0. ensure ready (rd_experiments table + experiments git repo + base main)
|
||||
rd_trace_init
|
||||
|
||||
# 1. start — inserts the row, resolves evolved_from, forks+pushes the branch.
|
||||
# Returns {experiment_id, branch, evolved_from, base_branch} as JSON.
|
||||
rd_trace_start rational="5-day forward label, RankIC early stop, 50-ETF universe" \
|
||||
details="LGBModel mse lr=0.02 num_leaves=15 num_boost_round=3000; TopkDropout topk=2; benchmark QQQ" \
|
||||
experiment_name="tac-rd-expN" \
|
||||
evolved_from="auto" \
|
||||
session_id="<this chat's opencode session id, if started from a chat>"
|
||||
# -> {"experiment_id": N, "branch": "exp/N-...", "evolved_from": ..., "base_branch": ...}
|
||||
|
||||
# 2. write the workflow YAML INSIDE the experiments submodule
|
||||
# (e.g. <repo-root>/experiments/workflows/<exp>/workflow.yaml), then commit it:
|
||||
rd_trace_commit experiment_id=<N> message="add workflow yaml"
|
||||
|
||||
# 2b. if the workflow uses a NEW custom module, snapshot it onto the branch
|
||||
# (start/finish auto-snapshot contrib+data; do this to capture mid-run):
|
||||
rd_trace_snapshot experiment_id=<N> # default contrib+data
|
||||
# or: rd_trace_snapshot experiment_id=<N> paths="tac-qlib/tac_qlib/contrib/model/rank_gbdt.py"
|
||||
|
||||
# 3. run the backtest through the WORKFLOW with the recorder (MUST write mlruns):
|
||||
rd_run_workflow config_path=<repo-root>/experiments/workflows/<exp>/workflow.yaml experiment_name=tac-rd-expN
|
||||
# -> returns run_id (= experiment_ref_id) + metrics
|
||||
|
||||
# 4. inspect with rd_exp_result / rd_exp_blotter, then finish — updates the row
|
||||
# (re-embeds rational/details, sets metrics/eval/end_ts), snapshots the custom
|
||||
# code, and commits+pushes. finish also surfaces parent-repo side effects.
|
||||
rd_trace_finish experiment_id=<N> \
|
||||
ref_id=<mlflow-run-id> \
|
||||
evaluation="IC 0.0645, RankIC 0.075; net excess +0.85% ann" \
|
||||
metrics='{"IC":0.0645,"RankIC":0.075,"ann_excess":0.85}' \
|
||||
mlruns_dir=<lake>/mlruns/<exp_id>/<run_id>
|
||||
```
|
||||
|
||||
Helpers (MCP tools): `rd_trace_search` (semantic), `rd_trace_get` (one row),
|
||||
`rd_trace_list`, `rd_trace_mlruns_dir` (resolves the mlruns dir for an
|
||||
experiment name), `rd_trace_guard` (parent-repo side-effect check).
|
||||
|
||||
Rules:
|
||||
- **Always** run backtests as workflows with the `record` block (req 2) — never a bare
|
||||
`rd_backtest` for a traced experiment.
|
||||
- **Always** open the lineage (`rd_trace_start`) BEFORE the run and **Always**
|
||||
`rd_trace_finish` + push after it completes (req 5) — this happens as part of the run,
|
||||
do not wait for the user to ask; intermediate `rd_trace_commit` is encouraged (req 5).
|
||||
- **Always** snapshot the custom qlib code (`rd_trace_snapshot`, or rely on the
|
||||
auto-snapshot at start/finish) so the experiment branch carries the exact contrib/data
|
||||
modules the run used — descendants fork and reuse `code/` instead of reinventing it.
|
||||
- Keep rational/details ≤ 512 tokens (paper-summary style) so embeddings are exact —
|
||||
no truncation.
|
||||
- **Confine experiments to the `experiments/` submodule** — never write to, stage, or
|
||||
commit parent tradeac repo files; run `rd_trace_guard` to check for side effects.
|
||||
(Custom code edits under `tac-qlib/tac_qlib/contrib` and `.../data` are the sanctioned
|
||||
exception — they are the snapshotted modules; see "Custom code is part of the lineage".)
|
||||
- **Follow the Secrets policy above** — no secrets in files, no reading `.env*`, ask the
|
||||
user to set env vars; use `uri: "sqlite:///mlruns.db"` for the tracking store.
|
||||
- Workflow YAMLs are jinja-rendered with `os.environ` as the context, so env-var
|
||||
placeholders work (`{%- set LAKE = TAC_LAKE_DIR %}` then `{{ LAKE }}`). Use them for
|
||||
paths/config — never for secrets that get committed.
|
||||
|
||||
## Extending Qlib — the 4 class families you can override
|
||||
|
||||
### 1. Custom Model (train-time) — `tac_qlib/contrib/model/`
|
||||
Subclass `qlib.contrib.model.gbdt.LGBModel` (or `qlib.model.base.BaseModel`) and
|
||||
implement `fit(dataset, ...)` + `predict(dataset)`. `LGBModel.fit` calls
|
||||
`self._prepare_data(dataset)` → `lgb.Dataset`s, then `lgb.train` with
|
||||
`early_stopping` on the valid set. Override points that matter:
|
||||
|
||||
- `_prepare_data` → build the `lgb.Dataset` with `group=` (per-day query groups)
|
||||
when you need ranking metrics per trading day.
|
||||
- `fit` → change what early-stops training (the biggest IC/backtest lever, see §Knobs).
|
||||
- `predict` → return the Series keyed (datetime, instrument).
|
||||
|
||||
Reference: `tac_qlib/tac_qlib/contrib/model/rank_gbdt.py` — `RankICLGBModel`
|
||||
subclasses `LGBModel`, adds per-day `group` in `_prepare_data`, injects
|
||||
`feval=rankic_feval` (mean per-day Spearman) into `lgb.train`, and forces
|
||||
`metric='None'` + `first_metric_only=True` so early-stopping tracks RankIC only.
|
||||
|
||||
### 2. Custom Strategy (backtest-time) — `tac_qlib/contrib/strategy/`
|
||||
Subclass `qlib.contrib.strategy.signal_strategy.BaseSignalStrategy` (which wraps
|
||||
`qlib.strategy.base.BaseStrategy`) and implement:
|
||||
|
||||
```python
|
||||
def generate_trade_decision(self, execute_result=None):
|
||||
# trade_step, trade_start/end = self.trade_calendar.get_step_time(trade_step)
|
||||
# pred = self.signal.get_signal(start_time=pred_shift, end_time=pred_shift) # shift=-1 => signal known at t-1
|
||||
# self.trade_position / self.trade_exchange / self.trade_calendar injected by the executor
|
||||
# build qlib.backtest.Order(stock_id, amount, start_time, end_time, direction=Order.BUY/SELL)
|
||||
# return TradeDecisionWO(orders, self)
|
||||
```
|
||||
|
||||
Wire it into the YAML under `PortAnaRecord.config.strategy`:
|
||||
|
||||
```yaml
|
||||
strategy:
|
||||
class: OptimalStopControl
|
||||
module_path: tac_qlib.contrib.strategy.optimal_stop
|
||||
kwargs:
|
||||
signal: "<PRED>" # placeholder replaced with the recorded pred
|
||||
topk: 10
|
||||
entry_pct: 0.85
|
||||
exit_pct: 0.7
|
||||
max_hold_days: 10
|
||||
min_hold_days: 2
|
||||
sl: -0.08
|
||||
risk_degree: 0.95
|
||||
```
|
||||
|
||||
Reference: `tac_qlib/tac_qlib/contrib/strategy/optimal_stop.py`
|
||||
(`OptimalStopControl` — entry gated by cross-sectional signal percentile, exits
|
||||
by percentile/time/stop-loss, equal-weight control sizing).
|
||||
|
||||
### 3. Custom DataHandler / processors — `tac_qlib/contrib/data/handler.py`
|
||||
`TACHandler(DataHandlerLP)` already wraps the lake via `QlibDataLoader` +
|
||||
`LakeFeatureProvider`. Key config surface (all usable from YAML without new code):
|
||||
- `feature_fields` — explicit list; the handler prefixes `$` and de-dups. Anything
|
||||
the provider can route is usable: bar fields, `$amount` (v*vw), and any column
|
||||
present in the lake `features/.../symbol=*.parquet` files.
|
||||
- `infer_processors` / `learn_processors` — add `CSRankNorm`, `CSZScoreNorm`
|
||||
(label), `ZScoreNorm`, `DropnaLabel`, `Fillna`, etc. `DropAllNaN` is a
|
||||
tac-qlib processor (drops all-NaN columns on the fit window).
|
||||
- `label` — any qlib expression, e.g. `Ref($close,-6)/Ref($close,-1)-1`.
|
||||
|
||||
To add a *new feature family*: compute it once (see `examples/sp_features.py` +
|
||||
`examples/persist_sp_features.py`), persist extra columns into
|
||||
`features/market=US/timeframe=1d/symbol=*.parquet` (drop stale `sp_*` columns
|
||||
first on re-runs), then reference them in `feature_fields`.
|
||||
|
||||
**The Rust engine already ships the SP feature pipeline as a lake MCP tool**:
|
||||
`get_lake_sp` (tac-engine, stochastic-rs) computes `sp_ou_*`, `sp_hmm_*`,
|
||||
`sp_jump_*`, `sp_rv*`/`sp_vol_ratio_*` (+ `sp_rv_ac1`, `sp_rv_cv_22`),
|
||||
`sp_max_up`/`sp_max_down`, `sp_trend_slope_*`, `sp_logp`,
|
||||
`sp_hurst_exponent`, `sp_sig_*` (levels 1/2 at lag 1 and 5),
|
||||
`sp_rskew_*`/`sp_rkurt_*`/`sp_dsv_*` (realized moments via stochastic-rs
|
||||
`realized`) + `sp_ret` from lake bars and persists them into
|
||||
the feature parquets (replacing stale `sp_*`), all in one call:
|
||||
```json
|
||||
{"symbol": "AAPL", "timeframe": "1d", "start": "2015-01-03", "end": "2026-08-10", "fit_end": "2025-09-01"}
|
||||
```
|
||||
`fit_end` pins the Gaussian-HMM fit to the train window (no lookahead), matching
|
||||
the `FIT_END` convention. **Deferred families** (`garch`, `entropy`, `catch22`)
|
||||
are still computed with the Python `sp_features.py` path until their ports land.
|
||||
Note two deliberate differences vs the Python reference: the Rust HMM uses the
|
||||
causal *forward filter* (`filtered_state_probs`) rather than hmmlearn's smoothed
|
||||
`predict_proba`, and `hurst` is estimated on the returns series directly
|
||||
(`take_differences=false`) rather than the reference's double-differenced
|
||||
`kind="random_walk"` — regime *state* assignments agree, probability levels are
|
||||
comparable but not identical.
|
||||
|
||||
### 4. Custom Record (artifact writers)
|
||||
Subclass `qlib.workflow.record_temp.SignalRecord` / a `Record` and log metrics +
|
||||
artifacts into the MLflow run. There is no shipped example Record in `contrib/`
|
||||
yet — write one against the pattern in `qlib.workflow.record_temp` when a
|
||||
workflow needs a bespoke simulator (e.g. beta-neutral 3L/3S) that
|
||||
`PortAnaRecord` doesn't cover.
|
||||
|
||||
## Empirical knobs that moved the numbers (measured on the 50-ETF lake)
|
||||
|
||||
All experiments used: 50-ETF universe, train 2015-01-03..2025-09-01 / valid
|
||||
2025-09-03..2026-01-03 / test 2026-01-04..2026-08-10, benchmark SPY, TopkDropout
|
||||
or OptimalStopControl, costs open 0.0005 / close 0.0015 / min 5.
|
||||
|
||||
> **Rank-dimension reminder**: when the goal is to improve the *ranking* quality of
|
||||
> a signal (RankIC, long-short spread, top-decile precision), do NOT reinvent the
|
||||
> stack — use the contrib modules already shipped and verified in this repo:
|
||||
> `tac_qlib.contrib.model.rank_gbdt.RankICLGBModel` (early-stops training on
|
||||
> per-day cross-sectional RankIC, `metric='None'` + `first_metric_only`) and
|
||||
> `tac_qlib.contrib.strategy.optimal_stop.OptimalStopControl` (entry/exit gated by
|
||||
> signal percentile instead of raw levels). Both are loadable from a workflow YAML
|
||||
> via `module_path` — see the canonical `tac-qlib/workflows/workflow_lgb_sp5d_rankic.yaml`
|
||||
> (rank dimension: model) and `workflow_lgb_sp5d_optstop.yaml` (rank dimension:
|
||||
> portfolio construction). Verified end-to-end on 2026-01-04..2026-08-10:
|
||||
> RankIC 0.071 / net-of-cost excess +20.7% ann (IR 0.70) vs SPY. Only write a new
|
||||
> custom Model/Strategy when these proven paths are insufficient.
|
||||
|
||||
### Label
|
||||
- **5-day forward return `Ref($close,-6)/Ref($close,-1)-1` ≫ 2-day.** IC nearly
|
||||
tripled (0.0207 → 0.0645 standalone; the biggest single lever found). The 2-day
|
||||
target is too noisy.
|
||||
|
||||
### Features
|
||||
- **Stochastic-process features beat hand-rolled TA.** 55-feature set: OU
|
||||
(`sp_ou_*`), 2-state HMM (`sp_hmm_*`), jump intensity (`sp_jump_*`, incl.
|
||||
`sp_max_up`/`sp_max_down`), HARRV vol (`sp_rv*` + `sp_rv_ac1`/`sp_rv_cv_22`),
|
||||
trend (`sp_trend_slope_*`, `sp_logp`), GARCH (`sp_garch_*`), Hurst
|
||||
(`sp_hurst_exponent`), path signatures (`sp_sig_*`, lag 1 & 5), entropy
|
||||
(`sp_ent_*`), realized moments (`sp_rskew_*`/`sp_rkurt_*`/`sp_dsv_*`),
|
||||
catch22 (`sp_c22_*`). IC 0.036 → 0.047 vs the 19-feature v1.
|
||||
- **Do NOT add ta-lib indicators on top** (SP+TA, 74 feats): IC dropped
|
||||
0.047 → 0.031, RankIC 0.047 → 0.020. They're redundant with rv22/hmm/garch/catch22
|
||||
and dilute CSRankNorm + LGBM.
|
||||
- **CSRankNorm** (per-day cross-sectional rank) is important for the rank signal.
|
||||
- Warm-up rows persist as all-NaN feature rows — expected; DropAllNaN/DropnaLabel
|
||||
handle them.
|
||||
|
||||
### Model / training loop
|
||||
- **LambdaRank / rank_xendcg objectives FAIL here** (RankIC → ~0): with only ~50
|
||||
"documents" per query the rank gradient is noise.
|
||||
- **Early-stopping metric beats objective.** MSE objective + early-stop on a
|
||||
**RankIC feval** (mean per-day Spearman) lifted RankIC 0.047 → 0.075 (standalone).
|
||||
- **The workflow gap was qlib's training loop**: `lgb.train` default
|
||||
`first_metric_only=False` + `metric=l2` keeps training while l2 improves after
|
||||
RankIC peaks. `RankICLGBModel` sets `metric='None'` + `first_metric_only=True`
|
||||
so early-stopping tracks RankIC only.
|
||||
- **RankIC-only early stop + bigger/smaller budget is the win**: `num_boost_round
|
||||
3000`, `learning_rate 0.02`, `early_stopping_rounds 200`, `min_data_in_leaf 20`,
|
||||
`lambda_l2 0.5` → test excess **+9.1% ann w/o cost (IR 1.03, maxDD −3.8%)** and
|
||||
**+0.85% ann after costs** — the only config that beat SPY net. Note IC/RankIC
|
||||
themselves were slightly lower (0.042) than the 500-tree run (0.051); the tuned
|
||||
budget selects the iteration maximizing *valid* RankIC, converting to realized
|
||||
excess return.
|
||||
|
||||
### Strategy / portfolio construction
|
||||
- **Long-only construction leaves the edge on the table.** The SP-5d signal has
|
||||
long-short **+31.6% ann (Sharpe 2.51)**, but TopkDropout long-only ≈ flat vs SPY,
|
||||
and OptimalStopControl underperformed (valid-window threshold overfit: valid
|
||||
+7.5% → test −17.7% on one calibration).
|
||||
- **Costs eat most of the gross edge** (+9.1% → +0.85% net). Reduce turnover or go
|
||||
long-short to widen the net edge.
|
||||
- OptimalStopControl thresholds must be calibrated on the *valid* window and are
|
||||
sensitive to overfit — prefer robust defaults or penalize turnover in selection.
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **Installed package copy**: `tac_qlib` in the venv is a copy under
|
||||
`/opt/venv/lib/python3.12/site-packages/tac_qlib/`. After editing any
|
||||
`tac_qlib/contrib/**` module, `cp` it there or the workflow imports the stale
|
||||
version. New subpackages need `mkdir -p` first.
|
||||
- `qlib.backtest` exports `Order` but not `OrderDir`/`Position` at top level —
|
||||
import `Order` from `qlib.backtest`, `OrderDir`/`TradeDecisionWO` from
|
||||
`qlib.backtest.decision`, `Position` from `qlib.backtest.position`.
|
||||
- `qlib.backtest.high_performance_ds` may not export `Order` in this build — don't
|
||||
import from it.
|
||||
- HMM / GARCH / catch22 features must not see test data at fit time: fit the HMM
|
||||
on the train window only (`fit_end=FIT_END`), and compute rolling windows ending
|
||||
at each day. GARCH/entropy use a stride + forward-fill for speed (~5x).
|
||||
- `pycatch22`, `arch`, `hurst`, `antropy`, `hmmlearn` are required for the full
|
||||
feature set; install with `uv pip install --python /app/.venv/bin/python <pkg>`
|
||||
(a C compiler is needed for `pycatch22`). `duckdb` and `pyarrow` are declared in
|
||||
`tac-qlib/pyproject.toml`; if a workflow import fails on either, lazy-install with
|
||||
`uv pip install --python /app/.venv/bin/python duckdb pyarrow`.
|
||||
- `rd_run_workflow` defaults to `wait=false`: it returns immediately with
|
||||
`status: started` and the workflow runs in a background thread — poll
|
||||
`rd_exp_get_run` / `rd_exp_list` for the newest run of the experiment
|
||||
(status `RUNNING` until it finishes), then reuse its `run_id`. Pass
|
||||
`wait=true` only for small windows that finish within the MCP call timeout.
|
||||
- After fixing a YAML model/handler change, remember both `/app/tac-qlib/...` and
|
||||
the `/opt/venv` copy stay in sync.
|
||||
|
||||
## Files this skill is based on
|
||||
|
||||
Minimal, runnable examples live next to this skill in `examples/` — they are the
|
||||
canonical reference for every artifact the skill describes:
|
||||
|
||||
- Workflows (full `record` block → MLflow on disk):
|
||||
- `examples/workflow_minimal.yaml` — the canonical backtest template (req: every
|
||||
traced backtest runs through a workflow like this via `rd_run_workflow`)
|
||||
- `examples/workflow_rankic.yaml` — RankIC-early-stop model wired in
|
||||
- Repo workflows for reference: `tac-qlib/workflows/workflow_lgb_taclake.yaml`,
|
||||
`tune_run1_wider_5d.yaml`, `tune_run2_regularized.yaml`, `tune_run3_label5d_clean_universe.yaml`,
|
||||
`tune_run4_fix_universe_longtrain.yaml`, `tune_run5_longtest.yaml`
|
||||
- Models: `examples/model_rank_gbdt.py` (`RankICLGBModel`: per-day groups +
|
||||
`feval=rankic` + `metric='None'`). Repo: `tac_qlib/contrib/model/rank_gbdt.py`
|
||||
- Strategies: `examples/strategy_optimal_stop.py` (`OptimalStopControl`),
|
||||
`examples/strategy_beta_neutral.py` (doc-only 3L/3S stub — pattern for a
|
||||
custom strategy + Record; not wired into the package)
|
||||
- Handler: `examples/handler.py` (how to subclass `TACHandler`); repo:
|
||||
`tac_qlib/contrib/data/handler.py`; providers: `tac_qlib/data/providers.py`
|
||||
- Feature engineering: `examples/sp_features.py` (OU + Hurst) and
|
||||
`examples/persist_sp_features.py` (persist `sp_*` into the lake features parquet)
|
||||
- Ranking experiments: `examples/run_rank_objectives.py` (mse vs lambdarank vs
|
||||
rank_xendcg ablation on the lake)
|
||||
- Optstop calibration: `examples/run_optstop_compare.py` (valid-window grid +
|
||||
overfit warning)
|
||||
- Traceability tooling: the `rd_trace_*` MCP tools (tac-qlib-rd,
|
||||
`tac_qlib/trace.py`) — see the traceability section above
|
||||
@@ -0,0 +1,70 @@
|
||||
"""Minimal custom DataHandler — how to extend TACHandler for a new feature family.
|
||||
|
||||
`TACHandler(DataHandlerLP)` already routes lake bars + ta-lib features via
|
||||
`LakeFeatureProvider` (see tac_qlib/contrib/data/handler.py). To add a NEW
|
||||
feature family (computed once, persisted into the lake features parquet — see
|
||||
examples/persist_sp_features.py), you only need to:
|
||||
|
||||
1. persist extra columns into features/market=US/timeframe=1d/symbol=*.parquet
|
||||
2. list them in `feature_fields` (they are prefixed with `$` and de-duped)
|
||||
|
||||
A subclass is only needed when the feature must be computed *inside* the qlib
|
||||
pipeline (e.g. as an extra processor). This file sketches that pattern.
|
||||
|
||||
Reference handler structure (from tac_qlib/contrib/data/handler.py):
|
||||
|
||||
class TACHandler(DataHandlerLP):
|
||||
def __init__(self, instruments, start_time, end_time, freq,
|
||||
fit_start_time=None, fit_end_time=None,
|
||||
feature_fields=None, label=None, lake_root=None, market="US",
|
||||
infer_processors=None, learn_processors=None, **kwargs):
|
||||
loader = QlibDataLoader(configured=(feature_fields or self.DEFAULT_FIELDS), freq=freq)
|
||||
super().__init__(instruments, start_time, end_time, freq=freq,
|
||||
data_loader=loader,
|
||||
infer_processors=infer_processors or DEFAULT_INFER_PROCESSORS,
|
||||
learn_processors=learn_processors or DEFAULT_LEARN_PROCESSORS,
|
||||
fit_start_time=fit_start_time, fit_end_time=fit_end_time,
|
||||
process_type=DataHandlerLP.PTYPE_A, **kwargs)
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, List, Optional
|
||||
|
||||
from tac_qlib.contrib.data.handler import DEFAULT_INFER_PROCESSORS, DEFAULT_LEARN_PROCESSORS, TACHandler
|
||||
|
||||
|
||||
class CustomFeaturesHandler(TACHandler):
|
||||
"""TACHandler variant that also loads the lake feature columns passed in.
|
||||
|
||||
Usage from YAML — only the handler kwargs change:
|
||||
|
||||
handler:
|
||||
class: CustomFeaturesHandler
|
||||
module_path: tac_qlib.contrib.data.handler # after adding this class there
|
||||
kwargs:
|
||||
instruments: AAPL,MSFT,QQQ
|
||||
start_time: 2026-03-01
|
||||
end_time: 2026-08-06
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
feature_fields: "$close,sp_ou_alpha,sp_hurst_exponent"
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
feature_fields: Optional[List[str]] = None,
|
||||
infer_processors: Optional[List[Any]] = None,
|
||||
learn_processors: Optional[List[Any]] = None,
|
||||
**kwargs: Any,
|
||||
):
|
||||
# `feature_fields` are passed through with the leading `$` stripped by
|
||||
# TACHandler; infer/learn default to the lake-tuned processor stacks.
|
||||
super().__init__(
|
||||
feature_fields=feature_fields,
|
||||
infer_processors=infer_processors or DEFAULT_INFER_PROCESSORS,
|
||||
learn_processors=learn_processors or DEFAULT_LEARN_PROCESSORS,
|
||||
**kwargs,
|
||||
)
|
||||
@@ -0,0 +1,78 @@
|
||||
"""Minimal RankIC early-stopping LightGBM model (the biggest IC/backtest lever).
|
||||
|
||||
Drop-in replacement for `qlib.contrib.model.gbdt.LGBModel` in a workflow YAML:
|
||||
|
||||
task.model:
|
||||
class: RankICLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_gbdt
|
||||
kwargs: { loss: mse, learning_rate: 0.02, num_boost_round: 3000,
|
||||
early_stopping_rounds: 200, lambda_l2: 0.5 }
|
||||
|
||||
What it changes vs stock LGBModel:
|
||||
* `_prepare_data` builds `lgb.Dataset` with per-day `group` query groups, so
|
||||
metrics are computed per trading day.
|
||||
* `fit` injects `feval=rankic_feval` (mean per-day Spearman) into `lgb.train`
|
||||
and forces `metric='None'` + `first_metric_only=True` so early stopping
|
||||
tracks RankIC — not l2, which keeps improving after RankIC peaks.
|
||||
|
||||
Why: with ~50 instruments per day, ranking objectives (lambda_rank/xendcg)
|
||||
produce near-zero RankIC; MSE objective + RankIC early-stop is what lifts it.
|
||||
|
||||
Install: copy to tac_qlib/contrib/model/rank_gbdt.py AND the /opt/venv copy.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from qlib.contrib.model.gbdt import LGBModel
|
||||
|
||||
|
||||
def rankic_feval(preds: np.ndarray, dataset) -> tuple[str, float, bool]:
|
||||
"""Mean per-day Spearman rank IC between predictions and the label."""
|
||||
label = dataset.get_label()
|
||||
group = dataset.get_group() if hasattr(dataset, "get_group") else None
|
||||
if group is None:
|
||||
return "rankic", _spearman(preds, label), False
|
||||
|
||||
start = 0
|
||||
ics = []
|
||||
for g in group:
|
||||
sl = slice(start, start + g)
|
||||
start += g
|
||||
ics.append(_spearman(preds[sl], label[sl]))
|
||||
return "rankic", float(np.mean(ics)), False
|
||||
|
||||
|
||||
def _spearman(x: np.ndarray, y: np.ndarray) -> float:
|
||||
if len(x) < 2:
|
||||
return 0.0
|
||||
from scipy.stats import spearmanr
|
||||
|
||||
rho, _ = spearmanr(x, y)
|
||||
return float(rho) if rho == rho else 0.0
|
||||
|
||||
|
||||
class RankICLGBModel(LGBModel):
|
||||
"""LGBModel with per-day query groups and RankIC-only early stopping."""
|
||||
|
||||
def _prepare_data(self, dataset, *args, **kwargs):
|
||||
"""Attach per-day group sizes to the train/valid lgb.Dataset."""
|
||||
dtrain, dvalid = super()._prepare_data(dataset, *args, **kwargs)
|
||||
for d, index in ((dtrain, dataset.get_index_by_segment("train")), (dvalid, dataset.get_index_by_segment("valid"))):
|
||||
if d is not None and index is not None:
|
||||
# group by calendar day in order
|
||||
days = pd.Series([i[0] for i in index])
|
||||
group = days.value_counts().sort_index().tolist()
|
||||
d.set_group(np.array(group, dtype=np.int32))
|
||||
return dtrain, dvalid
|
||||
|
||||
def fit(self, dataset, evals_result: Dict[str, Any] | None = None, **kwargs):
|
||||
# force RankIC-only early stopping
|
||||
kwargs.setdefault("feval", rankic_feval)
|
||||
kwargs.setdefault("metric", "None")
|
||||
kwargs.setdefault("first_metric_only", True)
|
||||
return super().fit(dataset, evals_result=evals_result, **kwargs)
|
||||
@@ -0,0 +1,66 @@
|
||||
"""Minimal persistence of computed SP features into the lake features parquet.
|
||||
|
||||
Flow: compute sp_* features per symbol (examples/sp_features.py) and MERGE them
|
||||
into features/market=US/timeframe=1d/symbol=*.parquet so TACHandler /
|
||||
LakeFeatureProvider can route `$sp_ou_theta` etc. from the workflow YAML.
|
||||
|
||||
Run after backfilling bars; re-run drops stale sp_* columns first (see note).
|
||||
|
||||
python examples/persist_sp_features.py --market US --timeframe 1d
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import os
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from tac_qlib.data.config import LakeConfig, NON_FEATURE_COLUMNS
|
||||
from examples.sp_features import build_sp_features
|
||||
|
||||
#: columns owned by this feature family (replaced on re-runs, never duplicated)
|
||||
SP_PREFIX = "sp_"
|
||||
|
||||
|
||||
def persist_symbol(lake: LakeConfig, timeframe: str, symbol: str) -> None:
|
||||
bars_path = lake.bar_path(timeframe, symbol)
|
||||
feats_path = lake.features_path(timeframe, symbol)
|
||||
if not bars_path.exists():
|
||||
return
|
||||
bars = pd.read_parquet(bars_path)
|
||||
feats = build_sp_features(bars)
|
||||
# bars have a single 't'/'date' column; align feature rows to it
|
||||
feats = feats.drop(columns=[c for c in NON_FEATURE_COLUMNS if c in feats.columns], errors="ignore")
|
||||
|
||||
feats_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
if feats_path.exists():
|
||||
existing = pd.read_parquet(feats_path)
|
||||
# drop stale sp_* columns before merging (idempotent re-runs)
|
||||
existing = existing[[c for c in existing.columns if not c.startswith(SP_PREFIX)]]
|
||||
merged = pd.merge(existing, feats, on="t", how="left", suffixes=("", "_dup"))
|
||||
merged = merged.loc[:, ~merged.columns.str.endswith("_dup")]
|
||||
# keep original column order + new sp_* appended
|
||||
merged.to_parquet(feats_path, index=False)
|
||||
else:
|
||||
feats.to_parquet(feats_path, index=False)
|
||||
print(f"persisted {symbol}: {len(feats.columns) - 1} sp_* features")
|
||||
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--market", default="US")
|
||||
ap.add_argument("--timeframe", default="1d")
|
||||
ap.add_argument("--symbols", default="", help="comma-separated; default: all lake symbols")
|
||||
args = ap.parse_args()
|
||||
lake_root = os.environ.get("TAC_LAKE_DIR")
|
||||
if not lake_root:
|
||||
raise SystemExit("TAC_LAKE_DIR is required")
|
||||
lake = LakeConfig(lake_root, args.market)
|
||||
symbols = [s.strip().upper() for s in args.symbols.split(",") if s.strip()] or lake.load_symbols()
|
||||
for symbol in symbols:
|
||||
persist_symbol(lake, args.timeframe, symbol)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,69 @@
|
||||
"""Minimal OptimalStopControl threshold calibration — valid-window grid search.
|
||||
|
||||
This repo found OptimalStopControl thresholds overfit the valid window (valid
|
||||
+7.5% → test −17.7% on one calibration). This script runs a small grid over
|
||||
(entry_pct, exit_pct, max_hold_days) on the VALID window, reports per-config
|
||||
excess return + turnover, and warns when the best valid config is a spike.
|
||||
|
||||
Reference repo impl: tac-qlib/examples/run_optstop_compare.py.
|
||||
|
||||
python examples/run_optstop_compare.py --universe AAPL,MSFT,QQQ
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import itertools
|
||||
import os
|
||||
|
||||
import pandas as pd
|
||||
|
||||
|
||||
GRID = {
|
||||
"entry_pct": [0.7, 0.85, 0.95],
|
||||
"exit_pct": [0.5, 0.7],
|
||||
"max_hold_days": [5, 10],
|
||||
}
|
||||
|
||||
|
||||
def evaluate_config(lake_root: str, universe: list[str], window: tuple, config: dict) -> dict:
|
||||
"""Simplified stand-in: train the RankIC model, backtest OptimalStopControl
|
||||
on `window`, return (ann_excess_return, turnover, max_drawdown).
|
||||
|
||||
The real repo impl calls qlib.backtest with the strategy and reads
|
||||
report_normal.csv + risk.csv. Keep the interface here so the grid loop is
|
||||
reusable.
|
||||
"""
|
||||
# placeholder — plug in the real backtest here
|
||||
return {"ann_excess": 0.0, "turnover": 0.0, "max_dd": 0.0}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--lake-root", default=os.environ.get("TAC_LAKE_DIR", ""))
|
||||
ap.add_argument("--universe", default="AAPL,MSFT,QQQ,IVV,SMH,TLT")
|
||||
args = ap.parse_args()
|
||||
universe = [s.strip().upper() for s in args.universe.split(",")]
|
||||
valid = ("2026-06-01", "2026-06-30")
|
||||
test = ("2026-07-01", "2026-08-06")
|
||||
|
||||
keys = list(GRID)
|
||||
results = []
|
||||
for combo in itertools.product(*[GRID[k] for k in keys]):
|
||||
config = dict(zip(keys, combo))
|
||||
v = evaluate_config(args.lake_root, universe, valid, config)
|
||||
t = evaluate_config(args.lake_root, universe, test, config)
|
||||
results.append({**config, "valid_excess": v["ann_excess"], "test_excess": t["ann_excess"]})
|
||||
|
||||
df = pd.DataFrame(results).sort_values("valid_excess", ascending=False)
|
||||
print(df.head(10).to_string(index=False))
|
||||
# Overfit check: how far is the best-valid config from the median test config?
|
||||
med = df["test_excess"].median()
|
||||
best = df.iloc[0]
|
||||
print(f"\nmedian test excess: {med:+.3f} | best-valid test excess: {best['test_excess']:+.3f}")
|
||||
if abs(best["test_excess"] - med) > 0.10:
|
||||
print("WARNING: best-valid config is an outlier on test — likely overfit, prefer robust defaults")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,90 @@
|
||||
"""Minimal ranking-objective ablation loop — why lambda_rank fails here.
|
||||
|
||||
This repo found that with only ~50 instruments per day the rank-gradient
|
||||
objectives (lambdarank / rank_xendcg) produce near-zero RankIC, while MSE
|
||||
objective + RankIC early-stop is the winner. This script replays that check by
|
||||
training a few LightGBM variants on the same lake split and printing RankIC.
|
||||
|
||||
Reference repo impl: tac-qlib/examples/run_rank_objectives.py.
|
||||
|
||||
python examples/run_rank_objectives.py --universe AAPL,MSFT,QQQ
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import os
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
OBJECTIVES = ["mse", "lambdarank", "rank_xendcg"]
|
||||
|
||||
|
||||
def load_frame(lake_root: str, universe: list[str], start: str, end: str) -> pd.DataFrame:
|
||||
"""Stack lake bars into a qlib-like (datetime, instrument) frame."""
|
||||
from tac_qlib.data.config import LakeConfig
|
||||
|
||||
lake = LakeConfig(lake_root, "US")
|
||||
frames = []
|
||||
for sym in universe:
|
||||
p = lake.bar_path("1d", sym)
|
||||
if p.exists():
|
||||
df = pd.read_parquet(p)[["t", "c"]].rename(columns={"t": "datetime", "c": "close"})
|
||||
df["instrument"] = sym
|
||||
frames.append(df)
|
||||
out = pd.concat(frames, ignore_index=True)
|
||||
out["datetime"] = pd.to_datetime(out["datetime"])
|
||||
out = out[(out["datetime"] >= start) & (out["datetime"] <= end)]
|
||||
return out.set_index(["datetime", "instrument"])
|
||||
|
||||
|
||||
def label_5d(frame: pd.DataFrame) -> pd.Series:
|
||||
close = frame["close"].unstack()
|
||||
lbl = close.shift(-6) / close.shift(-1) - 1
|
||||
return lbl.stack().rename("label")
|
||||
|
||||
|
||||
def train_one(lake_root: str, universe: list[str], objective: str, train: tuple, test: tuple):
|
||||
import lightgbm as lgb
|
||||
|
||||
frame = load_frame(lake_root, universe, train[0], test[1])
|
||||
label = label_5d(frame)
|
||||
data = pd.concat([frame["close"], label], axis=1).dropna()
|
||||
|
||||
tr = data.loc[(data.index.get_level_values(0) >= train[0]) & (data.index.get_level_values(0) <= train[1])]
|
||||
te = data.loc[(data.index.get_level_values(0) >= test[0]) & (data.index.get_level_values(0) <= test[1])]
|
||||
|
||||
dtrain = lgb.Dataset(tr[["close"]], label=tr["label"])
|
||||
dtest = lgb.Dataset(te[["close"]], label=te["label"])
|
||||
params = {"objective": objective, "learning_rate": 0.05, "num_leaves": 15, "verbosity": -1}
|
||||
model = lgb.train(params, dtrain, num_boost_round=100, valid_sets=[dtest])
|
||||
|
||||
pred = model.predict(te[["close"]], num_iteration=model.best_iteration)
|
||||
label_te = te["label"].to_numpy()
|
||||
# per-day RankIC
|
||||
days = te.index.get_level_values(0).unique()
|
||||
ics = []
|
||||
for d in days:
|
||||
m = te.index.get_level_values(0) == d
|
||||
if m.sum() >= 3:
|
||||
ics.append(pd.Series(pred[m]).rank().corr(pd.Series(label_te[m]).rank()))
|
||||
return float(np.nanmean(ics))
|
||||
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--lake-root", default=os.environ.get("TAC_LAKE_DIR", ""))
|
||||
ap.add_argument("--universe", default="AAPL,MSFT,QQQ,IVV,SMH,TLT")
|
||||
args = ap.parse_args()
|
||||
universe = [s.strip().upper() for s in args.universe.split(",")]
|
||||
train = ("2026-03-01", "2026-05-31")
|
||||
test = ("2026-07-01", "2026-08-06")
|
||||
print(f"{'objective':<14}{'test RankIC':>12}")
|
||||
for obj in OBJECTIVES:
|
||||
ic = train_one(args.lake_root, universe, obj, train, test)
|
||||
print(f"{obj:<14}{ic:>12.4f}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,80 @@
|
||||
"""Minimal stochastic-process feature computation — OU mean-reversion + Hurst.
|
||||
|
||||
These are the features that beat hand-rolled TA in this repo's 50-ETF runs.
|
||||
Compute them per symbol on a rolling window ENDING at each day (never let them
|
||||
see test data at fit time — see the HMM/GARCH note in SKILL.md).
|
||||
|
||||
Reference repo impl: tac-qlib/examples/sp_features.py (full 55-feature set:
|
||||
OU, HMM, jump, HARRV, trend, GARCH, Hurst, path signatures, entropy, catch22).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
#: rolling window for feature computation (days)
|
||||
LOOKBACK = 250
|
||||
|
||||
|
||||
def compute_ou_features(close: pd.Series) -> pd.DataFrame:
|
||||
"""Ornstein-Uhlenbeck fit: theta (reversion speed), sigma (vol), residual z.
|
||||
|
||||
OU: dx_t = theta (mu - x_t) dt + sigma dW_t (theta is the mean-reversion
|
||||
speed; higher = faster reversion = tradable mean-reversion signal).
|
||||
|
||||
Rolling OLS of dx on lagged log-price gives theta = -b (reversion speed);
|
||||
sigma is the residual std. Vectorized via rolling cov/var.
|
||||
"""
|
||||
logp = np.log(close)
|
||||
dx = logp.diff()
|
||||
x_prev = logp.shift(1)
|
||||
df = pd.DataFrame({"dx": dx, "x": x_prev})
|
||||
|
||||
out = pd.DataFrame(index=close.index, dtype=float)
|
||||
cov = df["dx"].rolling(LOOKBACK, min_periods=30).cov(df["x"])
|
||||
var = df["x"].rolling(LOOKBACK, min_periods=30).var()
|
||||
theta = (-cov / var).rename("sp_ou_theta")
|
||||
out["sp_ou_theta"] = theta
|
||||
out["sp_ou_sigma"] = df["dx"].rolling(LOOKBACK, min_periods=30).std()
|
||||
# standardized residual z = (x - mu) / sigma of the fitted process
|
||||
mu = df["x"].rolling(LOOKBACK, min_periods=30).mean()
|
||||
scale = np.sqrt(np.clip(1 / (2 * theta + 1e-9), 0, None))
|
||||
out["sp_ou_zscore"] = (df["x"] - mu) / (out["sp_ou_sigma"] * scale)
|
||||
return out
|
||||
|
||||
|
||||
def compute_hurst(close: pd.Series, lookback: int = 100) -> pd.Series:
|
||||
"""Rolling Hurst exponent via rescaled range (R/S). H>0.5 = trending."""
|
||||
def _hurst(x: np.ndarray) -> float:
|
||||
if len(x) < 20:
|
||||
return np.nan
|
||||
lags = range(2, min(len(x) // 2, 50))
|
||||
tau = []
|
||||
for lag in lags:
|
||||
diff = x[lag:] - x[:-lag]
|
||||
tau.append(np.sqrt(np.std(diff)))
|
||||
tau = np.array(tau)
|
||||
lags = np.array(lags, dtype=float)
|
||||
poly = np.polyfit(np.log(lags), np.log(tau), 1)
|
||||
return float(poly[0])
|
||||
|
||||
return close.rolling(lookback, min_periods=20).apply(lambda w: _hurst(w.to_numpy()), raw=False).rename(
|
||||
"sp_hurst_exponent"
|
||||
)
|
||||
|
||||
|
||||
def build_sp_features(bars: pd.DataFrame) -> pd.DataFrame:
|
||||
"""bars: lake 1d bars indexed by (datetime, instrument) or a symbol frame."""
|
||||
if isinstance(bars.index, pd.MultiIndex):
|
||||
frames = []
|
||||
for inst, sub in bars.groupby(level=1):
|
||||
close = sub.droplevel(1)["close"]
|
||||
feats = pd.concat([compute_ou_features(close), compute_hurst(close)], axis=1)
|
||||
feats["instrument"] = inst
|
||||
frames.append(feats.reset_index())
|
||||
out = pd.concat(frames).set_index(["datetime", "instrument"])
|
||||
else:
|
||||
close = bars["close"]
|
||||
out = pd.concat([compute_ou_features(close), compute_hurst(close)], axis=1)
|
||||
return out
|
||||
@@ -0,0 +1,82 @@
|
||||
"""Minimal beta-neutral 3L/3S strategy + record — stub of tac_qlib/contrib/strategy/beta_neutral.py.
|
||||
|
||||
Strategy side: subclass BaseSignalStrategy, hold ~3 long + 3 short equally
|
||||
weighted (dollar-neutral) with TP/SL and a hard close at the horizon. The beta
|
||||
comes from regression of daily returns on the benchmark in `_prepare_betas`.
|
||||
|
||||
Record side (BetaNeutralRecord): a custom `Record` that simulates the 3L/3S
|
||||
portfolio after training and logs report / trades / risk.csv into the MLflow
|
||||
run — the pattern to follow for any custom Record.
|
||||
|
||||
Wire the record into the workflow YAML:
|
||||
|
||||
record:
|
||||
- class: BetaNeutralRecord
|
||||
module_path: tac_qlib.contrib.strategy.beta_neutral
|
||||
kwargs: { benchmark: QQQ, n_long: 3, n_short: 3 }
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, List
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from qlib.backtest import Order
|
||||
from qlib.backtest.decision import OrderDir, TradeDecisionWO
|
||||
from qlib.contrib.strategy.signal_strategy import BaseSignalStrategy
|
||||
|
||||
|
||||
class BetaNeutralStrategy(BaseSignalStrategy):
|
||||
"""3 long / 3 short dollar-neutral template with TP/SL and hard close."""
|
||||
|
||||
def __init__(self, *, n_long: int = 3, n_short: int = 3, tp: float = 0.06, sl: float = -0.05, **kwargs: Any):
|
||||
super().__init__(**kwargs)
|
||||
self.n_long = n_long
|
||||
self.n_short = n_short
|
||||
self.tp = tp
|
||||
self.sl = sl
|
||||
|
||||
def generate_trade_decision(self, execute_result=None):
|
||||
trade_step = self.trade_calendar.get_trade_step()
|
||||
start_time, end_time = self.trade_calendar.get_step_time(trade_step)
|
||||
pred_start, pred_end = self.trade_calendar.get_step_time(trade_step - 1)
|
||||
pred = self.signal.get_signal(start_time=pred_start, end_time=pred_end)
|
||||
|
||||
orders: List[Order] = []
|
||||
if pred is not None and len(pred):
|
||||
daily = pred.groupby(level=0).mean().iloc[-1].dropna().sort_values()
|
||||
longs = daily.tail(self.n_long).index.tolist()
|
||||
shorts = daily.head(self.n_short).index.tolist()
|
||||
for inst in longs:
|
||||
orders.append(self._order(inst, 1, start_time, end_time))
|
||||
for inst in shorts:
|
||||
orders.append(self._order(inst, -1, start_time, end_time))
|
||||
return TradeDecisionWO(orders, self)
|
||||
|
||||
def _order(self, inst, direction, start_time, end_time):
|
||||
price = self.trade_exchange.get_close(inst, end_time) or 1.0
|
||||
qty = int(self.trade_exchange.account.cash / (len(self.trade_exchange.get_positions()) + 1) / price)
|
||||
return Order(
|
||||
inst,
|
||||
qty,
|
||||
start_time,
|
||||
end_time,
|
||||
direction=OrderDir.BUY if direction > 0 else OrderDir.SELL,
|
||||
type="market",
|
||||
)
|
||||
|
||||
|
||||
class BetaNeutralRecord: # subclass qlib.workflow.record_temp.Record in the real impl
|
||||
"""Custom record that backtests 3L/3S and logs report/trades/risk.csv."""
|
||||
|
||||
def __init__(self, *, benchmark: str = "QQQ", n_long: int = 3, n_short: int = 3, **_: Any):
|
||||
self.benchmark = benchmark
|
||||
self.n_long = n_long
|
||||
self.n_short = n_short
|
||||
|
||||
def generate(self, **kwargs):
|
||||
# Real impl: run qlib.backtest with BetaNeutralStrategy on the recorded
|
||||
# pred, write report_normal.csv / positions_normal.csv / risk.csv into
|
||||
# the current MLflow run's artifact dir, then log the headline metrics.
|
||||
print("BetaNeutralRecord.generate: simulate 3L/3S and log artifacts")
|
||||
@@ -0,0 +1,77 @@
|
||||
"""Minimal OptimalStopControl strategy — a stub of tac_qlib/contrib/strategy/optimal_stop.py.
|
||||
|
||||
Subclasses qlib's BaseSignalStrategy; override `generate_trade_decision` to build
|
||||
`qlib.backtest.Order`s and return a `TradeDecisionWO`. The real implementation
|
||||
gates entry by cross-sectional signal percentile, exits by percentile / time /
|
||||
stop-loss, and sizes equal-weight with `risk_degree` control.
|
||||
|
||||
Wire into a workflow YAML under PortAnaRecord.config.strategy:
|
||||
|
||||
strategy:
|
||||
class: OptimalStopControl
|
||||
module_path: tac_qlib.contrib.strategy.optimal_stop
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 10
|
||||
entry_pct: 0.85
|
||||
exit_pct: 0.7
|
||||
max_hold_days: 10
|
||||
min_hold_days: 2
|
||||
sl: -0.08
|
||||
risk_degree: 0.95
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
import numpy as np
|
||||
|
||||
from qlib.backtest import Order
|
||||
from qlib.backtest.decision import OrderDir, TradeDecisionWO
|
||||
from qlib.contrib.strategy.signal_strategy import BaseSignalStrategy
|
||||
|
||||
|
||||
class OptimalStopControl(BaseSignalStrategy):
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
topk: int = 10,
|
||||
entry_pct: float = 0.85,
|
||||
exit_pct: float = 0.7,
|
||||
max_hold_days: int = 10,
|
||||
min_hold_days: int = 2,
|
||||
sl: float = -0.08,
|
||||
risk_degree: float = 0.95,
|
||||
**kwargs: Any,
|
||||
):
|
||||
super().__init__(**kwargs)
|
||||
self.topk = topk
|
||||
self.entry_pct = entry_pct
|
||||
self.exit_pct = exit_pct
|
||||
self.max_hold_days = max_hold_days
|
||||
self.min_hold_days = min_hold_days
|
||||
self.sl = sl
|
||||
self.risk_degree = risk_degree
|
||||
|
||||
def generate_trade_decision(self, execute_result=None):
|
||||
"""Build orders for one trade step (minimal sketch — see repo impl)."""
|
||||
trade_step = self.trade_calendar.get_trade_step()
|
||||
# signal is known at t-1 via shift=-1 in the signal object
|
||||
start_time, end_time = self.trade_calendar.get_step_time(trade_step)
|
||||
pred_start, pred_end = self.trade_calendar.get_step_time(trade_step - 1)
|
||||
pred = self.signal.get_signal(start_time=pred_start, end_time=pred_end)
|
||||
|
||||
orders: List[Order] = []
|
||||
if pred is not None and len(pred):
|
||||
# take the top-k by cross-sectional percentile, equal-weight size
|
||||
cross = pred.groupby(level=0).rank(pct=True) # 0..1 per day
|
||||
keep = pred.index[cross >= 1.0 - self.entry_pct]
|
||||
for inst, (dt, _instr) in zip(keep, keep):
|
||||
price = self.trade_exchange.get_close(inst, end_time) or 1.0
|
||||
qty = int((self.risk_degree * self.trade_exchange.account.cash) / (self.topk * price))
|
||||
if qty > 0:
|
||||
orders.append(
|
||||
Order(inst, qty, start_time, end_time, direction=OrderDir.BUY, type="market")
|
||||
)
|
||||
return TradeDecisionWO(orders, self)
|
||||
@@ -0,0 +1,107 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# MINIMAL workflow — the canonical "run a backtest" template for the skill.
|
||||
#
|
||||
# Every traced backtest runs through a workflow YAML like this one via
|
||||
# rd_run_workflow, so the `record` blocks write MLflow artifacts to disk
|
||||
# (<lake>/mlruns/<exp_id>/<run_id>). The traced experiment's ref id IS the
|
||||
# mlflow run id returned by rd_run_workflow.
|
||||
#
|
||||
# Trigger:
|
||||
# rd_run_workflow config_path=examples/workflow_minimal.yaml \
|
||||
# experiment_name=tac-rd-minimal
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-minimal" }
|
||||
|
||||
task:
|
||||
model:
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.05
|
||||
num_leaves: 15
|
||||
n_estimators: 200
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.01
|
||||
reg_lambda: 0.01
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: AAPL,MSFT,QQQ,IVV,SMH,TLT
|
||||
start_time: 2026-03-01
|
||||
end_time: 2026-08-06
|
||||
fit_start_time: 2026-03-01
|
||||
fit_end_time: 2026-05-31
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
segments:
|
||||
train: [2026-03-01, 2026-05-31]
|
||||
valid: [2026-06-01, 2026-06-30]
|
||||
test: [2026-07-01, 2026-08-06]
|
||||
|
||||
# Record block — REQUIRED. Each entry writes one artifact family to mlruns:
|
||||
# SignalRecord pred.pkl + label.pkl
|
||||
# SigAnaRecord IC / Rank IC series + long-short group returns
|
||||
# PortAnaRecord backtest report / positions / risk
|
||||
record:
|
||||
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: { ana_long_short: true, ann_scaler: 252 }
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 2
|
||||
n_drop: 1
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-07-01
|
||||
end_time: 2026-08-06
|
||||
account: 1000000
|
||||
benchmark: QQQ
|
||||
exchange_kwargs:
|
||||
codes: AAPL,MSFT,QQQ,IVV,SMH,TLT
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,113 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# RankIC early-stop workflow — minimal example wiring the custom model.
|
||||
#
|
||||
# model_rank_gbdt.py must be importable: copy it (or symlink) into
|
||||
# tac_qlib/contrib/model/ and sync to /opt/venv site-packages (see SKILL.md
|
||||
# "Installed package copy" gotcha). Then run:
|
||||
#
|
||||
# rd_run_workflow config_path=examples/workflow_rankic.yaml \
|
||||
# experiment_name=tac-rd-rankic
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-rankic" }
|
||||
|
||||
task:
|
||||
model:
|
||||
# Custom model — see examples/model_rank_gbdt.py (RankICLGBModel):
|
||||
# per-day query groups + feval=rankic + metric='None' so early-stopping
|
||||
# tracks mean per-day Spearman instead of l2.
|
||||
class: RankICLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 15
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l1: 0.0
|
||||
lambda_l2: 0.5
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
seed: 2026
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: AAPL,MSFT,QQQ,IVV,SMH,TLT
|
||||
start_time: 2026-03-01
|
||||
end_time: 2026-08-06
|
||||
fit_start_time: 2026-03-01
|
||||
fit_end_time: 2026-05-31
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
infer_processors:
|
||||
- { class: DropAllNaN, kwargs: {} }
|
||||
- { class: ProcessInf, kwargs: {} }
|
||||
- { class: CSRankNorm, kwargs: {} }
|
||||
- { class: ZScoreNorm, kwargs: {} }
|
||||
- { class: Fillna, kwargs: {} }
|
||||
segments:
|
||||
train: [2026-03-01, 2026-05-31]
|
||||
valid: [2026-06-01, 2026-06-30]
|
||||
test: [2026-07-01, 2026-08-06]
|
||||
|
||||
record:
|
||||
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: { ana_long_short: true, ann_scaler: 252 }
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 2
|
||||
n_drop: 1
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-07-01
|
||||
end_time: 2026-08-06
|
||||
account: 1000000
|
||||
benchmark: QQQ
|
||||
exchange_kwargs:
|
||||
codes: AAPL,MSFT,QQQ,IVV,SMH,TLT
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,168 @@
|
||||
---
|
||||
name: tradeac-rd-explain
|
||||
description: Guide agents to retrieve, visualise and interpret TradeAC R&D workflow data — from the input qrun YAML to final IC / backtest metrics — via the tac-qlib-rd MCP tools (rd_exp_*) and the built-in R&D dashboard (/dashboard/rd). Use when asked about experiments, mlruns runs, workflow inputs, model hyper-parameters, IC/Rank IC evaluation, or backtest results.
|
||||
---
|
||||
|
||||
# tradeac-rd-explain
|
||||
|
||||
Every `qrun` workflow run is recorded into **mlflow** in the unified R&D store under the lake
|
||||
root — sqlite `mlruns.db` + artifact files under `mlruns/<experiment_id>/<run_uuid>/` in
|
||||
`$TAC_LAKE_DIR`. This skill tells you how to pull that data out with the
|
||||
`rd_exp_*` MCP tools (from `tac_qlib.rd_server`), how to read the raw files directly, and how to
|
||||
visualise/interpret everything — either from the built-in TradeAC UI or from the raw data.
|
||||
|
||||
Quick map of the R&D data:
|
||||
|
||||
| Step | Where it lives | `rd_exp_*` tool |
|
||||
|------|----------------|-----------------|
|
||||
| Input config (rendered YAML) | artifact `config` on the run | `rd_exp_input` |
|
||||
| Runs / experiments list | `mlruns.db` → `experiments`, `runs`, `tags`, `params`, `metrics` | `rd_exp_list`, `rd_exp_get_experiment`, `rd_exp_get_run` |
|
||||
| Predictions & labels | artifacts `pred.pkl`, `label.pkl` | `rd_exp_result` |
|
||||
| IC / Rank IC | artifacts `ic.pkl`, `ric.pkl` + metrics `IC`, `ICIR`, `Rank IC`, `Rank ICIR` | `rd_exp_result` |
|
||||
| Group returns | artifacts `long_short_r.pkl`, `long_avg_r.pkl` | `rd_exp_result` |
|
||||
| Backtest / risk | artifacts `portfolio_analysis/report_normal_1d.pkl`, `port_analysis_1d.pkl` + `1day.*` metrics | `rd_exp_result` |
|
||||
| Model + hyper-params | artifact `params.pkl` (qlib model), config `task.model` | `rd_exp_model` |
|
||||
| Hypothesis / evaluation notes | sidecar `rd-notes.json` | `rd_exp_get_notes` / `rd_exp_set_notes` |
|
||||
|
||||
## MCP-first policy
|
||||
|
||||
- **Use the `rd_exp_*` MCP tools to read all of the above** — do not reinvent them with `sqlite3`/pickle/pandas scripts. The tools are the canonical, JSON-safe way to pull experiment data (they fall back to the raw files automatically).
|
||||
- **NEVER script directly against the MCP server** (spawning `tac_qlib.rd_server`, stdio JSON-RPC, bash/curl) unless a tool genuinely can't do the job — then **stop and ask the user to confirm first**.
|
||||
- The raw-file/sqlite snippets in §3 below are **only** for cases where the MCP surface is unavailable or the user explicitly asks for a direct peek.
|
||||
- If the venv is missing a runtime dep (`duckdb`, `pyarrow`, `sqlite3`), lazy-install it (`uv pip install --python $VIRTUAL_ENV/bin/python duckdb pyarrow`) rather than working around it.
|
||||
|
||||
## 1. Prerequisites
|
||||
|
||||
- The tac-qlib-rd MCP server is registered in `opencode.json` (`.venv/bin/python -m tac_qlib.rd_server`).
|
||||
- The server resolves the unified R&D store from the lake root, so `uri` defaults to `sqlite:///<lake>/mlruns.db` (overridable via `MLRUNS_URI`).
|
||||
- Runs must exist first: use `rd_run_workflow` (or `rd_train` + records) to create them.
|
||||
|
||||
## Secrets policy
|
||||
|
||||
- NEVER write secrets into files: DB passwords, API keys, OAuth tokens, or credential-bearing URLs (`DATABASE_URL`, `MLRUNS_URI`) in scripts, configs, notes or committed code.
|
||||
- NEVER read `*.env` / `.env.*` directly (`cat`/`tail`/`grep`/`sed`/`head` on `.env`). That pulls secrets into this session and leaks them to any agent sharing it.
|
||||
- When a tool or command needs an env var, ASK the user to set it in the environment (shell/container env, or the user-owned `.env`) and reference it by name (`$VAR`), never by value. If it's missing, report which variable is required instead of reading it yourself.
|
||||
- If you find a committed secret, flag it, remove it, and replace it with a placeholder.
|
||||
|
||||
## 2. Getting the data
|
||||
|
||||
### 2.1 List experiments and their runs
|
||||
|
||||
```text
|
||||
rd_exp_list
|
||||
# -> [{experiment_id, name, run_count, latest_run: {run_id, status, headline_metrics}}]
|
||||
|
||||
rd_exp_get_experiment experiment_id=1
|
||||
# -> experiment meta + every run: run_id, status, start/end, git, metrics, params, tags, notes, artifacts
|
||||
```
|
||||
|
||||
### 2.2 Input configuration (what went in)
|
||||
|
||||
```text
|
||||
rd_exp_input experiment_id=1 run_id=<run_uuid>
|
||||
```
|
||||
|
||||
Returns the **saved `config` artifact** (the fully-rendered workflow YAML: `qlib_init`,
|
||||
`task.model.kwargs` hyper-parameters, `task.dataset.kwargs.handler` universe/window/features,
|
||||
`segments`, `record` list) plus the resolved `universe` and `feature_fields`. If a run has no
|
||||
`config` artifact (e.g. older `rd_train` runs) the tool falls back to reconstructing from
|
||||
recorded params/tags and marks `source: "reconstructed"` / `"partial"`.
|
||||
|
||||
> Rule of thumb: **the `config` artifact is the most complete input record**; the sqlite
|
||||
> `params` table alone (only `cmd-sys.argv`) is not enough to reconstruct the input.
|
||||
|
||||
### 2.3 Results & evaluation (what came out)
|
||||
|
||||
```text
|
||||
rd_exp_result experiment_id=1 run_id=<run_uuid>
|
||||
```
|
||||
|
||||
Returns: headline `metrics` (IC / ICIR / Rank IC / Rank ICIR, `l2.train`/`l2.valid`, `1day.*`
|
||||
risk metrics), per-day `ic_series` (`[{date, ic, ric}]`), `pred_stats`, `group_returns`
|
||||
(`long_short` / `long_avg`), and the `backtest` report (per-day cumulative return vs benchmark)
|
||||
+ `risk` table.
|
||||
|
||||
### 2.4 Model & hyper-parameters
|
||||
|
||||
```text
|
||||
rd_exp_model experiment_id=1 run_id=<run_uuid>
|
||||
rd_exp_model experiment_id=1 run_id=<run_uuid> tree_id=7
|
||||
```
|
||||
|
||||
Returns `hyperparams` (from config, preferred), `feature_names`, `feature_importances`,
|
||||
`num_trees`, `best_iteration`, and a **pruned top-layers tree** for LightGBM:
|
||||
`tree: {nodes: [{id, depth, feature, threshold, gain, leaf_value, node_count, left, right}]}`.
|
||||
|
||||
### 2.5 Notes (hypothesis / evaluation)
|
||||
|
||||
```text
|
||||
rd_exp_get_notes experiment_id=1 run_id=<run_uuid>
|
||||
rd_exp_set_notes experiment_id=1 run_id=<run_uuid> hypothesis="..." evaluation="..."
|
||||
# persisted to mlruns/<experiment_id>/<run_uuid>/rd-notes.json
|
||||
```
|
||||
|
||||
## 3. Reading the raw files directly
|
||||
|
||||
Everything above is a JSON view of these files (all under `$TAC_LAKE_DIR`):
|
||||
|
||||
- `mlruns.db` (sqlite) — `experiments`, `runs`, `tags`, `params`, `metrics`, `latest_metrics`.
|
||||
Quick peek: `sqlite3 $TAC_LAKE_DIR/mlruns.db "SELECT * FROM latest_metrics;"`.
|
||||
- `mlruns/<experiment_id>/<run_uuid>/artifacts/` — pickle files:
|
||||
- `config` → the input YAML (dict); carries the resolved `feature_fields`
|
||||
- `params.pkl` → the trained model (qlib `LGBModel`; `.model` is a `lightgbm.Booster`)
|
||||
- `pred.pkl`, `label.pkl`, `ic.pkl`, `ric.pkl`, `long_short_r.pkl`, `long_avg_r.pkl`
|
||||
- `portfolio_analysis/report_normal_1d.pkl`, `port_analysis_1d.pkl`
|
||||
- `mlruns/<experiment_id>/<run_uuid>/rd-notes.json` — hypothesis/evaluation notes.
|
||||
|
||||
In Python:
|
||||
|
||||
```python
|
||||
import os
|
||||
import pickle
|
||||
from pathlib import Path
|
||||
|
||||
run_dir = Path(os.environ["TAC_LAKE_DIR"]) / "mlruns/1/<run_uuid>"
|
||||
cfg = pickle.loads((run_dir / "artifacts/config").read_bytes()) # input config dict
|
||||
ic = pickle.loads((run_dir / "artifacts/ic.pkl").read_bytes()) # per-day IC Series
|
||||
import lightgbm
|
||||
model = pickle.loads((run_dir / "artifacts/params.pkl").read_bytes()) # needs qlib import
|
||||
tree = model.model.dump_model()["tree_info"] # LightGBM trees
|
||||
```
|
||||
|
||||
## 4. Visualising & interpreting
|
||||
|
||||
### 4.1 Built-in TradeAC UI
|
||||
|
||||
Open the dashboard: `/dashboard/rd` lists experiments + runs with headline metrics and
|
||||
`Input` / `Result` / `Model` action buttons:
|
||||
|
||||
- `/dashboard/rd/input?expId=<id>` — universe, windows, features, model settings (tables).
|
||||
- `/dashboard/rd/result?expId=<id>` — ECharts IC/Rank IC, cumulative group returns, backtest vs
|
||||
benchmark, per-day IC table, training-loss curves.
|
||||
- `/dashboard/rd/model?expId=<id>` — hyper-parameter table, feature importances, LightGBM tree
|
||||
viewer (pick a tree id).
|
||||
|
||||
### 4.2 Interpreting the numbers
|
||||
|
||||
- **IC / ICIR**: mean per-day IC (predictive power of the signal); ICIR = mean/std × √252.
|
||||
|IC| ≥ ~0.02 daily with stable sign is notable for cross-sectional signals; ICIR ≥ 1 is decent,
|
||||
≥ 2 strong. Rank IC is the Spearman version (more robust to outliers).
|
||||
- **Training loss (`l2.train`/`l2.valid`)**: watch the gap — widening gap ⇒ overfitting;
|
||||
valid flat/rising ⇒ underfitting or stale features.
|
||||
- **Group returns (`long_short_r`)**: cumulative return of top-decile-minus-bottom-decile signal
|
||||
baskets; steady positive slope = the ranking carries money.
|
||||
- **Backtest risk** (`annualized_return`, `information_ratio`, `max_drawdown`): IR = excess
|
||||
return / tracking error; max drawdown shows path risk. Compare against the benchmark column
|
||||
in the cumulative chart.
|
||||
- **Tree viewer**: root splits on the strongest features (high gain). Repeated use of a feature
|
||||
across the top layers ⇒ it dominates; suspicious thresholds near feature extremes often
|
||||
indicate leakage/sample bias.
|
||||
|
||||
## 5. Troubleshooting
|
||||
|
||||
| Symptom | Cause / fix |
|
||||
|---------|-------------|
|
||||
| `experiment_id` not found | Check `rd_exp_list`; ids are the mlflow `experiment_id`, not the name. |
|
||||
| `no recorder` / empty input | Run lacks a `config` artifact (pre-fix `rd_train`). Re-run via `rd_run_workflow` or `rd_train` on the fixed server to record config. |
|
||||
| Pickle errors on `params.pkl` | Ensure qlib + lightgbm importable (server venv). Tool returns a warning and skips the artifact rather than failing. |
|
||||
| Empty result series | Records were not run (only `rd_train`). Use `rd_run_workflow` or add `SignalRecord`/`SigAnaRecord`/`PortAnaRecord`. |
|
||||
@@ -0,0 +1,296 @@
|
||||
---
|
||||
name: tradeac-rd
|
||||
description: Guide agents to run quant R&D on the TradeAC data lake with a Qlib-based research server exposed over MCP (tac-qlib-rd). Train GBDT (LightGBM/XGBoost) and Linear/QDA ML models on lake bars + TA features, generate cross-sectional alpha predictions, evaluate IC/Rank IC, run TopkDropout backtests with benchmark comparison, and execute one-shot YAML workflows — all through `tac_qlib.rd_server`, an MCP server in the repo venv.
|
||||
---
|
||||
|
||||
# tradeac-rd
|
||||
|
||||
Quant R&D server for the TradeAC data lake. Wraps [Qlib](https://github.com/microsoft/qlib) in a local **MCP server** (`tac_qlib.rd_server` in the repo `.venv`) and uses custom qlib data providers that read directly from the lake (see `tac-engine/skills/tradeac-lake/SKILL.md` for the lake itself, and `tac-qlib/README.md` for the package).
|
||||
|
||||
Registered in `opencode.json` as `tac-qlib-rd` — the tools below are available directly once opencode is restarted.
|
||||
|
||||
## MCP-first policy
|
||||
|
||||
- **Prefer the tac-qlib-rd MCP tools** (`rd_train`, `rd_predict`, `rd_evaluate`, `rd_backtest`, `rd_strategy_targets`, `rd_run_workflow`, `rd_exp_*`, `rd_status`) over writing scripts that reimplement the R&D loop (custom qlib glue, own train/predict/eval/backtest, hand-rolled mlruns readers, own JSON-RPC clients).
|
||||
- **NEVER script directly against the MCP server** (spawning `python -m tac_qlib.rd_server`, driving it via bash/curl/stdio) unless a tool genuinely can't do the job — then **stop and ask the user to confirm first**.
|
||||
- Data prep (bars/features backfill) is done with the tac-engine lake MCP tools — see `tac-engine/skills/tradeac-lake/SKILL.md`. Inspect runs with `rd_exp_*` instead of reading `mlruns.db`/pickles directly.
|
||||
- If the venv is missing a runtime dep (e.g. `duckdb`, `pyarrow`), lazy-install it (`uv pip install --python $VIRTUAL_ENV/bin/python duckdb pyarrow`) instead of switching to another tool.
|
||||
|
||||
## The R&D loop
|
||||
|
||||
| Tool | Purpose |
|
||||
|------|---------|
|
||||
| `rd_train` | Fit a model on lake data + TA features, log to MLflow, return run metadata. |
|
||||
| `rd_predict` | Generate out-of-sample predictions from a trained model (by `model_path` or `run_id`). |
|
||||
| `rd_evaluate` | IC / Rank IC stats of a `pred.pkl` vs `label.pkl`. |
|
||||
| `rd_backtest` | TopkDropout backtest of predictions vs a benchmark, with risk metrics + artifacts. |
|
||||
| `rd_strategy_targets` | Turn a prediction's signal day into a deterministic target buy list (TopkDropout selection + sizing). |
|
||||
| `rd_run_workflow` | One-shot: run an entire YAML workflow (train → predict → sig-ana → backtest) and return metrics + artifacts. |
|
||||
| `rd_exp_list` | List MLflow experiments with run ids on the local sqlite store. |
|
||||
| `rd_exp_get_experiment` | Experiment detail: all runs (meta, metrics, notes, artifact files). |
|
||||
| `rd_exp_get_run` | Single run meta + latest metrics. |
|
||||
| `rd_exp_input` | What went into a run: qlib_init, model kwargs, dataset handler kwargs, segments, features, universe, label. |
|
||||
| `rd_exp_result` | What came out: headline IC/ICIR/Rank IC/Rank ICIR, per-day IC series, group returns, prediction stats, backtest report + risk (benchmark-relative). |
|
||||
| `rd_exp_model` | Trained model: class, hyperparameters, LightGBM tree/feature importance. |
|
||||
| `rd_exp_blotter` | Execution log: account P&L summary, daily equity, current positions, trade table, signal blotter. |
|
||||
| `rd_exp_get_notes` / `rd_exp_set_notes` | Read / write hypothesis + evaluation notes on a run. |
|
||||
| `rd_exp_delete` | **HARD-delete** an MLflow experiment: all runs (metrics/params/tags), the traced `rd_experiments` rows that reference them (FK is `ON DELETE CASCADE`), and on-disk artifacts. Irreversible — confirm with the user first. |
|
||||
| `rd_exp_delete_run` | **HARD-delete** a single MLflow run + its traced `rd_experiments` row + artifacts. Irreversible — confirm with the user first. |
|
||||
| `rd_status` | Lake + qlib readiness: data window, symbols, persisted features, qlib version. |
|
||||
|
||||
The standard flow is `rd_train` → `rd_predict` → `rd_evaluate` → `rd_backtest`; `rd_run_workflow` replaces all of it with a YAML config. The `rd_exp_*` inspection tools read saved MLflow artifacts, so every page in the R&D app (`/rd/input`, `/rd/result`, `/rd/model`, `/rd/blotter`) is backed by an MCP call (`rd_exp_input`, `rd_exp_result`, `rd_exp_model`, `rd_exp_blotter`) keyed by `experiment_id` + `run_id`.
|
||||
|
||||
## Setup / env
|
||||
|
||||
| Var | Default | Purpose |
|
||||
|-----|---------|---------|
|
||||
| `TAC_LAKE_DIR` | **required** (no default) | lake root (bars + features + metadata). Local dev: absolute path (e.g. `/home/data/lake`). |
|
||||
| `TAC_RD_MARKET` | `US` | market partition for lake reads |
|
||||
| `DATABASE_URL` | – | Postgres tracking store for MLflow (its own `experiments`/`runs`/… tables) when set |
|
||||
| `MLRUNS_URI` | postgres (`$DATABASE_URL`) or `sqlite:///<lake>/mlruns.db` | MLflow tracking URI override. Artifact files always live under `<lake>/mlruns/<exp_id>/<run_uuid>/` |
|
||||
|
||||
The server lives in the repo `.venv`; the MCP config is already registered. Restart opencode after editing `opencode.json`.
|
||||
|
||||
## Secrets policy
|
||||
|
||||
- NEVER write secrets into files: DB passwords, API keys, OAuth tokens, or credential-bearing URLs (`DATABASE_URL`, `MLRUNS_URI`, `EMBEDDING_API_KEY`) in workflow YAMLs, scripts, configs, notes or committed code.
|
||||
- NEVER read `*.env` / `.env.*` directly (`cat`/`tail`/`grep`/`sed`/`head` on `.env`). That pulls secrets into this session and leaks them to any agent sharing it.
|
||||
- When a tool or command needs an env var, ASK the user to set it in the environment (shell/container env, or the user-owned `.env`) and reference it by name (`$VAR`), never by value. If it's missing, report which variable is required instead of reading it yourself.
|
||||
- Tracking store: use `uri: "sqlite:///mlruns.db"` (relative) in workflows — `rd_run_workflow` normalizes it to Postgres when `$DATABASE_URL` is set, else the lake sqlite. Never hardcode a `postgres://user:pass@…` URI.
|
||||
- If you find a committed secret, flag it, remove it, and replace it with a placeholder.
|
||||
|
||||
## `rd_train`
|
||||
|
||||
Args: `universe` (comma-separated), `train_start/valid_end/test_end` (`YYYY-MM-DD`), `experiment_name`, `out_dir`, `model` (`lgb` default, `xgb`, `linear`, `qda`), optional `label` (default `Ref($close,-2)/Ref($close,-1)-1`, the next-day return), `topk`/`n_drop` for later backtests.
|
||||
|
||||
- Loads 1d bars + all persisted TA features (`features/` dir) for the universe from the lake.
|
||||
- Splits into train / valid / test; fits on train with early stopping on valid.
|
||||
- Logs the run to MLflow (`run_id`), saves `params.pkl` (model) + `pred.pkl` + `label.pkl` to `out_dir`.
|
||||
- With `record_analysis=true` (default) also runs SignalRecord / SigAnaRecord (`ana_long_short`) / PortAnaRecord inside the run, so the result page gets IC/Rank IC series, long-short group returns, monthly IC and the portfolio backtest. `benchmark`, `topk`, `n_drop`, `account`, `risk_degree`, `open_cost`/`close_cost`/`min_cost` tune that backtest.
|
||||
- Returns `run_id`, `status`, `fit_seconds`, `feature_fields`, `label`, per-segment rows/date/instrument counts, and artifact paths.
|
||||
- **`wait=false`** runs the fit in a background thread and returns immediately (`status: started`, `background: true`) — poll `rd_exp_get_run` / `rd_exp_list` for the newest run of `experiment_name` until its status is `FINISHED`, then use that `run_id`. Use it for slow windows (e.g. a 4y retrain) where a blocking MCP call can time out.
|
||||
|
||||
> If the lake lacks bars or features for `universe`, backfill first via the tac-engine `get_lake_bars` / `get_lake_ta` tools, or raise the training start date.
|
||||
|
||||
## `rd_predict`
|
||||
|
||||
Args: `universe`, `model_path` **or** `run_id`+`experiment_name` (artifact `params.pkl` is loaded from MLflow), same date ranges as `rd_train`, `out_dir`, optional `top` (number of top-scored rows in the `head` list).
|
||||
|
||||
- Rebuilds the same feature matrix for `test_start..test_end`, produces scores.
|
||||
- Writes `pred.pkl` (scores) and `label.pkl` (labels) to `out_dir`.
|
||||
- Returns paths, `count`, `date_min/max`, instruments, score distribution stats, and a small `head`.
|
||||
|
||||
## `rd_evaluate`
|
||||
|
||||
Args: `pred_path`, `label_path` (the two pkl files from `rd_train`/`rd_predict`).
|
||||
|
||||
- Returns `IC` and `RankIC` tables (`days`, `mean`, `std`, `ann_vol`, `ir`, `skew`, `kurt`, `maxdd`) and a `headline` (`IC`, `ICIR`, `Rank IC`, `Rank ICIR`).
|
||||
|
||||
## `rd_backtest`
|
||||
|
||||
Args: `pred_path`, `start_time`/`end_time`, `topk`, `n_drop`, `benchmark`, optional `out_dir`.
|
||||
|
||||
- TopkDropoutStrategy (topk long, n_drop drop), $100k account, `risk_degree 0.95`, day freq, benchmark comparison.
|
||||
- Returns `start_time`, `end_time`, `trading_days`, `risk` (`mean`, `std`, `annualized_return`, `information_ratio`, `max_drawdown`), `benchmark`, and artifact paths (`report_normal.csv`, `positions_normal.csv`, `risk.csv`).
|
||||
|
||||
## `rd_strategy_targets`
|
||||
|
||||
Args: `pred_path`, optional `signal_date` (defaults to the last day in the prediction), `account`, `risk_degree`, `topk`, `n_drop`, optional `prices` (JSON `{symbol: price}`).
|
||||
|
||||
- Applies the **exact TopkDropout selection** for one signal day: rank the cross-sectional scores, drop the top `n_drop`, take the next `topk` as buys, sized at `account × risk_degree / topk` per name. Use this to chain a prediction straight into an order list — no manual strategy replication.
|
||||
- With `prices`, floors each order to whole shares (`qty`) and reports `expected_price` / `invested`.
|
||||
- Returns `signal_date`, `per_name_notional`, the deterministic `targets` list (`symbol`, `rank`, `score`, `side`, `notional`, `qty`), and the top-20 `ranking` for context. If fewer than `topk + n_drop` names have a score that day it returns empty `targets` with a `reason`.
|
||||
|
||||
## `rd_run_workflow`
|
||||
|
||||
Args: `config_path` (YAML, see `tac-qlib/workflows/workflow_lgb_taclake.yaml`), `experiment_name`, optional `wait` (default `false`), optional `run_in_new_process` (default `false`).
|
||||
|
||||
- Runs the full pipeline (qlib `signal` + `records`), returns `run_id`, `status`, the resolved `qlib_init`/`model`/`dataset`/`records` config, and `metrics` (train/valid loss, IC/ICIR/Rank IC/Rank ICIR, and the `1day.*` backtest metrics).
|
||||
- `wait=false` (default) returns immediately with `status: started`; the workflow runs in a background thread — poll `rd_exp_get_run` / `rd_exp_list` for the newest run of `experiment_name` until it finishes. `wait=true` blocks until completion (only for small windows that finish inside the MCP call timeout).
|
||||
- `run_in_new_process=true` runs the workflow in a **separate OS process** instead of a thread. qlib `init` sets process-global state, so this is the safe mode for concurrent or long workflows — it isolates crashes, releases memory on exit, and avoids the thread-safety race. stdout/stderr are redirected to `<lake>/logs/rd-workflow-<exp>-<ts>.log` (returned as `log_path`; the child must never write to the MCP stdio pipe). Polling works identically because the child writes to the same mlflow store. The process `pid` is returned.
|
||||
|
||||
## Tracing every run started from a chat (REQUIRED)
|
||||
|
||||
**Every experiment you start from this chat must be traced FIRST.** The R&D
|
||||
lineage (`/rd/lineage`) and the round book build on the `rd_experiments` table —
|
||||
an experiment created by `rd_run_workflow` / `rd_train` without a
|
||||
`rd_trace_start` is invisible there (no lineage node, no chat link). So before
|
||||
triggering any run, use the `rd_trace_*` MCP tools (tac-qlib-rd):
|
||||
|
||||
1. Open the trace BEFORE the run (see `tac-qlib/skills/tac-qlib-custom/SKILL.md`,
|
||||
"Experiment traceability" — the skill that owns the trace flow):
|
||||
```
|
||||
rd_trace_start rational="<what this run tests, in one line>" \
|
||||
details="<universe / features / label / model / strategy sizing>" \
|
||||
experiment_name=<the experiment you will run into> \
|
||||
evolved_from=<predecessor traced id or auto> \
|
||||
session_id="<this chat's opencode session id>"
|
||||
# -> {"experiment_id": N, "branch": "...", "evolved_from": ..., "base_branch": ...}
|
||||
```
|
||||
2. Run the workflow into that **same** `experiment_name`:
|
||||
```
|
||||
rd_run_workflow config_path=<yaml> experiment_name=<the experiment name>
|
||||
```
|
||||
3. On success, **finish the trace** (links the run, copies metrics/evaluation):
|
||||
```
|
||||
rd_trace_finish experiment_id=<N> ref_id=<run_id> \
|
||||
evaluation="<outcome>" metrics='{...headline...}' \
|
||||
mlruns_dir=<lake>/mlruns/<experiment_id>/<run_id>
|
||||
```
|
||||
|
||||
If you are NOT tracing (quick throwaway exploration), say so explicitly and note
|
||||
the run will not appear in the lineage graph. The default for any run started
|
||||
from a chat is to trace it.
|
||||
|
||||
## Building a workflow YAML and triggering a run
|
||||
|
||||
A workflow YAML is a qrun config: `qlib_init` (lake providers + MLflow exp manager), `task.model`, `task.dataset`, and `task.record`. Copy `tac-qlib/workflows/workflow_lgb_taclake.yaml` as the template.
|
||||
|
||||
```yaml
|
||||
# jinja is available: {%- set LAKE = TAC_LAKE_DIR %} (TAC_LAKE_DIR is required)
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
calendar_provider: {class: tac_qlib.data.providers.LakeCalendarProvider, kwargs: {lake_root: "{{ LAKE }}", market: US}}
|
||||
instrument_provider: {class: tac_qlib.data.providers.LakeInstrumentProvider, kwargs: {lake_root: "{{ LAKE }}", market: US, markets: {}}}
|
||||
feature_provider: {class: tac_qlib.data.providers.LakeFeatureProvider, kwargs: {lake_root: "{{ LAKE }}", market: US}}
|
||||
exp_manager: {class: MLflowExpManager, module_path: qlib.workflow.expm, kwargs: {uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd"}}
|
||||
|
||||
task:
|
||||
model:
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs: {loss: mse, learning_rate: 0.05, num_leaves: 15, n_estimators: 200,
|
||||
colsample_bytree: 0.8, subsample: 0.8, subsample_freq: 1,
|
||||
reg_alpha: 0.01, reg_lambda: 0.01}
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ # universe
|
||||
start_time: 2000-01-03 # lake look-back for features
|
||||
end_time: 2026-08-06
|
||||
fit_start_time: 2026-03-01 # normalization fit window
|
||||
fit_end_time: 2026-05-31
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-2)/Ref($close,-1)-1" # next-day return
|
||||
segments: # train/valid/test split
|
||||
train: [2026-03-01, 2026-05-31]
|
||||
valid: [2026-06-01, 2026-06-30]
|
||||
test: [2026-07-01, 2026-08-06]
|
||||
record:
|
||||
- {class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {}}
|
||||
- {class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: {ana_long_short: true, ann_scaler: 252}}
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs: {signal: "<PRED>", topk: 2, n_drop: 1, only_tradable: true, risk_degree: 0.95}
|
||||
backtest:
|
||||
start_time: 2026-07-01
|
||||
end_time: 2026-08-06
|
||||
account: 1000000
|
||||
benchmark: QQQ # any symbol in the lake; empty = no benchmark
|
||||
exchange_kwargs:
|
||||
codes: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
```
|
||||
|
||||
Then trigger it (each call = one new run in the named experiment):
|
||||
|
||||
```text
|
||||
rd_run_workflow config_path=tac-qlib/workflows/tune_run1_wider_5d.yaml experiment_name=tac-rd-tune
|
||||
# -> run_id <uuid>; save it, then inspect via rd_exp_*.
|
||||
```
|
||||
|
||||
The saved `config` artifact (same shape as above) is what `rd_exp_input` returns, so runs are reproducible from their YAML.
|
||||
|
||||
## Inspecting a run given experiment_id + run_id
|
||||
|
||||
The URL on the R&D app is `/rd/input|result|model|blotter?expId=<id>&run=<uuid>`; the underlying MCP calls are:
|
||||
|
||||
| You want | MCP call | Args |
|
||||
|----------|----------|------|
|
||||
| Full results (IC/ICIR/Rank IC, group returns, backtest risk) | `rd_exp_result` | `experiment_id`, `run_id` |
|
||||
| Input config (universe, windows, features, label, model) | `rd_exp_input` | `experiment_id`, `run_id` |
|
||||
| Execution blotter (P&L, positions, trades, signals) | `rd_exp_blotter` | `experiment_id`, `run_id` |
|
||||
| Model (hyperparams, tree, importances) | `rd_exp_model` | `experiment_id`, `run_id` |
|
||||
| Run notes | `rd_exp_get_notes` / `rd_exp_set_notes` | `experiment_id`, `run_id` (+ `hypothesis`/`evaluation`) |
|
||||
|
||||
Get the run ids first: `rd_exp_list` → pick an experiment → `rd_exp_get_experiment` returns its runs (meta + latest metrics), or `rd_exp_get_run run_id=<uuid>` for one run.
|
||||
|
||||
## Evaluating a run and proposing the next one
|
||||
|
||||
Treat each run as one hypothesis. To evaluate and iterate:
|
||||
|
||||
1. **Read the input** (`rd_exp_input`): universe, train/valid/test windows, label expression, features, model + hyperparams. Note what was held fixed vs changed.
|
||||
2. **Read the signal metrics** (`rd_exp_result.headline`): IC (predictive power), ICIR (stability — |ICIR| ≥ 0.5 strong, 0.2–0.5 weak but persistent, < 0.2 noise), Rank IC/Rank ICIR. A decent IC with Rank IC ≈ 0 means the ranking is noisy even if the mean cross-section is predictive.
|
||||
3. **Read the backtest** (`rd_exp_result.backtest`): `annualized_return`, `information_ratio`, `max_drawdown` are **excess vs the benchmark** (qlib mean-daily × 238). Compare against `return_annualized` (raw strategy) and `benchmark_annualized`; check the benchmark is a sensible peer (a single high-flying stock like AAPL is a brutal benchmark for an ETF universe).
|
||||
4. **Read the blotter** (`rd_exp_blotter.summary`): `n_trades`/`trading_days` reveal turnover; `total_cost` vs account is the cost drag; positions show concentration. High turnover + low topk on correlated names = cost-heavy, undiversified book.
|
||||
5. **Diagnose** and pick ONE lever for the next run — change one thing, hold the rest fixed so the comparison is clean:
|
||||
- *Weak/noisy signal* (ICIR < 0.3, Rank IC ≈ 0): longer label horizon (e.g. 5-day `Ref($close,-6)/Ref($close,-1)-1`), stronger regularization (`reg_alpha`/`reg_lambda` up, `subsample`/`colsample` down), or a cleaner universe (drop leveraged/duplicate names).
|
||||
- *Good signal, bad book* (high IC but poor excess return): raise `topk` for diversification, tune `n_drop` for rotation, reduce turnover, check `total_cost`.
|
||||
- *Benchmark mismatch*: pick an index ETF (QQQ/IVV) the universe tracks instead of a single stock.
|
||||
- *Data window*: a 3-month fit window is short; consider rolling/expanding if the lake history allows.
|
||||
6. **Write the next run as a YAML** (see section above), **trace it first** (`rd_trace_start experiment_name=<exp>`), then trigger with `rd_run_workflow` into that **same new experiment** (e.g. `tac-rd-tune`), and `rd_trace_finish experiment_id=<N> ref_id=<run_id>` when it succeeds. Then `rd_exp_get_experiment` to compare run-to-run. Record the hypothesis/evaluation via `rd_exp_set_notes`.
|
||||
|
||||
Example: the baseline `Exp-1 Run-f29f5446` shows IC 0.071 / ICIR 0.17 / Rank IC 0.014 with excess return −0.94 ann (IR −2.23) vs a +89% ann benchmark — the 1-day signal is unstable, the topk=2 book turned 24 trades in 27 days (~1.1% cost drag) on correlated ETFs + leveraged hedges, and AAPL is an unfair benchmark. Two improvement runs are ready in `tac-qlib/workflows/tune_run1_wider_5d.yaml` (5-day label, topk=5, deduped 10-name universe, benchmark QQQ) and `tune_run2_regularized.yaml` (stronger regularization, topk=3/n_drop=2, same-day label) — trigger both into `experiment_name=tac-rd-tune` and compare.
|
||||
|
||||
## Example session
|
||||
|
||||
```text
|
||||
# 1) train
|
||||
rd_train universe=AAPL,MSFT,TSLA,USO,SLV,TLT train_start=2026-03-01 train_end=2026-05-31
|
||||
valid_start=2026-06-01 valid_end=2026-06-30 test_start=2026-07-01 test_end=2026-08-06
|
||||
experiment_name=tac-rd-mcp out_dir=/tmp/rd_out
|
||||
# -> run_id ...
|
||||
|
||||
# 2) predict on the test window (from the mlflow run)
|
||||
rd_predict universe=AAPL,MSFT,TSLA,USO,SLV,TLT run_id=<run_id> experiment_name=tac-rd-mcp
|
||||
train_start=2026-03-01 train_end=2026-05-31 valid_start=2026-06-01 valid_end=2026-06-30
|
||||
test_start=2026-07-01 test_end=2026-08-06 out_dir=/tmp/rd_out top=5
|
||||
|
||||
# 3) evaluate the alpha
|
||||
rd_evaluate pred_path=/tmp/rd_out/pred.pkl label_path=/tmp/rd_out/label.pkl
|
||||
|
||||
# 4) backtest the signal
|
||||
rd_backtest pred_path=/tmp/rd_out/pred.pkl start_time=2026-07-01 end_time=2026-08-06 topk=2 n_drop=1 benchmark=AAPL
|
||||
|
||||
# 5) one-shot equivalent — trace first, then run, then finish
|
||||
rd_trace_start --rational "<hypothesis>" --experiment-name tac-rd-one-shot --evolved-from auto
|
||||
rd_run_workflow config_path=tac-qlib/workflows/workflow_lgb_taclake.yaml experiment_name=tac-rd-one-shot
|
||||
rd_trace_finish --id <EXPERIMENT_ID> --ref-id <run_id> --evaluation "<outcome>"
|
||||
|
||||
# 6) inspect that run later — given experiment_id + run_id (the /rd pages call exactly these)
|
||||
rd_exp_get_experiment experiment_id=1 # -> runs with meta + latest metrics
|
||||
rd_exp_input experiment_id=1 run_id=<run_id> # what went in: universe, windows, features, label, model
|
||||
rd_exp_result experiment_id=1 run_id=<run_id> # what came out: IC/ICIR/Rank IC, backtest risk
|
||||
rd_exp_blotter experiment_id=1 run_id=<run_id> # execution: P&L, positions, trades, signals
|
||||
rd_exp_model experiment_id=1 run_id=<run_id> # hyperparameters + tree / importances
|
||||
rd_exp_set_notes experiment_id=1 run_id=<run_id> hypothesis="5d label + topk5" evaluation="ICIR 0.5, ann +12%"
|
||||
```
|
||||
|
||||
## Notes
|
||||
|
||||
- The server reads the lake lazily via the custom `LakeCalendarProvider` / `LakeInstrumentProvider` / `LakeFeatureProvider`; if data is missing the relevant provider raises a clear error — backfill through the tac-engine lake tools first.
|
||||
- All tools return JSON via stdio (MCP). Diagnostics/logs go to stderr.
|
||||
- MLflow runs are stored in the tracking store at `$DATABASE_URL` (Postgres) when
|
||||
set, else the unified lake sqlite `mlruns.db`; artifact files always live under
|
||||
`<lake>/mlruns/<exp_id>/<run_uuid>/`. Override the tracking URI with `MLRUNS_URI` if needed.
|
||||
- `rd_run_workflow` / `rd_train` pin each experiment's MLflow `artifact_location` to `<lake>/mlruns` so DB and artifacts stay co-located even when the server process runs from another cwd. Readers (`rd_exp_*`) resolve each run's artifact dir from its recorded `artifact_uri`, falling back to the side-by-side `<lake>/mlruns` layout — so runs whose artifacts were written elsewhere (e.g. `<cwd>/mlruns`) still display.
|
||||
@@ -0,0 +1,18 @@
|
||||
"""tac-qlib: read the TradeAC parquet lake from within the qlib research workflow.
|
||||
|
||||
This package is fully decoupled from the upstream ``qlib`` checkout. It provides:
|
||||
|
||||
- ``tac_qlib.qlib_init.qlib_init``: drop-in ``qlib.init`` configured against the lake.
|
||||
- ``tac_qlib.data.providers``: qlib data providers (calendar / instruments / features)
|
||||
backed by the lake parquet files, so ``qlib.init`` + ``D.features`` work without any
|
||||
``*.bin`` data.
|
||||
- ``tac_qlib.contrib.data.handler``: a ``DataHandlerLP`` subclass (``TACHandler``) that
|
||||
builds a train/test dataset from raw OHLCV + pre-computed ta-lib features.
|
||||
|
||||
Upstream ``qlib/`` is never modified.
|
||||
"""
|
||||
|
||||
from .qlib_init import qlib_init, provider_config
|
||||
|
||||
__all__ = ["qlib_init", "provider_config"]
|
||||
__version__ = "0.1.0"
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,388 @@
|
||||
"""Round book MCP server (stdio transport) — the execution trail for algo rounds.
|
||||
|
||||
Exposes the round-book tools over MCP so the agent (tac-algo-trade skill) and
|
||||
the R&D UI can read AND write the same execution trail in Postgres:
|
||||
|
||||
scheduler_runs ──► ROUND ──► rd_experiments
|
||||
|
||||
round_create / round_update / round_update_status / round_list / round_get windows
|
||||
fact_record / fact_query evidence
|
||||
intent_set / intent_get / intent_list target portfolios
|
||||
decision_record / decision_query / order_link gates + orders
|
||||
round_sync_fills Alpaca fills
|
||||
book_reconcile / trail_query / trail_funnel / round_metrics investigation
|
||||
|
||||
Run::
|
||||
|
||||
.venv/bin/python -m tac_qlib.book_server # stdio MCP server
|
||||
|
||||
All tools return JSON-safe dicts. Logging goes to stderr; stdout is reserved
|
||||
for the MCP protocol. DB access is via tac_qlib.book_db (psycopg + DATABASE_URL);
|
||||
fill sync additionally needs APCA_API_KEY_ID / APCA_API_SECRET_KEY when the
|
||||
agent does not pass `orders` explicitly.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import functools
|
||||
import json
|
||||
import sys
|
||||
from typing import Any, Dict, List, Optional, Sequence
|
||||
|
||||
from mcp.server.mcpserver import MCPServer
|
||||
|
||||
from tac_qlib import book_db
|
||||
|
||||
book_db._load_repo_env()
|
||||
|
||||
server = MCPServer(
|
||||
name="tac-rd-book",
|
||||
title="TradeAC round book",
|
||||
instructions=(
|
||||
"Execution trail for algo trading rounds on the TradeAC stack: create "
|
||||
"round windows, record evidence facts, set versioned target intents, "
|
||||
"record placed/skipped decisions, link Alpaca orders, sync fills, and "
|
||||
"reconcile / trace the funnel. Backed by Postgres (DATABASE_URL)."
|
||||
),
|
||||
version="0.1.0",
|
||||
)
|
||||
|
||||
|
||||
def _log(message: str) -> None:
|
||||
print(f"[tac-rd-book] {message}", file=sys.stderr)
|
||||
|
||||
|
||||
def _as_obj(value: Any) -> Any:
|
||||
"""Accept structured MCP input as JSON strings or as already-parsed dicts/lists."""
|
||||
if isinstance(value, str):
|
||||
if not value.strip():
|
||||
return None
|
||||
try:
|
||||
return json.loads(value)
|
||||
except json.JSONDecodeError:
|
||||
return value
|
||||
return value
|
||||
|
||||
|
||||
def _obj(value: Any) -> Optional[Dict[str, Any]]:
|
||||
parsed = _as_obj(value)
|
||||
return parsed if isinstance(parsed, dict) else None
|
||||
|
||||
|
||||
def _arr(value: Any) -> Optional[List[Any]]:
|
||||
parsed = _as_obj(value)
|
||||
return parsed if isinstance(parsed, list) else None
|
||||
|
||||
|
||||
def _open_round(func):
|
||||
"""Ensure the round tables exist before any round-book operation.
|
||||
Uses ``functools.wraps`` so ``inspect.signature`` follows ``__wrapped__``
|
||||
and the MCP tool schema keeps the real typed parameters."""
|
||||
@functools.wraps(func)
|
||||
def wrapper(*args, **kwargs):
|
||||
try:
|
||||
book_db.ensure_schema()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
_log(f"schema check failed: {exc}")
|
||||
return func(*args, **kwargs)
|
||||
|
||||
return wrapper
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- windows
|
||||
|
||||
@_open_round
|
||||
def round_create(
|
||||
target_date: str,
|
||||
source: str = "scheduled",
|
||||
signal_date: str = "",
|
||||
scheduler_run_id: int = 0,
|
||||
rd_experiment_id: int = 0,
|
||||
experiment_name: str = "",
|
||||
run_id: str = "",
|
||||
model_path: str = "",
|
||||
strategy_snapshot: str = "{}",
|
||||
account_equity_at_sizing: float = 0.0,
|
||||
) -> dict:
|
||||
"""Open a round window for a target trading date. Idempotent per
|
||||
(source, target_date): an already-open round for the same window is
|
||||
returned unchanged (``reused=True``). Returns the full round row."""
|
||||
snap = _obj(strategy_snapshot) or {}
|
||||
return book_db.create_round(
|
||||
source=source,
|
||||
target_date=target_date,
|
||||
signal_date=signal_date or None,
|
||||
scheduler_run_id=scheduler_run_id or None,
|
||||
rd_experiment_id=rd_experiment_id or None,
|
||||
experiment_name=experiment_name or None,
|
||||
run_id=run_id or None,
|
||||
model_path=model_path or None,
|
||||
strategy_snapshot=snap,
|
||||
account_equity_at_sizing=account_equity_at_sizing or None,
|
||||
)
|
||||
|
||||
|
||||
def round_update(
|
||||
round_id: int,
|
||||
source: str = "",
|
||||
target_date: str = "",
|
||||
signal_date: str = "",
|
||||
scheduler_run_id: int = 0,
|
||||
rd_experiment_id: int = 0,
|
||||
experiment_name: str = "",
|
||||
run_id: str = "",
|
||||
model_path: str = "",
|
||||
strategy_snapshot: str = "",
|
||||
account_equity_at_sizing: float = 0.0,
|
||||
) -> dict:
|
||||
"""Update a round window's metadata — e.g. pin the new training run
|
||||
(``run_id`` / ``model_path``) and strategy snapshot after the retrain.
|
||||
Empty / zero values leave the field unchanged."""
|
||||
return book_db.update_round(
|
||||
round_id,
|
||||
source=source or None,
|
||||
target_date=target_date or None,
|
||||
signal_date=signal_date or None,
|
||||
scheduler_run_id=scheduler_run_id or None,
|
||||
rd_experiment_id=rd_experiment_id or None,
|
||||
experiment_name=experiment_name or None,
|
||||
run_id=run_id or None,
|
||||
model_path=model_path or None,
|
||||
strategy_snapshot=_obj(strategy_snapshot) if strategy_snapshot else None,
|
||||
account_equity_at_sizing=account_equity_at_sizing or None,
|
||||
)
|
||||
|
||||
|
||||
def round_update_status(
|
||||
round_id: int,
|
||||
status: str = "",
|
||||
locked_intent_id: int = 0,
|
||||
summary_metrics: str = "{}",
|
||||
feedback_note: str = "",
|
||||
) -> dict:
|
||||
"""Advance a round (open → locked → settled | aborted). ``locked_intent_id``
|
||||
pins the intent reconciliation uses. ``summary_metrics`` / ``feedback_note``
|
||||
update the round summary."""
|
||||
return book_db.update_round_status(
|
||||
round_id,
|
||||
status=status or None,
|
||||
locked_intent_id=locked_intent_id or None,
|
||||
summary_metrics=_obj(summary_metrics),
|
||||
feedback_note=feedback_note or None,
|
||||
)
|
||||
|
||||
|
||||
def round_list(
|
||||
source: str = "",
|
||||
target_date: str = "",
|
||||
status: str = "",
|
||||
limit: int = 20,
|
||||
include_detail: bool = False,
|
||||
) -> dict:
|
||||
"""List round windows (newest first), optionally filtered by source /
|
||||
target_date / status. ``include_detail`` attaches each round's funnel
|
||||
counts + roll-up metrics (used by the /dashboard/rounds list)."""
|
||||
return {
|
||||
"rounds": book_db.list_rounds(
|
||||
source=source or None,
|
||||
target_date=target_date or None,
|
||||
status=status or None,
|
||||
limit=limit,
|
||||
with_detail=bool(include_detail),
|
||||
)
|
||||
}
|
||||
|
||||
|
||||
def round_get(round_id: int) -> dict:
|
||||
"""Full detail of one round: window row + intents, decisions, orders, facts,
|
||||
funnel and reconciliation — everything the UI's round detail page needs."""
|
||||
round_row = book_db.get_round(round_id)
|
||||
if round_row is None:
|
||||
return {"error": f"round {round_id} not found"}
|
||||
return {
|
||||
"round": round_row,
|
||||
"intents": book_db.list_intents(round_id),
|
||||
"decisions": book_db.query_decisions(round_id),
|
||||
"orders": book_db.list_orders(round_id),
|
||||
"facts": book_db.query_facts(round_id, limit=500),
|
||||
"funnel": book_db.funnel(round_id),
|
||||
"reconcile": book_db.reconcile(round_id),
|
||||
"metrics": book_db.metrics(round_id),
|
||||
}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- facts
|
||||
|
||||
def fact_record(round_id: int, kind: str, payload: str = "{}", symbol: str = "", source: str = "") -> dict:
|
||||
"""Append an evidence event (signal_score, quote, news_sentiment,
|
||||
account_state, position_state, strategy_config, ...) to a round."""
|
||||
return book_db.record_fact(
|
||||
round_id,
|
||||
kind=kind,
|
||||
payload=_obj(payload),
|
||||
symbol=symbol or None,
|
||||
source=source or None,
|
||||
)
|
||||
|
||||
|
||||
def fact_query(round_id: int, kind: str = "", symbol: str = "", limit: int = 200) -> dict:
|
||||
"""Query a round's recorded facts (newest first), optionally filtered by kind/symbol."""
|
||||
return {"facts": book_db.query_facts(round_id, kind=kind or None, symbol=symbol or None, limit=limit)}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- intents
|
||||
|
||||
def intent_set(round_id: int, target_portfolio: str, raw_strategy_output: str = "{}", reason: str = "") -> dict:
|
||||
"""Write the next target-portfolio version for a round (auto-increments and
|
||||
supersedes the previous active version). ``target_portfolio`` is a JSON
|
||||
array of {symbol, side, qty, notional, expected_price, score, rank, weight}."""
|
||||
return book_db.set_intent(
|
||||
round_id,
|
||||
target_portfolio=_arr(target_portfolio) or [],
|
||||
raw_strategy_output=_obj(raw_strategy_output),
|
||||
reason=reason or None,
|
||||
)
|
||||
|
||||
|
||||
def intent_get(round_id: int, version: int = 0) -> dict:
|
||||
"""Get a round's intent — the given version, or the active (max) version
|
||||
when ``version`` is omitted."""
|
||||
intent = book_db.get_intent(round_id, version=version or None)
|
||||
return {"intent": intent} if intent else {"intent": None, "error": f"no intent for round {round_id}"}
|
||||
|
||||
|
||||
def intent_list(round_id: int) -> dict:
|
||||
"""List every target-portfolio version for a round (oldest first)."""
|
||||
return {"intents": book_db.list_intents(round_id)}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- decisions / orders
|
||||
|
||||
def decision_record(
|
||||
round_id: int,
|
||||
symbol: str,
|
||||
side: str,
|
||||
qty: float = 0.0,
|
||||
order_type: str = "",
|
||||
expected_price: float = 0.0,
|
||||
status: str = "intended",
|
||||
reason: str = "",
|
||||
reason_detail: str = "",
|
||||
intent_id: int = 0,
|
||||
supersedes_decision_id: int = 0,
|
||||
alpaca_order_id: str = "",
|
||||
client_order_id: str = "",
|
||||
) -> dict:
|
||||
"""Record one per-symbol decision by the gates. Use status ``skipped`` /
|
||||
``rejected`` with a ``reason`` for deliberate skips; placed orders carry
|
||||
``alpaca_order_id`` / ``client_order_id`` (an execution row is created).
|
||||
Passing ``supersedes_decision_id`` marks the previous decision superseded."""
|
||||
return book_db.record_decision(
|
||||
round_id,
|
||||
symbol=symbol,
|
||||
side=side,
|
||||
qty=qty or None,
|
||||
order_type=order_type or None,
|
||||
expected_price=expected_price or None,
|
||||
status=status,
|
||||
reason=reason or None,
|
||||
reason_detail=reason_detail or None,
|
||||
intent_id=intent_id or None,
|
||||
supersedes_decision_id=supersedes_decision_id or None,
|
||||
alpaca_order_id=alpaca_order_id or None,
|
||||
client_order_id=client_order_id or None,
|
||||
)
|
||||
|
||||
|
||||
def decision_query(round_id: int, symbol: str = "", include_superseded: bool = True) -> dict:
|
||||
"""List a round's decisions, optionally filtered by symbol."""
|
||||
return {"decisions": book_db.query_decisions(round_id, symbol=symbol or None, include_superseded=include_superseded)}
|
||||
|
||||
|
||||
def order_link(
|
||||
round_id: int,
|
||||
decision_id: int,
|
||||
alpaca_order_id: str = "",
|
||||
client_order_id: str = "",
|
||||
qty_filled: float = -1.0,
|
||||
avg_fill_price: float = -1.0,
|
||||
status: str = "",
|
||||
) -> dict:
|
||||
"""Create or update the execution row for a placed decision (idempotent per
|
||||
decision). Use ``qty_filled=-1`` to leave the value unchanged."""
|
||||
return book_db.link_order(
|
||||
round_id,
|
||||
decision_id=decision_id,
|
||||
alpaca_order_id=alpaca_order_id or None,
|
||||
client_order_id=client_order_id or None,
|
||||
qty_filled=qty_filled if qty_filled >= 0 else None,
|
||||
avg_fill_price=avg_fill_price if avg_fill_price >= 0 else None,
|
||||
status=status or None,
|
||||
)
|
||||
|
||||
|
||||
def round_sync_fills(round_id: int, orders: str = "", feed: str = "iex") -> dict:
|
||||
"""Pull Alpaca order state into the round. Pass ``orders`` as a JSON array
|
||||
(tac-engine ``list_orders`` output) or omit it to fetch from Alpaca with
|
||||
APCA_API_* env vars. Marks orders superseded when the effective intent no
|
||||
longer targets their symbol+side. Returns updated / superseded / unmatched."""
|
||||
parsed = _arr(orders) if isinstance(orders, str) else orders
|
||||
return book_db.sync_fills(round_id, orders=parsed if isinstance(parsed, list) else None, feed=feed or "iex")
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- investigation
|
||||
|
||||
def book_reconcile(round_id: int) -> dict:
|
||||
"""Reconcile the round: effective intent targets vs decisions vs fills, with
|
||||
per-symbol residuals and roll-ups (cash/BP impact, slippage bps, cost)."""
|
||||
return book_db.reconcile(round_id)
|
||||
|
||||
|
||||
def trail_query(round_id: int, symbol: str = "") -> dict:
|
||||
"""Per-symbol waterfall: intent target → decision → order → fill."""
|
||||
return {"trail": book_db.trail(round_id, symbol=symbol or None)}
|
||||
|
||||
|
||||
def trail_funnel(round_id: int) -> dict:
|
||||
"""Decision funnel counts for a round: targets → decided → placed → filled,
|
||||
plus skipped-reason breakdown and superseded count."""
|
||||
return book_db.funnel(round_id)
|
||||
|
||||
|
||||
def round_metrics(round_id: int) -> dict:
|
||||
"""Roll-up metrics: placed/filled order counts, invested notional, turnover."""
|
||||
return book_db.metrics(round_id)
|
||||
|
||||
|
||||
def register_tools(mcp_server: MCPServer) -> None:
|
||||
"""Attach all round-book tools to an ``MCPServer`` instance."""
|
||||
for fn in (
|
||||
round_create,
|
||||
round_update,
|
||||
round_update_status,
|
||||
round_list,
|
||||
round_get,
|
||||
fact_record,
|
||||
fact_query,
|
||||
intent_set,
|
||||
intent_get,
|
||||
intent_list,
|
||||
decision_record,
|
||||
decision_query,
|
||||
order_link,
|
||||
round_sync_fills,
|
||||
book_reconcile,
|
||||
trail_query,
|
||||
trail_funnel,
|
||||
round_metrics,
|
||||
):
|
||||
mcp_server.tool(structured_output=False)(fn)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
register_tools(server)
|
||||
server.run()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,11 @@
|
||||
from . import data # noqa: F401 (registers tac_qlib.contrib.data)
|
||||
from . import model, strategy # noqa: F401
|
||||
from .data import TACHandler # noqa: F401
|
||||
from .model import RankICLGBModel # noqa: F401
|
||||
from .strategy import OptimalStopControl # noqa: F401
|
||||
|
||||
__all__ = [
|
||||
"TACHandler",
|
||||
"RankICLGBModel",
|
||||
"OptimalStopControl",
|
||||
]
|
||||
@@ -0,0 +1,3 @@
|
||||
from .handler import TACHandler
|
||||
|
||||
__all__ = ["TACHandler"]
|
||||
@@ -0,0 +1,254 @@
|
||||
"""TACHandler: a qlib DataHandlerLP that builds datasets from the TradeAC lake.
|
||||
|
||||
This is the "custom DataHandler" entry point (Option B): the handler is referenced from the
|
||||
workflow yaml's ``dataset.handler`` and reads OHLCV + pre-computed ta-lib features straight
|
||||
from the lake parquet files through ``QLibDataLoader`` + the tac_qlib feature provider.
|
||||
|
||||
The standard qlib processor pipeline (``infer_processors`` / ``learn_processors``) still runs
|
||||
on top, so existing recipes such as ``DropnaLabel``, ``CSZScoreNorm`` or ``RobustZScoreNorm``
|
||||
keep working unchanged.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from inspect import getfullargspec
|
||||
from typing import List, Optional, Tuple, Union
|
||||
|
||||
from qlib.data.dataset import processor as processor_module
|
||||
from qlib.data.dataset.handler import DataHandlerLP
|
||||
from qlib.utils import get_callable_kwargs
|
||||
|
||||
from ...data.config import (
|
||||
LakeConfig,
|
||||
timeframe_for_freq,
|
||||
NON_FEATURE_COLUMNS,
|
||||
)
|
||||
|
||||
DEFAULT_INFER_PROCESSORS = [
|
||||
{"class": "DropAllNaN", "kwargs": {}},
|
||||
{"class": "ProcessInf", "kwargs": {}},
|
||||
{"class": "ZScoreNorm", "kwargs": {}},
|
||||
{"class": "Fillna", "kwargs": {}},
|
||||
]
|
||||
DEFAULT_LEARN_PROCESSORS = [
|
||||
{"class": "DropnaLabel"},
|
||||
{"class": "CSZScoreNorm", "kwargs": {"fields_group": "label"}},
|
||||
]
|
||||
|
||||
#: always include raw OHLCV; ta-lib columns are discovered from the lake and appended.
|
||||
RAW_FEATURE_FIELDS = ("$open", "$high", "$low", "$close", "$vwap", "$volume")
|
||||
|
||||
DEFAULT_LABEL = "Ref($close,-2)/Ref($close,-1)-1"
|
||||
|
||||
|
||||
def check_transform_proc(proc_l, fit_start_time, fit_end_time):
|
||||
"""Port of ``qlib.contrib.data.handler.check_transform_proc`` (inject fit window into procs)."""
|
||||
new_l = []
|
||||
for p in proc_l:
|
||||
if not isinstance(p, processor_module.Processor):
|
||||
klass, pkwargs = get_callable_kwargs(p, processor_module)
|
||||
args = getfullargspec(klass).args
|
||||
if "fit_start_time" in args and "fit_end_time" in args:
|
||||
assert fit_start_time is not None and fit_end_time is not None, (
|
||||
"Make sure `fit_start_time` and `fit_end_time` are not None."
|
||||
)
|
||||
pkwargs.update({"fit_start_time": fit_start_time, "fit_end_time": fit_end_time})
|
||||
proc_config = {"class": klass.__name__, "kwargs": pkwargs}
|
||||
if isinstance(p, dict) and "module_path" in p:
|
||||
proc_config["module_path"] = p["module_path"]
|
||||
new_l.append(proc_config)
|
||||
else:
|
||||
new_l.append(p)
|
||||
return new_l
|
||||
|
||||
|
||||
def get_common_feature_fields(lake_root=None, market="US", timeframe="1d") -> List[str]:
|
||||
"""Discover feature columns present in *every* feature file of the lake.
|
||||
|
||||
Walks the `family=ta|sp` partition layout (plus any legacy flat files).
|
||||
TA and SP columns are disjoint by construction, so the common set is
|
||||
computed per family (columns shared by all symbol files of that family),
|
||||
then the per-family results are unioned. Returns sorted field names
|
||||
(without the ``$`` prefix). Empty if no features are persisted.
|
||||
"""
|
||||
cfg = LakeConfig(lake_root, market)
|
||||
feat_dir = cfg.features_dir(timeframe)
|
||||
if not feat_dir.exists():
|
||||
return []
|
||||
import pyarrow.parquet as pq
|
||||
|
||||
def _family_common(fam_dir: Path) -> set:
|
||||
common = None
|
||||
for p in sorted(fam_dir.glob("symbol=*.parquet")):
|
||||
try:
|
||||
cols = set(pq.read_schema(p).names) - set(NON_FEATURE_COLUMNS)
|
||||
except Exception: # pragma: no cover - skip unreadable files
|
||||
continue
|
||||
common = cols if common is None else (common & cols)
|
||||
if not common:
|
||||
break
|
||||
return common or set()
|
||||
|
||||
common: set = set()
|
||||
# family tier: features/market=*/timeframe=*/family=*/symbol=*.parquet
|
||||
for fam in ("ta", "sp"):
|
||||
fam_dir = feat_dir / f"family={fam}"
|
||||
if fam_dir.is_dir():
|
||||
common |= _family_common(fam_dir)
|
||||
# legacy flat: features/market=*/timeframe=*/symbol=*.parquet
|
||||
if (feat_dir / "family=ta").exists() or (feat_dir / "family=sp").exists():
|
||||
pass # family layout already covered
|
||||
else:
|
||||
common |= _family_common(feat_dir)
|
||||
return sorted(common)
|
||||
|
||||
|
||||
class DropAllNaN(processor_module.Processor):
|
||||
"""Drop feature columns that are all-NaN over the fit window.
|
||||
|
||||
The lake can hold fully-empty indicator columns (e.g. a ta-lib output that was NaN
|
||||
from the start). Such columns carry no learnable signal and make ``ZScoreNorm.fit``
|
||||
warn on empty slices, so we drop them before any other processor runs. The drop set
|
||||
is fixed on the fit window once (during ``fit``), then applied consistently to every
|
||||
segment so train/valid/test keep identical feature columns.
|
||||
"""
|
||||
|
||||
def __init__(self, fit_start_time=None, fit_end_time=None):
|
||||
self.fit_start_time = fit_start_time
|
||||
self.fit_end_time = fit_end_time
|
||||
self.cols_to_drop = []
|
||||
|
||||
def fit(self, df=None):
|
||||
if df is None or len(df) == 0:
|
||||
return self
|
||||
window = df
|
||||
if self.fit_start_time is not None and self.fit_end_time is not None:
|
||||
try:
|
||||
from qlib.data.dataset.utils import fetch_df_by_index
|
||||
|
||||
window = fetch_df_by_index(
|
||||
df, slice(self.fit_start_time, self.fit_end_time), level="datetime"
|
||||
)
|
||||
except Exception: # pragma: no cover - defensive
|
||||
window = df
|
||||
if len(window) == 0:
|
||||
return self
|
||||
self.cols_to_drop = [c for c in window.columns if window[c].isna().all()]
|
||||
return self
|
||||
|
||||
def __call__(self, df):
|
||||
if self.cols_to_drop:
|
||||
return df.drop(columns=self.cols_to_drop, errors="ignore")
|
||||
return df
|
||||
|
||||
|
||||
class TACHandler(DataHandlerLP):
|
||||
"""DataHandlerLP backed by the TradeAC parquet lake.
|
||||
|
||||
Parameters mirror ``Alpha158``: ``instruments``/``start_time``/``end_time``/``freq`` define
|
||||
the queried window; ``feature_fields`` selects the features (default: raw OHLCV + all common
|
||||
ta-lib columns found in the lake); ``label`` is a qlib expression for the target.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
instruments="all",
|
||||
start_time=None,
|
||||
end_time=None,
|
||||
freq="day",
|
||||
infer_processors=DEFAULT_INFER_PROCESSORS,
|
||||
learn_processors=DEFAULT_LEARN_PROCESSORS,
|
||||
fit_start_time=None,
|
||||
fit_end_time=None,
|
||||
process_type=DataHandlerLP.PTYPE_A,
|
||||
filter_pipe=None,
|
||||
feature_fields=None,
|
||||
label=DEFAULT_LABEL,
|
||||
lake_root=None,
|
||||
market="US",
|
||||
**kwargs,
|
||||
):
|
||||
# default the processor fit window to the queried window (like Alpha158 without a split)
|
||||
if fit_start_time is None:
|
||||
fit_start_time = start_time
|
||||
if fit_end_time is None:
|
||||
fit_end_time = end_time
|
||||
|
||||
infer_processors = check_transform_proc(infer_processors, fit_start_time, fit_end_time)
|
||||
learn_processors = check_transform_proc(learn_processors, fit_start_time, fit_end_time)
|
||||
|
||||
feature_fields = self._normalize_feature_fields(feature_fields, freq, lake_root, market)
|
||||
if not feature_fields:
|
||||
raise ValueError(
|
||||
"no feature fields available for the lake; set `feature_fields` explicitly "
|
||||
"(e.g. ['$close', '$rsi_14', '$sma_20'])"
|
||||
)
|
||||
|
||||
label_expr, label_names = self._normalize_label(label)
|
||||
|
||||
data_loader = {
|
||||
"class": "QlibDataLoader",
|
||||
"kwargs": {
|
||||
"config": {
|
||||
"feature": (feature_fields, feature_fields),
|
||||
"label": (label_expr, label_names),
|
||||
},
|
||||
"filter_pipe": filter_pipe,
|
||||
"freq": freq,
|
||||
},
|
||||
}
|
||||
super().__init__(
|
||||
instruments=instruments,
|
||||
start_time=start_time,
|
||||
end_time=end_time,
|
||||
data_loader=data_loader,
|
||||
infer_processors=infer_processors,
|
||||
learn_processors=learn_processors,
|
||||
process_type=process_type,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
# ------------------------------------------------------------------ config
|
||||
@staticmethod
|
||||
def _normalize_feature_fields(feature_fields, freq, lake_root, market) -> List[str]:
|
||||
if feature_fields is None:
|
||||
common = get_common_feature_fields(lake_root, market, timeframe_for_freq(freq))
|
||||
feature_fields = list(RAW_FEATURE_FIELDS) + ["$" + f for f in common if "$" + f not in RAW_FEATURE_FIELDS]
|
||||
elif isinstance(feature_fields, str):
|
||||
feature_fields = [f.strip() for f in feature_fields.split(",") if f.strip()]
|
||||
fields = [f if f.startswith("$") else "$" + f for f in feature_fields]
|
||||
# de-dup while preserving order
|
||||
seen, out = set(), []
|
||||
for f in fields:
|
||||
if f not in seen:
|
||||
seen.add(f)
|
||||
out.append(f)
|
||||
return out
|
||||
|
||||
@staticmethod
|
||||
def _normalize_label(label) -> Tuple[List[str], List[str]]:
|
||||
if isinstance(label, str):
|
||||
return [label], ["LABEL0"]
|
||||
if isinstance(label, (list, tuple)):
|
||||
if len(label) == 2 and isinstance(label[0], str):
|
||||
return [label[0]], list(label[1]) if isinstance(label[1], (list, tuple)) else [label[1]]
|
||||
return list(label), ["LABEL%d" % i for i in range(len(label))]
|
||||
raise TypeError(f"unsupported label config: {label!r}")
|
||||
|
||||
# ------------------------------------------------------------------ utils
|
||||
def get_label_config(self):
|
||||
return DEFAULT_LABEL
|
||||
|
||||
@staticmethod
|
||||
def discover_feature_fields(lake_root=None, market="US", freq="day") -> List[str]:
|
||||
return get_common_feature_fields(lake_root, market, timeframe_for_freq(freq))
|
||||
|
||||
|
||||
__all__ = ["TACHandler", "DropAllNaN", "get_common_feature_fields"]
|
||||
|
||||
|
||||
# Make `DropAllNaN` resolvable by bare name from processor configs (e.g. the default
|
||||
# ``infer_processors`` and workflow yamls that reference it without a ``module_path``),
|
||||
# mirroring how qlib registers its own processors in ``qlib.data.dataset.processor``.
|
||||
processor_module.DropAllNaN = DropAllNaN
|
||||
@@ -0,0 +1,4 @@
|
||||
from .rank_ensemble import RankICEnsembleLGBModel # noqa: F401
|
||||
from .rank_gbdt import RankICLGBModel, rankic_feval # noqa: F401
|
||||
|
||||
__all__ = ["RankICLGBModel", "rankic_feval", "RankICEnsembleLGBModel"]
|
||||
@@ -0,0 +1,189 @@
|
||||
"""Seed-ensembled LightGBM that early-stops on cross-sectional RankIC.
|
||||
|
||||
``RankICEnsembleLGBModel`` wraps ``RankICLGBModel`` (per-day RankIC feval +
|
||||
``metric='None'`` + ``first_metric_only`` early stopping) over a seed ensemble:
|
||||
one sub-model is trained per seed with identical hyper-parameters, and
|
||||
predictions are averaged across seeds. This is the model class the
|
||||
``tac-rd-rank-ensemble-isolated`` reference run wires into its workflow
|
||||
(``module_path: tac_qlib.contrib.model.rank_ensemble``).
|
||||
|
||||
The ensemble inherits the RankIC early-stopping behaviour of the single-seed
|
||||
model (valid RankIC drives the stopping iteration) while the seed averaging
|
||||
stabilizes the prediction against any single seed's early-stopping path.
|
||||
|
||||
Training is parallelized: the seed sub-models train in a thread pool —
|
||||
``lgb.train`` is C++ and releases the GIL, so concurrent seeds do not block on
|
||||
the GIL (5 seeds ~40min/5 on this box). Measured on a 6-physical-core / 12 SMT
|
||||
host: the seeds scale ~2x, not linearly — the runs are memory-bandwidth bound
|
||||
and each Booster caps its threads at ``cores // workers`` so 5 concurrent
|
||||
boosters don't oversubscribe; larger-core hosts scale better. The qlib data
|
||||
pipeline is warmed once on the calling thread (fills the handler cache), and
|
||||
each worker then prepares its **own** ``lgb.Dataset`` (independent handle, so
|
||||
no concurrent ``construct()`` on a shared handle — LightGBM's ``Dataset`` is
|
||||
not thread-safe to build). qlib's ``R`` recorder is also not thread-safe, so
|
||||
the per-seed evaluation curves are logged on the calling thread after the pool
|
||||
finishes.
|
||||
|
||||
Wired into a workflow yaml like:
|
||||
|
||||
model:
|
||||
class: RankICEnsembleLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 31
|
||||
n_estimators: 3000
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l2: 0.5
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seeds: "42,7,2026,99,123"
|
||||
parallel: 5
|
||||
|
||||
Any ``**kwargs`` other than ``seeds``/``parallel`` are forwarded unchanged to
|
||||
every ``RankICLGBModel`` sub-model (same params, different ``seed``).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from typing import List, Optional
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from qlib.data.dataset import DatasetH
|
||||
from qlib.data.dataset.handler import DataHandlerLP
|
||||
|
||||
from tac_qlib.contrib.model.rank_gbdt import RankICLGBModel
|
||||
|
||||
__all__ = ["RankICEnsembleLGBModel"]
|
||||
|
||||
|
||||
class RankICEnsembleLGBModel(RankICLGBModel):
|
||||
"""Seed ensemble of RankIC-early-stopping LightGBM models.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
seeds : comma-separated integers, one sub-model per seed.
|
||||
parallel : number of seeds to train concurrently. ``0`` (default) = auto
|
||||
(all seeds, bounded by the available cores); ``1`` = sequential.
|
||||
**kwargs : forwarded to every ``RankICLGBModel`` sub-model (model
|
||||
hyper-parameters). ``seeds``/``parallel`` are consumed here and not
|
||||
forwarded.
|
||||
"""
|
||||
|
||||
def __init__(self, seeds: str = "42", parallel: int = 0, **kwargs):
|
||||
self.seeds = [int(s.strip()) for s in str(seeds).split(",") if s.strip()]
|
||||
if not self.seeds:
|
||||
raise ValueError("seeds must contain at least one integer")
|
||||
self.parallel = int(parallel)
|
||||
# drop seed/parallel handling from the base kwargs, keep everything else
|
||||
self._model_kwargs = dict(kwargs)
|
||||
super().__init__(**self._model_kwargs)
|
||||
self._models: List[RankICLGBModel] = []
|
||||
|
||||
# --------------------------------------------------------------- helpers
|
||||
@staticmethod
|
||||
def _cores() -> int:
|
||||
try:
|
||||
return max(1, len(os.sched_getaffinity(0)))
|
||||
except AttributeError:
|
||||
return max(1, os.cpu_count() or 1)
|
||||
|
||||
def _worker_count(self) -> int:
|
||||
if self.parallel > 0:
|
||||
return min(len(self.seeds), self.parallel)
|
||||
return min(len(self.seeds), self._cores())
|
||||
|
||||
# ------------------------------------------------------------------ fit
|
||||
def fit(
|
||||
self,
|
||||
dataset: DatasetH,
|
||||
num_boost_round: Optional[int] = None,
|
||||
early_stopping_rounds: Optional[int] = None,
|
||||
verbose_eval: int = 20,
|
||||
evals_result=None,
|
||||
reweighter=None,
|
||||
**kwargs,
|
||||
):
|
||||
"""Train one RankICLGBModel per seed and keep them for prediction.
|
||||
|
||||
The qlib data pipeline is warmed once on this thread (handler cache),
|
||||
then each seed sub-model trains in a parallel worker thread on its own
|
||||
``lgb.Dataset`` (LightGBM releases the GIL in ``lgb.train``). Evals
|
||||
are logged on this thread after the pool (qlib's ``R`` is not
|
||||
thread-safe).
|
||||
"""
|
||||
n_round = num_boost_round or self.num_boost_round
|
||||
n_es = early_stopping_rounds or self.early_stopping_rounds
|
||||
|
||||
if len(self.seeds) == 1:
|
||||
m = RankICLGBModel(seed=self.seeds[0], **self._model_kwargs)
|
||||
m.fit(
|
||||
dataset,
|
||||
num_boost_round=n_round,
|
||||
early_stopping_rounds=n_es,
|
||||
verbose_eval=verbose_eval,
|
||||
evals_result=evals_result,
|
||||
reweighter=reweighter,
|
||||
**kwargs,
|
||||
)
|
||||
self._models = [m]
|
||||
return
|
||||
|
||||
# Warm the qlib handler cache once on this thread so the workers'
|
||||
# concurrent prepare() calls only hit cached frames (no first-write race).
|
||||
proto = RankICLGBModel(seed=self.seeds[0], **self._model_kwargs)
|
||||
proto._prepare_data(dataset, reweighter)
|
||||
|
||||
workers = self._worker_count()
|
||||
# Cap per-Booster threads so concurrent seeds don't oversubscribe
|
||||
# (LightGBM's num_threads=0 uses ALL cores per Booster).
|
||||
per_booster = max(1, self._cores() // workers)
|
||||
|
||||
def fit_seed(seed):
|
||||
m = RankICLGBModel(seed=seed, **self._model_kwargs)
|
||||
if workers > 1 and "num_threads" not in m.params:
|
||||
m.params["num_threads"] = per_booster
|
||||
ds_l = m._prepare_data(dataset, reweighter)
|
||||
booster, evals, names = m._train_from_datasets(
|
||||
ds_l,
|
||||
num_boost_round=n_round,
|
||||
early_stopping_rounds=n_es,
|
||||
verbose_eval=verbose_eval,
|
||||
**kwargs,
|
||||
)
|
||||
m.model = booster
|
||||
return m, evals, names
|
||||
|
||||
with ThreadPoolExecutor(max_workers=workers) as ex:
|
||||
results = list(ex.map(fit_seed, self.seeds))
|
||||
|
||||
self._models = [m for m, _, _ in results]
|
||||
|
||||
# Merge + log evals on the main thread (qlib's R is not thread-safe).
|
||||
if evals_result is not None:
|
||||
for m, evals, names in results:
|
||||
for k in names:
|
||||
for key, val in evals.get(k, {}).items():
|
||||
evals_result.setdefault(f"{k}.seed{m.params['seed']}", {})[key] = val
|
||||
for m, evals, names in results:
|
||||
self._log_evals(evals, names, prefix=f"seed{m.params['seed']}.")
|
||||
|
||||
# -------------------------------------------------------------- predict
|
||||
def predict(self, dataset: DatasetH, segment="test") -> pd.Series:
|
||||
"""Average the per-seed predictions over the given segment."""
|
||||
if not self._models:
|
||||
raise ValueError("model is not fitted yet!")
|
||||
preds = [m.predict(dataset, segment=segment) for m in self._models]
|
||||
if len(preds) == 1:
|
||||
return preds[0]
|
||||
frame = pd.concat(preds, axis=1)
|
||||
return frame.mean(axis=1)
|
||||
@@ -0,0 +1,200 @@
|
||||
"""LGBModel variant that early-stops on cross-sectional RankIC instead of l2.
|
||||
|
||||
Standard qlib ``LGBModel`` early-stops on the regression loss (mse). For
|
||||
cross-sectional alpha signals the quantity we actually care about is the per-day
|
||||
rank correlation (Rank IC), which mse early-stopping does not optimize for.
|
||||
Experiments on the 50-ETF lake (SP-5d 55-feature panel) show that early-stopping
|
||||
on a custom RankIC feval lifts RankIC 0.047 -> 0.075 vs. the mse-stopped model.
|
||||
|
||||
This class reuses ``LGBModel``'s data preparation but:
|
||||
|
||||
- tags each ``lgb.Dataset`` with per-day query ``group`` sizes so a ranking
|
||||
metric can be computed per trading day;
|
||||
- injects a custom ``feval`` (mean per-day Spearman of pred vs label) into
|
||||
``lgb.train``; early stopping then selects the iteration that maximizes
|
||||
RankIC on the valid set;
|
||||
- forces ``metric='None'`` + ``first_metric_only=True`` so early-stopping
|
||||
tracks RankIC only (not the regression loss).
|
||||
|
||||
Wired into a workflow yaml like:
|
||||
|
||||
model:
|
||||
class: RankICLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.03
|
||||
num_leaves: 31
|
||||
n_estimators: 500
|
||||
...
|
||||
|
||||
The rank feval is used for early-stopping selection only; the objective stays
|
||||
the configured loss (default mse). Set ``rank_eval=False`` to fall back to the
|
||||
plain LGBModel behaviour (early-stop on the loss).
|
||||
|
||||
Generic: works for any cross-sectional panel whose qlib dataset index has a
|
||||
``datetime`` level (each level value = one query group). The per-day groups are
|
||||
derived automatically, so no universe-specific configuration is needed.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import List, Optional, Tuple
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import lightgbm as lgb
|
||||
|
||||
from qlib.data.dataset import DatasetH
|
||||
from qlib.data.dataset.handler import DataHandlerLP
|
||||
from qlib.contrib.model.gbdt import LGBModel
|
||||
from qlib.workflow import R
|
||||
|
||||
__all__ = ["RankICLGBModel", "rankic_feval"]
|
||||
|
||||
|
||||
def _per_day_spearman(preds: np.ndarray, labels: np.ndarray, group: np.ndarray) -> float:
|
||||
"""Mean per-day Spearman rank correlation of preds vs labels.
|
||||
|
||||
``group`` holds the number of rows of each trading day (query group), in
|
||||
order. Days with <3 valid rows or a constant pred/label are skipped.
|
||||
"""
|
||||
if group is None or len(group) == 0:
|
||||
return 0.0
|
||||
offs = np.concatenate([[0], np.cumsum(group.astype(int))])
|
||||
vals = []
|
||||
for i in range(len(group)):
|
||||
s = slice(offs[i], offs[i + 1])
|
||||
p, l = preds[s], labels[s]
|
||||
if len(p) < 3 or np.std(p) == 0 or np.std(l) == 0:
|
||||
continue
|
||||
vals.append(np.corrcoef(pd.Series(p).rank(), pd.Series(l).rank())[0, 1])
|
||||
return float(np.mean(vals)) if vals else 0.0
|
||||
|
||||
|
||||
def rankic_feval(preds, dataset):
|
||||
"""LightGBM feval: mean RankIC (higher is better in lgb convention)."""
|
||||
labels = dataset.get_label()
|
||||
group = dataset.get_group()
|
||||
ric = _per_day_spearman(preds, labels, group)
|
||||
return "rankic", ric, True # (name, value, higher_is_better)
|
||||
|
||||
|
||||
class RankICLGBModel(LGBModel):
|
||||
"""LGBModel that early-stops on per-day RankIC via a custom feval."""
|
||||
|
||||
def __init__(self, rank_eval: bool = True, **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
self.rank_eval = rank_eval
|
||||
|
||||
def _prepare_data(self, dataset: DatasetH, reweighter=None) -> List[Tuple[lgb.Dataset, str]]:
|
||||
ds_l = []
|
||||
assert "train" in dataset.segments
|
||||
for key in ["train", "valid"]:
|
||||
if key in dataset.segments:
|
||||
df = dataset.prepare(key, col_set=["feature", "label"], data_key=DataHandlerLP.DK_L)
|
||||
if df.empty:
|
||||
raise ValueError("Empty data from dataset, please check your dataset config.")
|
||||
x, y = df["feature"], df["label"]
|
||||
if y.values.ndim == 2 and y.values.shape[1] == 1:
|
||||
y = np.squeeze(y.values)
|
||||
else:
|
||||
raise ValueError("LightGBM doesn't support multi-label training")
|
||||
|
||||
if reweighter is None:
|
||||
w = None
|
||||
elif hasattr(reweighter, "reweight"):
|
||||
w = reweighter.reweight(df)
|
||||
else:
|
||||
raise ValueError("Unsupported reweighter type.")
|
||||
|
||||
# per-day query groups: each trading day is one group
|
||||
if self.rank_eval and isinstance(df.index, pd.MultiIndex) and "datetime" in df.index.names:
|
||||
group = df.groupby(level="datetime").size().to_numpy(dtype=np.int32)
|
||||
else:
|
||||
group = None
|
||||
|
||||
d = lgb.Dataset(x.values, label=y, weight=w, group=group, free_raw_data=False)
|
||||
ds_l.append((d, key))
|
||||
return ds_l
|
||||
|
||||
def _train_from_datasets(
|
||||
self,
|
||||
ds_l: List[Tuple[lgb.Dataset, str]],
|
||||
num_boost_round: Optional[int] = None,
|
||||
early_stopping_rounds: Optional[int] = None,
|
||||
verbose_eval: int = 20,
|
||||
evals_result=None,
|
||||
**kwargs,
|
||||
) -> Tuple[lgb.Booster, dict, List[str]]:
|
||||
"""Train a Booster from already-prepared ``lgb.Dataset`` objects.
|
||||
|
||||
Pure training — no ``R.log_metrics`` — so it can be called from worker
|
||||
threads (qlib's ``R`` recorder is not thread-safe; the caller decides
|
||||
when/where to log). Returns ``(booster, evals_result, segment_names)``.
|
||||
"""
|
||||
if evals_result is None:
|
||||
evals_result = {}
|
||||
ds, names = list(zip(*ds_l))
|
||||
|
||||
callbacks = [
|
||||
lgb.early_stopping(
|
||||
self.early_stopping_rounds if early_stopping_rounds is None else early_stopping_rounds
|
||||
),
|
||||
lgb.log_evaluation(period=verbose_eval),
|
||||
lgb.record_evaluation(evals_result),
|
||||
]
|
||||
if self.rank_eval:
|
||||
# early-stopping must be driven ONLY by the RankIC feval, not l2.
|
||||
# metric='None' suppresses the default l2 metric; first_metric_only
|
||||
# makes early_stopping track the single remaining (rankic) metric.
|
||||
self.params["metric"] = "None"
|
||||
self.params["first_metric_only"] = True
|
||||
feval = rankic_feval
|
||||
else:
|
||||
self.params.pop("metric", None)
|
||||
self.params.pop("first_metric_only", None)
|
||||
feval = None
|
||||
|
||||
booster = lgb.train(
|
||||
self.params,
|
||||
ds[0],
|
||||
num_boost_round=self.num_boost_round if num_boost_round is None else num_boost_round,
|
||||
valid_sets=ds,
|
||||
valid_names=names,
|
||||
feval=feval,
|
||||
callbacks=callbacks,
|
||||
**kwargs,
|
||||
)
|
||||
return booster, evals_result, list(names)
|
||||
|
||||
def _log_evals(self, evals_result, names: List[str], prefix: str = "") -> None:
|
||||
"""Log recorded evaluation curves to qlib's active recorder."""
|
||||
for k in names:
|
||||
for key, val in evals_result.get(k, {}).items():
|
||||
name = f"{prefix}{key}.{k}"
|
||||
for epoch, m in enumerate(val):
|
||||
R.log_metrics(**{name.replace("@", "_"): m}, step=epoch)
|
||||
|
||||
def fit(
|
||||
self,
|
||||
dataset: DatasetH,
|
||||
num_boost_round: Optional[int] = None,
|
||||
early_stopping_rounds: Optional[int] = None,
|
||||
verbose_eval: int = 20,
|
||||
evals_result=None,
|
||||
reweighter=None,
|
||||
**kwargs,
|
||||
):
|
||||
if evals_result is None:
|
||||
evals_result = {}
|
||||
ds_l = self._prepare_data(dataset, reweighter)
|
||||
self.model, evals_result, names = self._train_from_datasets(
|
||||
ds_l,
|
||||
num_boost_round=num_boost_round,
|
||||
early_stopping_rounds=early_stopping_rounds,
|
||||
verbose_eval=verbose_eval,
|
||||
evals_result=evals_result,
|
||||
**kwargs,
|
||||
)
|
||||
self._log_evals(evals_result, names)
|
||||
@@ -0,0 +1,3 @@
|
||||
from .optimal_stop import OptimalStopControl # noqa: F401
|
||||
|
||||
__all__ = ["OptimalStopControl"]
|
||||
@@ -0,0 +1,217 @@
|
||||
"""Optimal-stopping / stochastic-control strategy for cross-sectional signals.
|
||||
|
||||
Entry is a control policy: a symbol opens a position only when its cross-sectional
|
||||
signal percentile is at or above ``entry_pct`` (i.e. it is one of the top-ranked
|
||||
names) and the portfolio has fewer than ``topk`` open positions.
|
||||
|
||||
Exit is an optimal-stopping rule: a held position is stopped (closed) when its
|
||||
signal percentile falls below ``exit_pct`` (the continuation value of holding is
|
||||
no longer worth the risk), OR after ``max_hold_days`` (time stop / finite
|
||||
horizon), OR when the position P&L breaches ``sl`` (loss control) and the
|
||||
position has been held at least ``min_hold_days``.
|
||||
|
||||
Sizing is fixed ``notional`` per position (equal-weight control), unlike the
|
||||
TopkDropout cash-allocation heuristic.
|
||||
|
||||
Wired into qrun workflows like any ``BaseStrategy`` (see ``PortAnaRecord``
|
||||
config). Mirrors the API usage of qlib's ``TopkDropoutStrategy``: ``Order``/
|
||||
``OrderDir`` from ``qlib.backtest.decision``, ``trade_calendar`` /
|
||||
``trade_exchange`` / ``trade_position`` injected by the backtest executor.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import List
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from qlib.backtest import Order
|
||||
from qlib.backtest.decision import OrderDir, TradeDecisionWO
|
||||
from qlib.contrib.strategy.signal_strategy import BaseSignalStrategy
|
||||
|
||||
__all__ = ["OptimalStopControl"]
|
||||
|
||||
DEFAULT_NOTIONAL = 20_000.0
|
||||
DEFAULT_ENTRY_PCT = 0.80
|
||||
DEFAULT_EXIT_PCT = 0.50
|
||||
DEFAULT_MAX_HOLD_DAYS = 10
|
||||
DEFAULT_MIN_HOLD_DAYS = 2
|
||||
DEFAULT_SL = -0.06
|
||||
|
||||
|
||||
class OptimalStopControl(BaseSignalStrategy):
|
||||
"""Optimal-stopping long-only strategy over a cross-sectional signal.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
topk : max number of concurrent positions.
|
||||
entry_pct : min cross-sectional score percentile required to OPEN (0..1).
|
||||
exit_pct : held positions are stopped when score percentile < exit_pct.
|
||||
max_hold_days : hard time stop (finite-horizon close).
|
||||
min_hold_days : minimum holding days before stop-loss is evaluated.
|
||||
notional : $ per position (equal-weight control).
|
||||
sl : stop-loss threshold as fraction of entry price (<= 0), disabled if 0.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
signal=None,
|
||||
topk: int = 10,
|
||||
entry_pct: float = DEFAULT_ENTRY_PCT,
|
||||
exit_pct: float = DEFAULT_EXIT_PCT,
|
||||
max_hold_days: int = DEFAULT_MAX_HOLD_DAYS,
|
||||
min_hold_days: int = DEFAULT_MIN_HOLD_DAYS,
|
||||
notional: float = DEFAULT_NOTIONAL,
|
||||
sl: float = DEFAULT_SL,
|
||||
risk_degree: float = 0.95,
|
||||
trade_exchange=None,
|
||||
level_infra=None,
|
||||
common_infra=None,
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__(
|
||||
signal=signal,
|
||||
trade_exchange=trade_exchange,
|
||||
level_infra=level_infra,
|
||||
common_infra=common_infra,
|
||||
**kwargs,
|
||||
)
|
||||
self.topk = topk
|
||||
self.entry_pct = entry_pct
|
||||
self.exit_pct = exit_pct
|
||||
self.max_hold_days = max_hold_days
|
||||
self.min_hold_days = min_hold_days
|
||||
self.notional = notional
|
||||
self.sl = sl
|
||||
|
||||
# ------------------------------------------------------------------ utils
|
||||
@staticmethod
|
||||
def _pct_rank(score: pd.Series) -> pd.Series:
|
||||
return score.rank(pct=True)
|
||||
|
||||
def _entry_price(self, pos) -> float:
|
||||
# Position stores avg entry price under key "price" (see Position.position)
|
||||
price = pos.position.get("price")
|
||||
if price is None:
|
||||
price = pos.get_stock_amount("price")
|
||||
return float(price)
|
||||
|
||||
def _pnl_pct(self, pos, mark: float) -> float:
|
||||
entry = self._entry_price(pos)
|
||||
if not entry or entry != entry:
|
||||
return 0.0
|
||||
return mark / entry - 1.0
|
||||
|
||||
def _is_tradable(self, code, start, end, direction) -> bool:
|
||||
try:
|
||||
return self.trade_exchange.is_stock_tradable(
|
||||
stock_id=code, start_time=start, end_time=end, direction=direction
|
||||
)
|
||||
except TypeError: # some exchanges take no direction kwarg
|
||||
return self.trade_exchange.is_stock_tradable(stock_id=code, start_time=start, end_time=end)
|
||||
|
||||
# ------------------------------------------------------------ decision
|
||||
def generate_trade_decision(self, execute_result=None):
|
||||
trade_step = self.trade_calendar.get_trade_step()
|
||||
trade_start, trade_end = self.trade_calendar.get_step_time(trade_step)
|
||||
pred_start, pred_end = self.trade_calendar.get_step_time(trade_step, shift=1)
|
||||
pred_score = self.signal.get_signal(start_time=pred_start, end_time=pred_end)
|
||||
if isinstance(pred_score, pd.DataFrame):
|
||||
pred_score = pred_score.iloc[:, 0]
|
||||
if pred_score is None or len(pred_score) == 0:
|
||||
return TradeDecisionWO([], self)
|
||||
|
||||
pct = self._pct_rank(pred_score)
|
||||
time_per_step = self.trade_calendar.get_freq()
|
||||
current_temp = __import__("copy").deepcopy(self.trade_position)
|
||||
|
||||
holdings = {}
|
||||
for code in current_temp.get_stock_list():
|
||||
if abs(current_temp.get_stock_amount(code)) > 1e-6:
|
||||
holdings[code] = current_temp
|
||||
|
||||
# ---- optimal stopping: close held positions -----------------------
|
||||
sell_orders: List[Order] = []
|
||||
closed_today = set()
|
||||
kept = {}
|
||||
for code, pos in holdings.items():
|
||||
held = current_temp.get_stock_count(code, bar=time_per_step)
|
||||
mark = self.trade_exchange.get_deal_price(
|
||||
stock_id=code, start_time=trade_start, end_time=trade_end, direction=Order.SELL
|
||||
)
|
||||
if mark is None or mark != mark:
|
||||
continue
|
||||
rank = pct.get(code, 0.0)
|
||||
stop_pnl = held >= self.min_hold_days and self.sl < 0 and self._pnl_pct(pos, mark) <= self.sl
|
||||
if held >= self.max_hold_days or rank < self.exit_pct or stop_pnl:
|
||||
amt = abs(current_temp.get_stock_amount(code))
|
||||
o = Order(stock_id=code, amount=amt, start_time=trade_start,
|
||||
end_time=trade_end, direction=Order.SELL)
|
||||
if self.trade_exchange.check_order(o):
|
||||
sell_orders.append(o)
|
||||
self.trade_exchange.deal_order(o, position=current_temp)
|
||||
closed_today.add(code)
|
||||
else:
|
||||
kept[code] = mark
|
||||
|
||||
# ---- equal-weight control: target notional per name -----------------
|
||||
# candidate opens: top-ranked names whose signal pct >= entry_pct
|
||||
rank_desc = pred_score.sort_values(ascending=False)
|
||||
held_codes = set(kept)
|
||||
opens = []
|
||||
for sym in rank_desc.index:
|
||||
if len(opens) >= self.topk:
|
||||
break
|
||||
if sym in held_codes:
|
||||
continue
|
||||
if pct.get(sym, 0.0) < self.entry_pct:
|
||||
continue
|
||||
if not self._is_tradable(sym, trade_start, trade_end, OrderDir.BUY):
|
||||
continue
|
||||
opens.append(sym)
|
||||
|
||||
targets = held_codes | set(opens)
|
||||
if not targets:
|
||||
return TradeDecisionWO(sell_orders, self)
|
||||
|
||||
# total value (cash + marked positions) -> per-target notional
|
||||
total_value = current_temp.get_cash()
|
||||
for code, mark in kept.items():
|
||||
total_value += abs(current_temp.get_stock_amount(code)) * mark
|
||||
|
||||
target_notional = total_value * self.risk_degree / max(1, len(targets))
|
||||
|
||||
# ---- rebalance kept positions toward target weight ------------------
|
||||
buy_orders: List[Order] = []
|
||||
for code, mark in kept.items():
|
||||
cur = abs(current_temp.get_stock_amount(code)) * mark
|
||||
diff_notional = target_notional - cur
|
||||
if abs(diff_notional) / target_notional < 0.02:
|
||||
continue # skip tiny rebalances
|
||||
amount_delta = diff_notional / mark
|
||||
direction = Order.BUY if amount_delta > 0 else Order.SELL
|
||||
o = Order(stock_id=code, amount=abs(amount_delta), start_time=trade_start,
|
||||
end_time=trade_end, direction=direction)
|
||||
if self.trade_exchange.check_order(o):
|
||||
(buy_orders if direction == Order.BUY else sell_orders).append(o)
|
||||
self.trade_exchange.deal_order(o, position=current_temp)
|
||||
|
||||
# ---- open new positions at target weight ----------------------------
|
||||
for sym in opens:
|
||||
px = self.trade_exchange.get_deal_price(
|
||||
stock_id=sym, start_time=trade_start, end_time=trade_end, direction=OrderDir.BUY
|
||||
)
|
||||
if px is None or px != px or px <= 0:
|
||||
continue
|
||||
amount = target_notional / px
|
||||
factor = self.trade_exchange.get_factor(
|
||||
stock_id=sym, start_time=trade_start, end_time=trade_end
|
||||
)
|
||||
amount = self.trade_exchange.round_amount_by_trade_unit(amount, factor)
|
||||
o = Order(stock_id=sym, amount=amount, start_time=trade_start,
|
||||
end_time=trade_end, direction=Order.BUY)
|
||||
if self.trade_exchange.check_order(o):
|
||||
buy_orders.append(o)
|
||||
|
||||
return TradeDecisionWO(sell_orders + buy_orders, self)
|
||||
@@ -0,0 +1,25 @@
|
||||
from .config import (
|
||||
LakeConfig,
|
||||
BAR_FIELD_MAP,
|
||||
FREQ_TO_TIMEFRAME,
|
||||
UNKNOWN_FIELD_NAMES,
|
||||
timeframe_for_freq,
|
||||
resolve_lake_root,
|
||||
)
|
||||
from .providers import (
|
||||
LakeCalendarProvider,
|
||||
LakeInstrumentProvider,
|
||||
LakeFeatureProvider,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"LakeConfig",
|
||||
"BAR_FIELD_MAP",
|
||||
"FREQ_TO_TIMEFRAME",
|
||||
"UNKNOWN_FIELD_NAMES",
|
||||
"timeframe_for_freq",
|
||||
"resolve_lake_root",
|
||||
"LakeCalendarProvider",
|
||||
"LakeInstrumentProvider",
|
||||
"LakeFeatureProvider",
|
||||
]
|
||||
@@ -0,0 +1,202 @@
|
||||
"""TradeAC lake configuration helpers.
|
||||
|
||||
The lake is a hive-partitioned parquet store (see ``tac-engine/skills/tradeac-lake``):
|
||||
|
||||
$TAC_LAKE_DIR/
|
||||
├── market=US/
|
||||
│ └── timeframe=1d/
|
||||
│ └── symbol=AAPL.parquet # OHLCV bars: t, date, o, h, l, c, v, n, vw
|
||||
├── features/ # indicators, wide format, family tier
|
||||
│ └── market=US/
|
||||
│ └── timeframe=1d/
|
||||
│ ├── family=ta/symbol=AAPL.parquet # t, sma_5, sma_20, rsi_14, ...
|
||||
│ └── family=sp/symbol=AAPL.parquet # t, sp_ou_*, sp_hmm_*, ...
|
||||
├── calendar.parquet # trading days per market
|
||||
├── coverage.parquet # per (market,timeframe,symbol) loaded windows
|
||||
└── symbols.parquet # asset master
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
import pandas as pd
|
||||
|
||||
#: qlib freq string (Freq.__str__) -> lake timeframe partition name
|
||||
FREQ_TO_TIMEFRAME: Dict[str, str] = {
|
||||
"day": "1d",
|
||||
"1d": "1d",
|
||||
"min": "1m",
|
||||
"1min": "1m",
|
||||
"5min": "5m",
|
||||
"10min": "10m",
|
||||
"15min": "15m",
|
||||
"30min": "30m",
|
||||
"hour": "1h",
|
||||
"1hour": "1h",
|
||||
"2hour": "2h",
|
||||
"4hour": "4h",
|
||||
"week": "1w",
|
||||
"1week": "1w",
|
||||
"month": "1M",
|
||||
"1month": "1M",
|
||||
}
|
||||
|
||||
#: bar-field map: qlib field name (without the leading ``$``) -> lake bar column
|
||||
BAR_FIELD_MAP: Dict[str, str] = {
|
||||
"open": "o",
|
||||
"high": "h",
|
||||
"low": "l",
|
||||
"close": "c",
|
||||
"volume": "v",
|
||||
"vwap": "vw",
|
||||
"avg_amount": "vw", # amount / volume
|
||||
}
|
||||
|
||||
#: fields that qlib core/backtest queries but the lake does not store -> all-NaN
|
||||
UNKNOWN_FIELD_NAMES = ("factor", "change", "trade_unit", "suspend_flag")
|
||||
|
||||
#: columns in the parquet files that are not features
|
||||
NON_FEATURE_COLUMNS = ("t", "date", "market", "timeframe", "symbol")
|
||||
|
||||
|
||||
def timeframe_for_freq(freq: str) -> str:
|
||||
"""Map a qlib frequency (e.g. ``day``, ``1min``) to a lake timeframe (e.g. ``1d``)."""
|
||||
f = str(freq).lower()
|
||||
if f not in FREQ_TO_TIMEFRAME:
|
||||
raise ValueError(
|
||||
f"unsupported qlib freq {freq!r}; supported freqs: {sorted(set(FREQ_TO_TIMEFRAME))}"
|
||||
)
|
||||
return FREQ_TO_TIMEFRAME[f]
|
||||
|
||||
|
||||
def resolve_lake_root(lake_root: Optional[str] = None) -> Path:
|
||||
"""Resolve the lake root: explicit arg > ``TAC_LAKE_DIR`` (no fallback).
|
||||
|
||||
``TAC_LAKE_DIR`` is **mandatory** — there is deliberately no default
|
||||
A missing/empty value raises so a
|
||||
misconfigured environment never silently points at a wrong directory.
|
||||
"""
|
||||
if lake_root is None:
|
||||
lake_root = os.environ.get("TAC_LAKE_DIR")
|
||||
if not lake_root:
|
||||
raise RuntimeError(
|
||||
"TAC_LAKE_DIR is not set. Point it at the TradeAC lake root, e.g. "
|
||||
"export TAC_LAKE_DIR=/home/data/lake (docker) or set an absolute "
|
||||
"path in your local .env."
|
||||
)
|
||||
return Path(str(lake_root)).expanduser().resolve()
|
||||
|
||||
|
||||
class LakeConfig:
|
||||
"""Path helpers + cached readers for a (lake_root, market) combination."""
|
||||
|
||||
def __init__(self, lake_root: Optional[str] = None, market: str = "US"):
|
||||
self.lake_root: Path = resolve_lake_root(lake_root)
|
||||
self.market: str = (market or "US").upper()
|
||||
|
||||
# ---- paths --------------------------------------------------------------
|
||||
def bar_dir(self, timeframe: str) -> Path:
|
||||
return self.lake_root / f"market={self.market}" / f"timeframe={timeframe}"
|
||||
|
||||
def bar_path(self, timeframe: str, symbol: str) -> Path:
|
||||
return self.bar_dir(timeframe) / f"symbol={str(symbol).upper()}.parquet"
|
||||
|
||||
def features_dir(self, timeframe: str) -> Path:
|
||||
return self.lake_root / "features" / f"market={self.market}" / f"timeframe={timeframe}"
|
||||
|
||||
def features_path(self, timeframe: str, symbol: str) -> Path:
|
||||
# Legacy flat path (no family tier). Prefer `load_features` which
|
||||
# resolves the family=ta|sp partition layout.
|
||||
return self.features_dir(timeframe) / f"symbol={str(symbol).upper()}.parquet"
|
||||
|
||||
def load_features(self, timeframe: str, symbol: str) -> pd.DataFrame:
|
||||
"""All feature columns for a symbol, merging the `family=ta` and
|
||||
`family=sp` partitions by timestamp. Returns an empty frame when no
|
||||
feature files exist (legacy flat layout falls back transparently)."""
|
||||
sym = str(symbol).upper()
|
||||
frames = []
|
||||
for family in ("ta", "sp"):
|
||||
p = self.features_dir(timeframe) / f"family={family}" / f"symbol={sym}.parquet"
|
||||
if p.exists():
|
||||
frames.append(pd.read_parquet(p))
|
||||
if not frames:
|
||||
flat = self.features_dir(timeframe) / f"symbol={sym}.parquet"
|
||||
if flat.exists():
|
||||
return pd.read_parquet(flat)
|
||||
return pd.DataFrame()
|
||||
if len(frames) == 1:
|
||||
return frames[0]
|
||||
merged = frames[0]
|
||||
for extra in frames[1:]:
|
||||
merged = merged.merge(extra, on="t", how="outer", suffixes=("", "_dup"))
|
||||
for c in [c for c in merged.columns if c.endswith("_dup")]:
|
||||
merged = merged.drop(columns=c)
|
||||
return merged
|
||||
|
||||
def calendar_path(self) -> Path:
|
||||
return self.lake_root / "calendar.parquet"
|
||||
|
||||
def symbols_path(self) -> Path:
|
||||
return self.lake_root / "symbols.parquet"
|
||||
|
||||
def coverage_path(self) -> Path:
|
||||
return self.lake_root / "coverage.parquet"
|
||||
|
||||
# ---- metadata readers ----------------------------------------------------
|
||||
def load_symbols(self) -> List[str]:
|
||||
"""All symbols known to the lake (from ``symbols.parquet``)."""
|
||||
p = self.symbols_path()
|
||||
if not p.exists():
|
||||
return []
|
||||
df = pd.read_parquet(p)
|
||||
if "symbol" not in df.columns:
|
||||
return []
|
||||
return sorted(df["symbol"].astype(str).str.upper().tolist())
|
||||
|
||||
def symbol_spans(self, symbol: str, timeframe: str) -> List[tuple]:
|
||||
"""Listing span(s) ``[(start_iso, end_iso)]`` for a symbol from coverage.parquet."""
|
||||
p = self.coverage_path()
|
||||
if p.exists():
|
||||
try:
|
||||
df = pd.read_parquet(p)
|
||||
except Exception: # pragma: no cover - defensive
|
||||
df = pd.DataFrame()
|
||||
if len(df):
|
||||
df = df[
|
||||
(df.get("market") == self.market)
|
||||
& (df.get("timeframe") == timeframe)
|
||||
& (df.get("symbol") == str(symbol).upper())
|
||||
]
|
||||
if len(df):
|
||||
row = df.iloc[0]
|
||||
first = pd.Timestamp(row["first_t"]).date()
|
||||
last = pd.Timestamp(row["last_t"]).date()
|
||||
return [(first.isoformat(), last.isoformat())]
|
||||
# fallback: derive from the bar file itself
|
||||
p = self.bar_path(timeframe, symbol)
|
||||
if p.exists():
|
||||
import pyarrow.parquet as pq
|
||||
|
||||
tbl = pq.read_table(p, columns=["t"])
|
||||
first = pd.Timestamp(tbl.column("t")[0].as_py()).date()
|
||||
last = pd.Timestamp(tbl.column("t")[-1].as_py()).date()
|
||||
return [(first.isoformat(), last.isoformat())]
|
||||
return [("1970-01-01", "2099-12-31")]
|
||||
|
||||
def load_calendar_dates(self) -> List[pd.Timestamp]:
|
||||
"""Trading days (midnight timestamps) for the market, from ``calendar.parquet``."""
|
||||
p = self.calendar_path()
|
||||
if p.exists():
|
||||
df = pd.read_parquet(p)
|
||||
if "date" in df.columns:
|
||||
if "market" in df.columns:
|
||||
df = df[df["market"] == self.market]
|
||||
dates = pd.to_datetime(df["date"]).dt.normalize().sort_values().unique()
|
||||
return [pd.Timestamp(x) for x in dates]
|
||||
return []
|
||||
|
||||
def __repr__(self) -> str: # pragma: no cover
|
||||
return f"LakeConfig(lake_root={self.lake_root}, market={self.market})"
|
||||
@@ -0,0 +1,230 @@
|
||||
"""qlib data providers backed by the TradeAC parquet lake.
|
||||
|
||||
These providers plug into the standard qlib mechanism: ``qlib.init(calendar_provider=...,
|
||||
instrument_provider=..., feature_provider=...)`` instantiates them and binds them to the
|
||||
``Cal`` / ``Inst`` / ``FeatureD`` wrappers (see ``qlib.data.data.register_all_wrappers``).
|
||||
The rest of qlib (``LocalDatasetProvider`` expression engine, backtest ``Exchange``) keeps
|
||||
working unchanged because the interface contract is identical to the file-based providers:
|
||||
|
||||
- ``feature()`` returns a ``pd.Series`` indexed by the **calendar position** range
|
||||
``[start_index, end_index]`` (matching ``FileFeatureStorage.__getitem__`` semantics).
|
||||
- ``list_instruments()`` returns ``{symbol: [(start, end), ...]}``.
|
||||
- ``load_calendar()`` returns a list of ``pd.Timestamp`` trading days.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import bisect
|
||||
from typing import Dict, List, Optional, Union
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from qlib.data.data import CalendarProvider, FeatureProvider, InstrumentProvider
|
||||
from qlib.log import get_module_logger
|
||||
|
||||
from .config import (
|
||||
BAR_FIELD_MAP,
|
||||
LakeConfig,
|
||||
UNKNOWN_FIELD_NAMES,
|
||||
timeframe_for_freq,
|
||||
)
|
||||
|
||||
logger = get_module_logger("tac_qlib.data.providers")
|
||||
|
||||
|
||||
def _day_freq(freq: str) -> bool:
|
||||
return str(freq).lower() in ("day", "1d")
|
||||
|
||||
|
||||
def _calendar_keys(cal: List[pd.Timestamp], freq: str) -> pd.Index:
|
||||
"""Convert calendar timestamps into the same key space as the lake parquet."""
|
||||
if _day_freq(freq):
|
||||
return pd.Index([pd.Timestamp(x).date() for x in cal])
|
||||
return pd.Index([pd.Timestamp(x) for x in cal])
|
||||
|
||||
|
||||
class LakeCalendarProvider(CalendarProvider):
|
||||
"""Trading calendar read from ``<lake>/calendar.parquet`` (fallback: derived from bars)."""
|
||||
|
||||
def __init__(self, lake_root: Optional[str] = None, market: str = "US"):
|
||||
super().__init__()
|
||||
self.cfg = LakeConfig(lake_root, market)
|
||||
|
||||
def load_calendar(self, freq, future):
|
||||
timeframe = timeframe_for_freq(freq)
|
||||
if not _day_freq(freq):
|
||||
raise NotImplementedError(
|
||||
f"freq={freq!r} (timeframe={timeframe}) is not supported yet: the lake calendar "
|
||||
f"only covers daily sessions; add a minute-level calendar to `calendar.parquet`"
|
||||
)
|
||||
|
||||
dates = self.cfg.load_calendar_dates()
|
||||
if not dates:
|
||||
# Fallback: derive the trading-day set from the persisted bar files.
|
||||
bar_dir = self.cfg.bar_dir(timeframe)
|
||||
if bar_dir.exists():
|
||||
import pyarrow.parquet as pq
|
||||
|
||||
cal: Dict[pd.Timestamp, None] = {}
|
||||
for p in sorted(bar_dir.glob("symbol=*.parquet")):
|
||||
tbl = pq.read_table(p, columns=["t"])
|
||||
for v in tbl.column("t"):
|
||||
cal[pd.Timestamp(v.as_py()).normalize()] = None
|
||||
dates = sorted(cal.keys())
|
||||
if not dates:
|
||||
return []
|
||||
|
||||
if future:
|
||||
# append the next calendar day so that "today" is a valid trade date
|
||||
last = dates[-1]
|
||||
dates = dates + [pd.Timestamp(last) + pd.Timedelta(days=1)]
|
||||
return dates
|
||||
|
||||
|
||||
class LakeInstrumentProvider(InstrumentProvider):
|
||||
"""Instruments from ``<lake>/symbols.parquet`` with listing spans from ``coverage.parquet``."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
lake_root: Optional[str] = None,
|
||||
market: str = "US",
|
||||
markets: Optional[Dict[str, list]] = None,
|
||||
):
|
||||
super().__init__()
|
||||
self.cfg = LakeConfig(lake_root, market)
|
||||
#: optional named pools, e.g. ``{"sp500": ["AAPL", "MSFT"], "etf": ["SPY"]}``.
|
||||
#: ``all`` / any unregistered name resolves to every symbol in the lake.
|
||||
self.markets: Dict[str, list] = markets or {}
|
||||
|
||||
def _resolve_symbols(self, market: Union[str, list]) -> List[str]:
|
||||
if isinstance(market, (list, tuple, pd.Index, np.ndarray)):
|
||||
return [str(s).upper() for s in market]
|
||||
if isinstance(market, str) and "," in market:
|
||||
return [s.strip().upper() for s in market.split(",") if s.strip()]
|
||||
if market in self.markets:
|
||||
return [str(s).upper() for s in self.markets[market]]
|
||||
return self.cfg.load_symbols()
|
||||
|
||||
def list_instruments(self, instruments, start_time=None, end_time=None, freq="day", as_list=False):
|
||||
market = instruments["market"]
|
||||
timeframe = timeframe_for_freq(freq)
|
||||
|
||||
symbols = self._resolve_symbols(market)
|
||||
if not symbols:
|
||||
if as_list:
|
||||
return []
|
||||
return {}
|
||||
|
||||
# clip listing spans to the queried window (mirror of LocalInstrumentProvider)
|
||||
from qlib.data.data import Cal # pylint: disable=C0415
|
||||
|
||||
cal = Cal.calendar(freq=freq)
|
||||
start_time = pd.Timestamp(start_time or cal[0])
|
||||
end_time = pd.Timestamp(end_time or cal[-1])
|
||||
|
||||
out: Dict[str, list] = {}
|
||||
for symbol in symbols:
|
||||
spans = []
|
||||
for begin, end in self.cfg.symbol_spans(symbol, timeframe):
|
||||
lo = max(start_time, pd.Timestamp(begin))
|
||||
hi = min(end_time, pd.Timestamp(end))
|
||||
if lo <= hi:
|
||||
spans.append((lo, hi))
|
||||
if spans:
|
||||
out[symbol] = spans
|
||||
|
||||
filter_pipe = instruments.get("filter_pipe") or []
|
||||
for filter_config in filter_pipe:
|
||||
from qlib.data import filter as F # pylint: disable=C0415
|
||||
|
||||
filter_t = getattr(F, filter_config["filter_type"]).from_config(filter_config)
|
||||
out = filter_t(out, start_time, end_time, freq)
|
||||
|
||||
if as_list:
|
||||
return list(out)
|
||||
return out
|
||||
|
||||
|
||||
class LakeFeatureProvider(FeatureProvider):
|
||||
"""Feature data from the lake parquet (OHLCV bars + pre-computed ta-lib features).
|
||||
|
||||
Field routing:
|
||||
- ``$open/$high/$low/$close/$volume/$vwap`` -> bar parquet columns
|
||||
- ``$amount`` (= v*vw), ``$avg_amount`` (= vw) -> derived from bar parquet
|
||||
- ``$factor/$change/...`` -> all-NaN (not stored)
|
||||
- anything else -> a ta-lib column in the features parquet
|
||||
"""
|
||||
|
||||
def __init__(self, lake_root: Optional[str] = None, market: str = "US"):
|
||||
super().__init__()
|
||||
self.cfg = LakeConfig(lake_root, market)
|
||||
self._bar_cache: Dict[tuple, pd.DataFrame] = {}
|
||||
self._feature_cache: Dict[tuple, pd.DataFrame] = {}
|
||||
|
||||
# ------------------------------------------------------------------ caches
|
||||
def _load_bar_df(self, instrument: str, timeframe: str) -> pd.DataFrame:
|
||||
key = (instrument, timeframe)
|
||||
if key not in self._bar_cache:
|
||||
p = self.cfg.bar_path(timeframe, instrument)
|
||||
self._bar_cache[key] = pd.read_parquet(p) if p.exists() else pd.DataFrame()
|
||||
return self._bar_cache[key]
|
||||
|
||||
def _load_feature_df(self, instrument: str, timeframe: str) -> pd.DataFrame:
|
||||
key = (instrument, timeframe)
|
||||
if key not in self._feature_cache:
|
||||
self._feature_cache[key] = self.cfg.load_features(timeframe, instrument)
|
||||
return self._feature_cache[key]
|
||||
|
||||
@staticmethod
|
||||
def _keys(df: pd.DataFrame, freq: str) -> pd.Index:
|
||||
ts = pd.to_datetime(df["t"])
|
||||
return ts.dt.date if _day_freq(freq) else ts
|
||||
|
||||
# ------------------------------------------------------------------ fields
|
||||
def _extract(self, instrument: str, field: str, timeframe: str, freq: str) -> Optional[pd.Series]:
|
||||
"""Return the field as a Series keyed by date/timestamp (None if not present in the lake)."""
|
||||
bar = self._load_bar_df(instrument, timeframe)
|
||||
|
||||
if field in BAR_FIELD_MAP:
|
||||
col = BAR_FIELD_MAP[field]
|
||||
if col in bar.columns:
|
||||
return bar[col].astype(float).set_axis(self._keys(bar, freq))
|
||||
return None
|
||||
if field == "amount":
|
||||
if "v" in bar.columns and "vw" in bar.columns:
|
||||
return (bar["v"] * bar["vw"]).astype(float).set_axis(self._keys(bar, freq))
|
||||
return None
|
||||
if field in UNKNOWN_FIELD_NAMES:
|
||||
return None
|
||||
|
||||
feat = self._load_feature_df(instrument, timeframe)
|
||||
if field in feat.columns:
|
||||
return feat[field].astype(float).set_axis(self._keys(feat, freq))
|
||||
return None
|
||||
|
||||
# ------------------------------------------------------------------ api
|
||||
def _get_calendar(self, freq: str) -> List[pd.Timestamp]:
|
||||
from qlib.data.data import Cal # pylint: disable=C0415
|
||||
|
||||
cal = Cal.calendar(freq=freq)
|
||||
return list(cal)
|
||||
|
||||
def feature(self, instrument, field, start_index, end_index, freq):
|
||||
field = str(field)[1:]
|
||||
timeframe = timeframe_for_freq(freq)
|
||||
|
||||
cal = self._get_calendar(freq)
|
||||
n = len(cal)
|
||||
lo = max(0, int(start_index))
|
||||
hi = min(n - 1, int(end_index))
|
||||
if lo > hi:
|
||||
return pd.Series(dtype=np.float32)
|
||||
|
||||
keys = _calendar_keys(cal[lo : hi + 1], freq)
|
||||
ser = self._extract(str(instrument).upper(), field, timeframe, freq)
|
||||
if ser is None:
|
||||
vals = np.full(len(keys), np.nan, dtype=np.float64)
|
||||
else:
|
||||
vals = ser.reindex(keys).to_numpy(dtype=np.float64)
|
||||
return pd.Series(vals, index=pd.RangeIndex(lo, hi + 1))
|
||||
@@ -0,0 +1,64 @@
|
||||
"""Drop-in replacement for ``qlib.init`` that configures qlib against the TradeAC lake.
|
||||
|
||||
Usage::
|
||||
|
||||
from tac_qlib.qlib_init import qlib_init
|
||||
|
||||
qlib_init(
|
||||
provider_uri="/path/to/lake", # same layout as tac-engine's TAC_LAKE_DIR
|
||||
market="US",
|
||||
freq="day",
|
||||
markets={"sp500": ["AAPL", "MSFT"]}, # optional named instrument pools
|
||||
**qlib_init_kwargs, # anything qlib.init accepts
|
||||
)
|
||||
|
||||
It sets ``provider_uri`` to the lake root and points the calendar / instrument / feature
|
||||
providers at the lake-backed implementations, then delegates to the upstream ``qlib.init``.
|
||||
The dataset provider (expression engine, backtest ``Exchange``) is left untouched, so the
|
||||
rest of the qlib workflow is byte-for-byte upstream code.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Dict, List, Optional, Union
|
||||
|
||||
import qlib
|
||||
from qlib.config import C
|
||||
|
||||
from .data.config import LakeConfig, resolve_lake_root
|
||||
|
||||
PROVIDERS = "tac_qlib.data.providers"
|
||||
|
||||
|
||||
def provider_config(cls: str, **kwargs) -> dict:
|
||||
return {"class": f"{PROVIDERS}.{cls}", "kwargs": kwargs}
|
||||
|
||||
|
||||
def qlib_init(
|
||||
provider_uri: Optional[str] = None,
|
||||
market: str = "US",
|
||||
freq: str = "day",
|
||||
markets: Optional[Dict[str, list]] = None,
|
||||
**qlib_kwargs,
|
||||
) -> qlib.Initialized:
|
||||
"""Initialize qlib with the lake-backed data providers and re-export qlib.init results."""
|
||||
if qlib_kwargs.pop("calendar_provider", None) is not None or qlib_kwargs.pop("instrument_provider", None) is not None:
|
||||
raise ValueError("calendar_provider / instrument_provider are managed by tac_qlib; use `market` instead")
|
||||
|
||||
lake_root = resolve_lake_root(provider_uri)
|
||||
if freq != "day":
|
||||
raise ValueError("freq must be 'day' for now: the lake calendar only covers daily sessions")
|
||||
|
||||
qlib_kwargs.setdefault("provider_uri", lake_root)
|
||||
qlib_kwargs.setdefault("region", "us")
|
||||
qlib_kwargs.setdefault("expression_cache", None)
|
||||
qlib_kwargs.setdefault("dataset_cache", None)
|
||||
qlib_kwargs["calendar_provider"] = provider_config("LakeCalendarProvider", lake_root=lake_root, market=market)
|
||||
qlib_kwargs["instrument_provider"] = provider_config(
|
||||
"LakeInstrumentProvider", lake_root=lake_root, market=market, markets=markets or {}
|
||||
)
|
||||
qlib_kwargs["feature_provider"] = provider_config("LakeFeatureProvider", lake_root=lake_root, market=market)
|
||||
return qlib.init(**qlib_kwargs)
|
||||
|
||||
|
||||
__all__ = ["qlib_init", "provider_config"]
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,144 @@
|
||||
"""Risk-limit spec shared by backtest and live executor.
|
||||
|
||||
One JSON spec is consulted by BOTH ``rd_backtest`` (as a strategy filter
|
||||
overlay) and ``rd_strategy_targets`` (as pre-gate + sizing caps), so a limit
|
||||
that holds in backtest holds in live — the round's ``strategy_snapshot``
|
||||
stores the exact spec used.
|
||||
|
||||
Supported keys (all optional, all pct are 0-100):
|
||||
liquidity_floor_adv : min avg daily dollar volume (USD) per symbol.
|
||||
Names below it are filtered out of the tradable set.
|
||||
size_cap_pct : max notional per name as % of account equity.
|
||||
concentration_cap_pct: max total deployed as % of account equity.
|
||||
drawdown_pause_pct : if equity drawdown from peak exceeds this, new buys
|
||||
are paused (executor gate; not expressible in a
|
||||
one-shot qlib backtest and therefore documented).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
import pandas as pd
|
||||
|
||||
|
||||
def parse_limits(spec: Optional[str]) -> Dict[str, float]:
|
||||
"""Parse a risk_limits JSON string into a flat float map (empty = no limits)."""
|
||||
if not spec or not str(spec).strip():
|
||||
return {}
|
||||
if isinstance(spec, dict):
|
||||
raw = spec
|
||||
else:
|
||||
raw = json.loads(str(spec))
|
||||
out: Dict[str, float] = {}
|
||||
for k in ("liquidity_floor_adv", "size_cap_pct", "concentration_cap_pct", "drawdown_pause_pct"):
|
||||
v = raw.get(k)
|
||||
if v is not None and str(v) != "":
|
||||
out[k] = float(v)
|
||||
return out
|
||||
|
||||
|
||||
def dollar_adv(
|
||||
symbols: List[str],
|
||||
lake_root: str = "",
|
||||
market: str = "US",
|
||||
asof: Optional[str] = None,
|
||||
lookback: int = 20,
|
||||
) -> Dict[str, float]:
|
||||
"""Average daily dollar volume per symbol over the ``lookback`` sessions
|
||||
ending at ``asof`` (inclusive), read straight from lake 1d bars. Symbols
|
||||
with no lake data map to 0.0 (treated as illiquid)."""
|
||||
from tac_qlib.data.config import LakeConfig, resolve_lake_root
|
||||
|
||||
cfg = LakeConfig(resolve_lake_root(lake_root or None), market)
|
||||
asof_ts = pd.Timestamp(asof) if asof else pd.Timestamp.utcnow()
|
||||
out: Dict[str, float] = {}
|
||||
for sym in sorted({str(s).upper() for s in symbols}):
|
||||
p = cfg.bar_path("1d", sym)
|
||||
if not p.exists():
|
||||
out[sym] = 0.0
|
||||
continue
|
||||
try:
|
||||
df = pd.read_parquet(p)
|
||||
except Exception:
|
||||
out[sym] = 0.0
|
||||
continue
|
||||
if not len(df):
|
||||
out[sym] = 0.0
|
||||
continue
|
||||
tcol = df["t"] if "t" in df.columns else df["date"]
|
||||
ts = pd.to_datetime(tcol)
|
||||
df = df.assign(_t=ts).sort_values("_t")
|
||||
df = df[df["_t"] <= asof_ts]
|
||||
if not len(df):
|
||||
out[sym] = 0.0
|
||||
continue
|
||||
df = df.tail(lookback)
|
||||
px = df["vw"] if "vw" in df.columns else df["c"]
|
||||
out[sym] = float((df["v"] * px).mean()) if len(df) else 0.0
|
||||
return out
|
||||
|
||||
|
||||
def apply_to_ranking(
|
||||
ranking: pd.Series,
|
||||
adv: Dict[str, float],
|
||||
limits: Dict[str, float],
|
||||
account: float,
|
||||
risk_degree: float,
|
||||
topk: int,
|
||||
) -> Tuple[pd.Series, Dict[str, Any]]:
|
||||
"""Executor-side overlay on the ranked signal (``pd.Series`` symbol -> score).
|
||||
|
||||
Returns (filtered_ranking, applied) where ``filtered_ranking`` has
|
||||
illiquid names removed and ``applied`` records what the limits did (audit
|
||||
trail). Per-name notional and total caps are reported but not folded into
|
||||
the ranking — the caller sizes targets and can read ``applied`` to cap.
|
||||
"""
|
||||
applied: Dict[str, Any] = {"notes": [], "dropped_liquidity": []}
|
||||
filtered = ranking
|
||||
floor = limits.get("liquidity_floor_adv")
|
||||
if floor:
|
||||
dropped = [s for s in filtered.index if adv.get(str(s).upper(), 0.0) < floor]
|
||||
if dropped:
|
||||
filtered = filtered.drop(index=[s for s in dropped if s in filtered.index])
|
||||
applied["dropped_liquidity"] = [str(s) for s in dropped]
|
||||
applied["notes"].append(f"liquidity floor ${floor:,.0f} ADV dropped {len(dropped)}")
|
||||
per_name = account * risk_degree / max(topk, 1)
|
||||
size_cap = limits.get("size_cap_pct")
|
||||
if size_cap:
|
||||
cap = account * size_cap / 100.0
|
||||
applied["size_cap_notional"] = round(cap, 2)
|
||||
if per_name > cap:
|
||||
applied["per_name_capped_from"] = round(per_name, 2)
|
||||
per_name = cap
|
||||
applied["notes"].append(f"size cap {size_cap:g}% cut per-name notional to ${cap:,.2f}")
|
||||
applied["per_name_notional"] = round(per_name, 2)
|
||||
n_buys = min(topk, max(len(filtered), 0))
|
||||
conc = limits.get("concentration_cap_pct")
|
||||
if conc:
|
||||
conc_cap = account * conc / 100.0
|
||||
applied["concentration_cap_notional"] = round(conc_cap, 2)
|
||||
total = per_name * max(n_buys, 1)
|
||||
if total > conc_cap:
|
||||
applied["total_capped_from"] = round(total, 2)
|
||||
applied["notes"].append(f"concentration cap {conc:g}% cut total to ${conc_cap:,.2f}")
|
||||
per_name = conc_cap / max(n_buys, 1)
|
||||
applied["per_name_capped_from"] = applied.get("per_name_capped_from") or round(total / max(n_buys, 1), 2)
|
||||
applied["per_name_notional"] = round(per_name, 2)
|
||||
applied["total_notional"] = round(min(total, conc_cap), 2)
|
||||
else:
|
||||
applied["total_notional"] = round(per_name * n_buys, 2)
|
||||
return filtered, applied
|
||||
|
||||
|
||||
def drawdown_pause(equity: float, peak_equity: float, limits: Dict[str, float]) -> Tuple[bool, Optional[str]]:
|
||||
"""Executor gate: True when drawdown from peak exceeds drawdown_pause_pct."""
|
||||
pct = limits.get("drawdown_pause_pct")
|
||||
if not pct or not peak_equity or not equity:
|
||||
return False, None
|
||||
dd = (peak_equity - equity) / peak_equity * 100.0
|
||||
if dd >= pct:
|
||||
return True, f"drawdown {dd:.1f}% >= pause {pct:g}% (peak ${peak_equity:,.2f}, equity ${equity:,.2f})"
|
||||
return False, None
|
||||
@@ -0,0 +1,635 @@
|
||||
"""Trace tools for the tac-qlib-rd MCP server — replace the `trace.sh`/`trace_db.py`/`git_exp.sh` scripts.
|
||||
|
||||
The R&D lineage (`/rd/lineage`) and the round book build on the `rd_experiments`
|
||||
Postgres table. This module exposes the full trace lifecycle as MCP tools so an
|
||||
agent can drive tracing through the long-lived tac-qlib-rd server instead of
|
||||
shelling out to bash scripts (which re-import psycopg + reconnect per call and
|
||||
force the agent to parse prose output).
|
||||
|
||||
Because the server is long-lived, `psycopg` is imported and the DB connection
|
||||
is opened once per call (not once per script invocation), and every tool returns
|
||||
a single JSON object — no output parsing, fully deterministic.
|
||||
|
||||
Git operations (fork / commit / push on the `experiments/` clone) are performed
|
||||
via `git` subprocess with the repo's mandated credential helper, exactly as the
|
||||
old `git_exp.sh` did.
|
||||
|
||||
Env (from the repo `.env`, already loaded by rd_server): DATABASE_URL,
|
||||
EMBEDDING_API_BASE_URL, EMBEDDING_API_KEY, GIT_USER, GIT_PASS, GIT_REPO_URL,
|
||||
TAC_LAKE_DIR.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import sqlite3
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
import psycopg
|
||||
from psycopg.rows import dict_row
|
||||
|
||||
from tac_qlib.trace_embed import embed
|
||||
|
||||
EMBEDDING_DIM = 384
|
||||
_MIN_SCORE = 0.5
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- db
|
||||
def _conn():
|
||||
url = (os.environ.get("DATABASE_URL") or "").strip()
|
||||
if not url:
|
||||
raise RuntimeError("DATABASE_URL is not set")
|
||||
return psycopg.connect(url, row_factory=dict_row)
|
||||
|
||||
|
||||
def _now_iso() -> str:
|
||||
from datetime import datetime, timezone
|
||||
|
||||
return datetime.now(timezone.utc).isoformat()
|
||||
|
||||
|
||||
def _vector_literal(vec: Optional[List[float]]) -> Optional[str]:
|
||||
if not vec:
|
||||
return None
|
||||
return "[" + ",".join(repr(float(v)) for v in vec) + "]"
|
||||
|
||||
|
||||
def _jsonable(obj: Any) -> Any:
|
||||
if isinstance(obj, dict):
|
||||
return {k: _jsonable(v) for k, v in obj.items()}
|
||||
if isinstance(obj, (list, tuple, set)):
|
||||
return [_jsonable(v) for v in obj]
|
||||
if hasattr(obj, "isoformat"):
|
||||
return obj.isoformat()
|
||||
return obj
|
||||
|
||||
|
||||
def _row_json(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
out: Dict[str, Any] = {}
|
||||
for k, v in row.items():
|
||||
if k in ("rational_embedding", "details_embedding"):
|
||||
out[k] = v.tolist() if hasattr(v, "tolist") else v
|
||||
elif k == "metrics" and isinstance(v, str):
|
||||
try:
|
||||
out[k] = json.loads(v)
|
||||
except Exception: # noqa: BLE001
|
||||
out[k] = v
|
||||
else:
|
||||
out[k] = v
|
||||
return out
|
||||
|
||||
|
||||
def _get_row(exp_id: int) -> Optional[Dict[str, Any]]:
|
||||
with _conn() as conn, conn.cursor() as cur:
|
||||
cur.execute("SELECT * FROM rd_experiments WHERE id = %s", (exp_id,))
|
||||
return cur.fetchone()
|
||||
|
||||
|
||||
def _all_rows(limit: int) -> List[Dict[str, Any]]:
|
||||
with _conn() as conn, conn.cursor() as cur:
|
||||
cur.execute("SELECT * FROM rd_experiments ORDER BY id DESC LIMIT %s", (limit,))
|
||||
return cur.fetchall()
|
||||
|
||||
|
||||
def _init_db() -> None:
|
||||
with _conn() as conn, conn.cursor() as cur:
|
||||
cur.execute("SELECT 1 FROM pg_extension WHERE extname = 'vector'")
|
||||
if not cur.fetchone():
|
||||
raise RuntimeError("pgvector extension is not installed. Run: CREATE EXTENSION IF NOT EXISTS vector;")
|
||||
cur.execute("SELECT to_regclass('public.rd_experiments')")
|
||||
exists = bool(cur.fetchone())
|
||||
if not exists:
|
||||
DDL = """
|
||||
CREATE TABLE IF NOT EXISTS rd_experiments (
|
||||
id bigserial PRIMARY KEY NOT NULL,
|
||||
experiment_name text,
|
||||
rational text NOT NULL,
|
||||
rational_embedding vector(384),
|
||||
details text,
|
||||
details_embedding vector(384),
|
||||
evaluation text,
|
||||
metrics jsonb,
|
||||
evolved_from bigint,
|
||||
start_ts timestamptz DEFAULT now() NOT NULL,
|
||||
end_ts timestamptz,
|
||||
git_branch text NOT NULL,
|
||||
experiment_ref_id text,
|
||||
session_id text,
|
||||
mlruns_dir text,
|
||||
status text DEFAULT 'starting' NOT NULL,
|
||||
created_at timestamptz DEFAULT now() NOT NULL,
|
||||
updated_at timestamptz DEFAULT now() NOT NULL
|
||||
);
|
||||
"""
|
||||
for stmt in DDL.split(";"):
|
||||
stmt = stmt.strip()
|
||||
if stmt:
|
||||
cur.execute(stmt)
|
||||
# Idempotent backfills so pre-existing tables gain new columns.
|
||||
for stmt in ["ALTER TABLE rd_experiments ADD COLUMN IF NOT EXISTS session_id text;"]:
|
||||
stmt = stmt.strip()
|
||||
if stmt:
|
||||
cur.execute(stmt)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _search_evolved_from(text: str, limit: int = 5) -> Optional[int]:
|
||||
vec = embed(text)
|
||||
if not vec:
|
||||
return None
|
||||
lit = _vector_literal(vec)
|
||||
with _conn() as conn, conn.cursor() as cur:
|
||||
cur.execute(
|
||||
"""
|
||||
SELECT id, 1 - LEAST(
|
||||
COALESCE(rational_embedding <=> %s::vector, 1),
|
||||
COALESCE(details_embedding <=> %s::vector, 1)
|
||||
) AS similarity
|
||||
FROM rd_experiments
|
||||
ORDER BY similarity DESC
|
||||
LIMIT %s
|
||||
""",
|
||||
(lit, lit, limit),
|
||||
)
|
||||
rows = cur.fetchall()
|
||||
for row in rows:
|
||||
if row["similarity"] is not None and float(row["similarity"]) >= _MIN_SCORE:
|
||||
return int(row["id"])
|
||||
return None
|
||||
|
||||
|
||||
def _search(query: str, limit: int = 10, min_score: float = _MIN_SCORE) -> List[Dict[str, Any]]:
|
||||
vec = embed(query)
|
||||
if not vec:
|
||||
needle = f"%{query.replace('%', ' ').strip()}%"
|
||||
with _conn() as conn, conn.cursor() as cur:
|
||||
cur.execute(
|
||||
"""
|
||||
SELECT id, rational, details, git_branch, experiment_ref_id, status,
|
||||
start_ts, end_ts, evaluation, session_id
|
||||
FROM rd_experiments
|
||||
WHERE rational ILIKE %s OR details ILIKE %s
|
||||
ORDER BY id DESC LIMIT %s
|
||||
""",
|
||||
(needle, needle, limit),
|
||||
)
|
||||
return [_row_json(r) for r in cur.fetchall()]
|
||||
lit = _vector_literal(vec)
|
||||
with _conn() as conn, conn.cursor() as cur:
|
||||
cur.execute(
|
||||
"""
|
||||
SELECT id, rational, details, git_branch, experiment_ref_id, status,
|
||||
start_ts, end_ts, evaluation, session_id,
|
||||
1 - LEAST(
|
||||
COALESCE(rational_embedding <=> %s::vector, 1),
|
||||
COALESCE(details_embedding <=> %s::vector, 1)
|
||||
) AS similarity
|
||||
FROM rd_experiments
|
||||
ORDER BY similarity DESC
|
||||
LIMIT %s
|
||||
""",
|
||||
(lit, lit, limit),
|
||||
)
|
||||
rows = cur.fetchall()
|
||||
out = []
|
||||
for r in rows:
|
||||
sim = float(r.get("similarity") or 0)
|
||||
if sim < min_score:
|
||||
continue
|
||||
r = dict(r)
|
||||
r["similarity"] = sim
|
||||
out.append(_row_json(r))
|
||||
return out
|
||||
|
||||
|
||||
def _start(
|
||||
rational: str,
|
||||
details: str = "",
|
||||
evolved_from: str = "none",
|
||||
experiment_name: str = "",
|
||||
branch: str = "",
|
||||
session_id: str = "",
|
||||
) -> Dict[str, Any]:
|
||||
rational = rational.strip()
|
||||
details = (details or "").strip()
|
||||
if not rational:
|
||||
raise ValueError("--rational is required")
|
||||
|
||||
rational_vec = _vector_literal(embed(rational))
|
||||
details_vec = _vector_literal(embed(details)) if details else None
|
||||
|
||||
evo: Optional[int] = None
|
||||
if evolved_from == "auto":
|
||||
evo = _search_evolved_from(f"{rational}\n{details}") if rational_vec or details_vec else None
|
||||
elif evolved_from and evolved_from.isdigit():
|
||||
evo = int(evolved_from)
|
||||
|
||||
with _conn() as conn, conn.cursor() as cur:
|
||||
cur.execute(
|
||||
"""
|
||||
INSERT INTO rd_experiments
|
||||
(experiment_name, rational, rational_embedding, details, details_embedding,
|
||||
evolved_from, start_ts, git_branch, status, session_id)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, 'starting', %s)
|
||||
RETURNING id
|
||||
""",
|
||||
(
|
||||
experiment_name or None,
|
||||
rational,
|
||||
rational_vec,
|
||||
details,
|
||||
details_vec,
|
||||
evo,
|
||||
_now_iso(),
|
||||
branch or "",
|
||||
session_id.strip() or None,
|
||||
),
|
||||
)
|
||||
row = cur.fetchone()
|
||||
exp_id = int(row["id"])
|
||||
conn.commit()
|
||||
|
||||
branch = branch or f"exp/{exp_id}"
|
||||
with _conn() as conn, conn.cursor() as cur:
|
||||
cur.execute("UPDATE rd_experiments SET git_branch = %s WHERE id = %s", (branch, exp_id))
|
||||
conn.commit()
|
||||
return _row_json(_get_row(exp_id) or {})
|
||||
|
||||
|
||||
def _finish(
|
||||
exp_id: int,
|
||||
ref_id: str = "",
|
||||
evaluation: Optional[str] = None,
|
||||
metrics: Optional[str] = None,
|
||||
mlruns_dir: str = "",
|
||||
experiment_name: str = "",
|
||||
rational: Optional[str] = None,
|
||||
details: Optional[str] = None,
|
||||
status: Optional[str] = None,
|
||||
) -> Dict[str, Any]:
|
||||
row = _get_row(exp_id)
|
||||
if not row:
|
||||
raise ValueError(f"experiment {exp_id} not found")
|
||||
|
||||
fields: List[str] = []
|
||||
params: List[Any] = []
|
||||
status = status or "done"
|
||||
if status is None and row.get("status") in ("starting", "running"):
|
||||
status = "done"
|
||||
|
||||
rational = (rational or row.get("rational") or "").strip()
|
||||
details = (details if details is not None else row.get("details") or "").strip()
|
||||
|
||||
fields.append("rational = %s"); params.append(rational)
|
||||
fields.append("rational_embedding = %s"); params.append(_vector_literal(embed(rational)))
|
||||
fields.append("details = %s"); params.append(details)
|
||||
fields.append("details_embedding = %s"); params.append(_vector_literal(embed(details)) if details else None)
|
||||
if evaluation is not None:
|
||||
fields.append("evaluation = %s"); params.append(evaluation.strip())
|
||||
if metrics is not None:
|
||||
fields.append("metrics = %s"); params.append(json.dumps(json.loads(metrics)))
|
||||
if ref_id:
|
||||
fields.append("experiment_ref_id = %s"); params.append(ref_id.strip())
|
||||
if mlruns_dir:
|
||||
fields.append("mlruns_dir = %s"); params.append(mlruns_dir.strip())
|
||||
if experiment_name:
|
||||
fields.append("experiment_name = %s"); params.append(experiment_name.strip())
|
||||
fields.append("status = %s"); params.append(status)
|
||||
fields.append("end_ts = %s"); params.append(_now_iso())
|
||||
fields.append("updated_at = %s"); params.append(_now_iso())
|
||||
params.append(exp_id)
|
||||
|
||||
with _conn() as conn, conn.cursor() as cur:
|
||||
cur.execute(f"UPDATE rd_experiments SET {', '.join(fields)} WHERE id = %s", params)
|
||||
conn.commit()
|
||||
return _row_json(_get_row(exp_id) or {})
|
||||
|
||||
|
||||
def _mlruns_dir(exp_name: str) -> str:
|
||||
uri = (os.environ.get("DATABASE_URL") or "").strip()
|
||||
if uri.startswith("postgres://"):
|
||||
uri = "postgresql+psycopg://" + uri[len("postgres://") :]
|
||||
if uri.startswith("postgresql://") or uri.startswith("postgresql+psycopg://"):
|
||||
with _conn() as conn, conn.cursor() as cur:
|
||||
cur.execute("SELECT artifact_location FROM experiments WHERE name = %s", (exp_name,))
|
||||
row = cur.fetchone()
|
||||
if not row:
|
||||
raise RuntimeError(f"mlflow experiment {exp_name!r} not found")
|
||||
return row["artifact_location"]
|
||||
|
||||
lake = (os.environ.get("TAC_LAKE_DIR") or "").strip()
|
||||
if not lake:
|
||||
raise RuntimeError("TAC_LAKE_DIR not set")
|
||||
db_path = Path(lake) / "mlruns.db"
|
||||
if not db_path.exists():
|
||||
raise RuntimeError(f"mlruns.db not found at {db_path}")
|
||||
conn = sqlite3.connect(db_path)
|
||||
try:
|
||||
row = conn.execute("SELECT artifact_location FROM experiments WHERE name = ?", (exp_name,)).fetchone()
|
||||
finally:
|
||||
conn.close()
|
||||
if not row:
|
||||
raise RuntimeError(f"mlflow experiment {exp_name!r} not found in {db_path}")
|
||||
return row[0]
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- git
|
||||
def _parent_root() -> Path:
|
||||
start = Path.cwd()
|
||||
dir = start
|
||||
while dir != Path(dir.anchor):
|
||||
if (dir / "pnpm-workspace.yaml").exists() or (dir / "Cargo.toml").exists() or (dir / "opencode.json").exists():
|
||||
return dir
|
||||
dir = dir.parent
|
||||
raise RuntimeError("not inside a tradeac workspace")
|
||||
|
||||
|
||||
def _gitc(*args: str) -> subprocess.CompletedProcess:
|
||||
root = _parent_root()
|
||||
exp = root / "experiments"
|
||||
user = (os.environ.get("GIT_USER") or "").strip()
|
||||
password = (os.environ.get("GIT_PASS") or "").strip()
|
||||
helper = f'!f() {{ echo "username={user}"; echo "password={password}"; }}; f'
|
||||
cmd = ["git", "-C", str(exp), "-c", f"credential.helper={helper}"] + list(args)
|
||||
return subprocess.run(cmd, capture_output=True, text=True)
|
||||
|
||||
|
||||
def _git_ok(proc: subprocess.CompletedProcess) -> bool:
|
||||
return proc.returncode == 0
|
||||
|
||||
|
||||
def _git_out(proc: subprocess.CompletedProcess) -> str:
|
||||
return (proc.stdout or "").strip() or (proc.stderr or "").strip()
|
||||
|
||||
|
||||
def _require_auth() -> None:
|
||||
if not (os.environ.get("GIT_REPO_URL") or "").strip() or not (os.environ.get("GIT_USER") or "").strip():
|
||||
raise RuntimeError("GIT_REPO_URL / GIT_USER not set")
|
||||
|
||||
|
||||
def _ensure_repo() -> None:
|
||||
root = _parent_root()
|
||||
exp = root / "experiments"
|
||||
if (exp / ".git").exists():
|
||||
_gitc("remote", "set-url", "origin", os.environ["GIT_REPO_URL"])
|
||||
else:
|
||||
_require_auth()
|
||||
(root / "experiments").mkdir(parents=True, exist_ok=True)
|
||||
subprocess.run(
|
||||
["git", "clone", "-q", os.environ["GIT_REPO_URL"], str(exp)],
|
||||
check=True, capture_output=True, text=True,
|
||||
)
|
||||
_gitc("config", "user.email", f"{os.environ.get('GIT_USER', '')}@tradeac.local")
|
||||
_gitc("config", "user.name", os.environ.get("GIT_USER", "tradeac-agent"))
|
||||
|
||||
|
||||
def _ensure_base(base: str = "main") -> str:
|
||||
_require_auth()
|
||||
_gitc("fetch", "origin", base)
|
||||
if _git_ok(_gitc("rev-parse", "--verify", f"origin/{base}")):
|
||||
return f"origin/{base}"
|
||||
if _git_ok(_gitc("rev-parse", "--verify", base)):
|
||||
return base
|
||||
return base
|
||||
|
||||
|
||||
def _fork_branch(base_ref: str, branch: str) -> str:
|
||||
if _git_ok(_gitc("rev-parse", "--verify", f"origin/{branch}")):
|
||||
_gitc("checkout", "-q", "-B", branch, f"origin/{branch}")
|
||||
_gitc("reset", "-q", "--hard", f"origin/{branch}")
|
||||
return "reused existing branch"
|
||||
_gitc("fetch", "-q", "origin")
|
||||
if _git_ok(_gitc("rev-parse", "--verify", f"origin/{branch}")):
|
||||
_gitc("checkout", "-q", "-B", branch, f"origin/{branch}")
|
||||
return "reused existing branch"
|
||||
base_commit = ""
|
||||
if _git_ok(_gitc("rev-parse", "--verify", f"{base_ref}^{{commit}}")):
|
||||
base_commit = base_ref
|
||||
elif not base_ref.startswith("origin/"):
|
||||
base_commit = f"origin/{base_ref}"
|
||||
if not base_commit:
|
||||
raise RuntimeError(f"base '{base_ref}' not found locally or on origin")
|
||||
_gitc("checkout", "-q", "-B", branch, base_commit)
|
||||
return f"forked from {base_ref}"
|
||||
|
||||
|
||||
def _snapshot_code(paths: Optional[List[str]] = None) -> str:
|
||||
root = _parent_root()
|
||||
exp = root / "experiments"
|
||||
paths = paths or ["tac-qlib/tac_qlib/contrib", "tac-qlib/tac_qlib/data"]
|
||||
parent_head = _git_out(_gitc("rev-parse", "HEAD")) or "unknown"
|
||||
import shutil
|
||||
|
||||
shutil.rmtree(exp / "code", ignore_errors=True)
|
||||
(exp / "code").mkdir(parents=True, exist_ok=True)
|
||||
manifest = [f"# TradeAC custom-qlib-code snapshot (auto-generated)", f"# parent repo HEAD : {parent_head}"]
|
||||
for p in paths:
|
||||
manifest.append(f"# {p}")
|
||||
manifest.append("# per-file hashes (git hash-object):")
|
||||
for p in paths:
|
||||
src = root / p
|
||||
if not src.exists():
|
||||
continue
|
||||
dst = exp / "code" / p
|
||||
dst.parent.mkdir(parents=True, exist_ok=True)
|
||||
if src.is_dir():
|
||||
shutil.copytree(src, dst, dirs_exist_ok=True)
|
||||
for f in sorted(src.rglob("*")):
|
||||
if f.is_file():
|
||||
rel = str(f.relative_to(root))
|
||||
h = _git_out(_gitc("hash-object", str(f)))
|
||||
manifest.append(f" {h} {rel}")
|
||||
else:
|
||||
shutil.copy2(src, dst)
|
||||
h = _git_out(_gitc("hash-object", str(src)))
|
||||
manifest.append(f" {h} {p}")
|
||||
(exp / "code" / "MANIFEST.txt").write_text("\n".join(manifest) + "\n")
|
||||
return f"code snapshotted -> experiments/code (parent @ {parent_head[:12]})"
|
||||
|
||||
|
||||
def _commit(message: str) -> str:
|
||||
_gitc("add", "-A")
|
||||
if _git_ok(_gitc("diff", "--cached", "--quiet")):
|
||||
return "nothing to commit"
|
||||
_gitc("commit", "-q", "-m", message)
|
||||
return "committed"
|
||||
|
||||
|
||||
def _commit_push(message: str) -> str:
|
||||
result = _commit(message)
|
||||
if result == "nothing to commit":
|
||||
return result
|
||||
branch = _git_out(_gitc("branch", "--show-current"))
|
||||
_require_auth()
|
||||
proc = _gitc("push", "-u", "origin", branch)
|
||||
if not _git_ok(proc):
|
||||
raise RuntimeError(f"push failed: {_git_out(proc)}")
|
||||
return f"pushed {branch}"
|
||||
|
||||
|
||||
def _parent_changes() -> str:
|
||||
root = _parent_root()
|
||||
proc = subprocess.run(["git", "-C", str(root), "status", "--porcelain"], capture_output=True, text=True)
|
||||
out = (proc.stdout or "").strip()
|
||||
if not out:
|
||||
return "parent repo clean (no changes)"
|
||||
lines = out.splitlines()
|
||||
filtered = [
|
||||
l for l in lines
|
||||
if not l.startswith(".. experiments/")
|
||||
and not l.startswith("?? experiments/")
|
||||
and not l.startswith(".. tac-qlib/tac_qlib/contrib/")
|
||||
and not l.startswith(".. tac-qlib/tac_qlib/data/")
|
||||
and not l.startswith("?? tac-qlib/tac_qlib/contrib/")
|
||||
and not l.startswith("?? tac-qlib/tac_qlib/data/")
|
||||
]
|
||||
expected = [l for l in lines if l.startswith(".. tac-qlib/tac_qlib/contrib/") or l.startswith(".. tac-qlib/tac_qlib/data/")]
|
||||
note = ""
|
||||
if expected:
|
||||
note = "note: custom qlib code changed in the parent repo (contrib/data) — snapshotted to the experiment branch via trace snapshot:\n" + "\n".join(expected)
|
||||
if not filtered:
|
||||
return "parent repo changes limited to experiments/ and snapshotted custom qlib code (ok)" + (f"\n{note}" if note else "")
|
||||
return "WARNING: unexpected parent-repo changes outside the experiments/ clone:\n" + "\n".join(filtered) + "\n→ review and revert before finishing" + (f"\n{note}" if note else "")
|
||||
|
||||
|
||||
def slugify(text: str) -> str:
|
||||
s = "".join(c for c in text.lower() if c.isalnum() or c in " -").replace(" ", "-")
|
||||
return s[:40].strip("-")
|
||||
|
||||
|
||||
def get_experiment_branch(exp_id: int) -> str:
|
||||
row = _get_row(exp_id)
|
||||
if not row:
|
||||
raise ValueError(f"experiment {exp_id} not found")
|
||||
return row["git_branch"] or f"exp/{exp_id}"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- MCP tools
|
||||
def rd_trace_init() -> dict:
|
||||
"""Ensure the traceability store + experiments git repo are ready (rd_experiments table, base main)."""
|
||||
_init_db()
|
||||
_ensure_repo()
|
||||
base = _ensure_base("main")
|
||||
return {"status": "ready", "base": base}
|
||||
|
||||
|
||||
def rd_trace_start(
|
||||
rational: str,
|
||||
details: str = "",
|
||||
experiment_name: str = "",
|
||||
evolved_from: str = "none",
|
||||
session_id: str = "",
|
||||
) -> dict:
|
||||
"""Open a traced experiment: insert the rd_experiments row, resolve evolved_from, fork + push the experiment branch. Pass `session_id` (the opencode chat id) so the lineage keeps a stable chat link. Returns experiment_id / branch / evolved_from / base_branch as one JSON object."""
|
||||
_init_db()
|
||||
_ensure_repo()
|
||||
|
||||
evo_id = evolved_from
|
||||
if evolved_from == "auto":
|
||||
evo_id = str(_search_evolved_from(rational) or "")
|
||||
|
||||
row = _start(rational, details, evolved_from=evo_id or "none", experiment_name=experiment_name, session_id=session_id)
|
||||
exp_id = int(row["id"])
|
||||
branch = f"exp/{exp_id}-{slugify(rational)}"
|
||||
_gitc("checkout", "-q", "-B", branch, "main") if False else None
|
||||
|
||||
# fork from the predecessor branch (or main)
|
||||
base_branch = "main"
|
||||
if evo_id and evo_id.isdigit():
|
||||
base_branch = get_experiment_branch(int(evo_id))
|
||||
base_ref = _ensure_base(base_branch)
|
||||
_fork_branch(base_ref, branch)
|
||||
_snapshot_code()
|
||||
_gitc("checkout", "-q", "-B", branch, branch) if False else None
|
||||
|
||||
# persist branch on the row
|
||||
with _conn() as conn, conn.cursor() as cur:
|
||||
cur.execute("UPDATE rd_experiments SET git_branch = %s WHERE id = %s", (branch, exp_id))
|
||||
conn.commit()
|
||||
|
||||
_commit_push(f"start experiment {exp_id} ({branch})")
|
||||
return {"experiment_id": exp_id, "branch": branch, "evolved_from": evo_id or "none", "base_branch": base_branch}
|
||||
|
||||
|
||||
def rd_trace_finish(
|
||||
experiment_id: int,
|
||||
ref_id: str = "",
|
||||
evaluation: str = "",
|
||||
metrics: str = "",
|
||||
mlruns_dir: str = "",
|
||||
experiment_name: str = "",
|
||||
) -> dict:
|
||||
"""Close a traced experiment: update the row (link the mlflow run, metrics/evaluation/end_ts), snapshot code, commit + push the branch. Returns the updated row."""
|
||||
_finish(experiment_id, ref_id=ref_id, evaluation=evaluation or None, metrics=metrics or None, mlruns_dir=mlruns_dir, experiment_name=experiment_name)
|
||||
branch = get_experiment_branch(experiment_id)
|
||||
_gitc("checkout", "-q", "-B", branch, branch)
|
||||
_snapshot_code()
|
||||
_commit_push(f"finish experiment {experiment_id} ({branch})")
|
||||
return {"experiment_id": experiment_id, "branch": branch, "row": _row_json(_get_row(experiment_id) or {})}
|
||||
|
||||
|
||||
def rd_trace_commit(experiment_id: int, message: str = "wip") -> dict:
|
||||
"""Commit the current experiment branch state (no push)."""
|
||||
branch = get_experiment_branch(experiment_id)
|
||||
_gitc("checkout", "-q", "-B", branch, branch)
|
||||
result = _commit(f"exp {experiment_id}: {message}")
|
||||
return {"experiment_id": experiment_id, "branch": branch, "result": result}
|
||||
|
||||
|
||||
def rd_trace_snapshot(experiment_id: int, paths: str = "") -> dict:
|
||||
"""Snapshot custom qlib contrib/data code onto the experiment branch (default contrib+data)."""
|
||||
branch = get_experiment_branch(experiment_id)
|
||||
_gitc("checkout", "-q", "-B", branch, branch)
|
||||
path_list = [p.strip() for p in paths.split(",") if p.strip()] if paths else None
|
||||
msg = _snapshot_code(path_list)
|
||||
_commit_push(f"exp {experiment_id}: snapshot custom qlib code")
|
||||
return {"experiment_id": experiment_id, "branch": branch, "result": msg}
|
||||
|
||||
|
||||
def rd_trace_guard() -> dict:
|
||||
"""Check the parent repo for unexpected changes outside the experiments clone."""
|
||||
return {"parent_changes": _parent_changes()}
|
||||
|
||||
|
||||
def rd_trace_search(query: str, limit: int = 10, min_score: float = _MIN_SCORE) -> dict:
|
||||
"""Semantic search over experiment rationals/details (pgvector, falls back to ILIKE)."""
|
||||
return {"results": _search(query, limit=limit, min_score=min_score)}
|
||||
|
||||
|
||||
def rd_trace_get(experiment_id: int) -> dict:
|
||||
"""Return one traced experiment row."""
|
||||
row = _get_row(experiment_id)
|
||||
if not row:
|
||||
raise ValueError(f"experiment {experiment_id} not found")
|
||||
return _row_json(row)
|
||||
|
||||
|
||||
def rd_trace_list(limit: int = 20) -> dict:
|
||||
"""List traced experiments (newest first)."""
|
||||
return {"experiments": [_row_json(r) for r in _all_rows(limit)]}
|
||||
|
||||
|
||||
def rd_trace_mlruns_dir(experiment_name: str) -> dict:
|
||||
"""Resolve the mlflow artifact location for an experiment name."""
|
||||
return {"mlruns_dir": _mlruns_dir(experiment_name)}
|
||||
|
||||
|
||||
def register_trace_tools(server) -> None:
|
||||
"""Attach all rd_trace_* tools to an MCPServer instance (called by rd_server.main())."""
|
||||
for fn in (
|
||||
rd_trace_init,
|
||||
rd_trace_start,
|
||||
rd_trace_finish,
|
||||
rd_trace_commit,
|
||||
rd_trace_snapshot,
|
||||
rd_trace_guard,
|
||||
rd_trace_search,
|
||||
rd_trace_get,
|
||||
rd_trace_list,
|
||||
rd_trace_mlruns_dir,
|
||||
):
|
||||
server.tool(structured_output=False)(fn)
|
||||
@@ -0,0 +1,59 @@
|
||||
"""Embed text via the self-hosted infinity embedding API (used by tac_qlib.trace).
|
||||
|
||||
Mirrors the standalone `embed.py` in the tac-qlib-custom skill lib so the trace
|
||||
MCP tools can embed rational/details without shelling out.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
import urllib.request
|
||||
|
||||
EMBEDDING_MODEL = "michaelfeil/bge-small-en-v1.5"
|
||||
MAX_TOKENS = 512
|
||||
CHARS_PER_TOKEN = 4
|
||||
|
||||
|
||||
def estimate_tokens(text: str) -> int:
|
||||
return max(1, -(-len(text) // CHARS_PER_TOKEN))
|
||||
|
||||
|
||||
def embed(text: str, timeout: int = 40) -> list[float] | None:
|
||||
base_url = (os.environ.get("EMBEDDING_API_BASE_URL") or "").strip()
|
||||
api_key = (os.environ.get("EMBEDDING_API_KEY") or "").strip()
|
||||
if not base_url or not api_key:
|
||||
return None
|
||||
|
||||
if estimate_tokens(text) > MAX_TOKENS:
|
||||
raise ValueError(
|
||||
f"text is ~{estimate_tokens(text)} tokens, exceeding the {MAX_TOKENS}-token embedding "
|
||||
"context. Write a <=512-token summary of the experiment and embed that instead."
|
||||
)
|
||||
|
||||
body = json.dumps({"model": EMBEDDING_MODEL, "input": text}).encode("utf-8")
|
||||
req = urllib.request.Request(
|
||||
base_url,
|
||||
data=body,
|
||||
headers={
|
||||
"accept": "application/json",
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
)
|
||||
user, _, password = api_key.partition(":")
|
||||
cred = base64.b64encode(f"{user}:{password}".encode("utf-8")).decode("ascii")
|
||||
req.add_header("Authorization", f"Basic {cred}")
|
||||
|
||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||
payload = json.loads(resp.read().decode("utf-8"))
|
||||
|
||||
data = payload.get("data") if isinstance(payload, dict) else None
|
||||
if isinstance(data, list) and data and isinstance(data[0], dict):
|
||||
emb = data[0].get("embedding")
|
||||
if isinstance(emb, list) and emb:
|
||||
return [float(v) for v in emb]
|
||||
embeddings = payload.get("embeddings") if isinstance(payload, dict) else None
|
||||
if isinstance(embeddings, list) and embeddings and isinstance(embeddings[0], list):
|
||||
return [float(v) for v in embeddings[0]]
|
||||
raise RuntimeError(f"unexpected embedding response shape: {str(payload)[:300]}")
|
||||
@@ -0,0 +1,108 @@
|
||||
"""Smoke tests for the tac_qlib contrib package (model/strategy).
|
||||
|
||||
Covers the pieces a workflow YAML resolves via ``module_path``:
|
||||
|
||||
- ``tac_qlib.contrib.model.rank_gbdt`` -> RankICLGBModel (+ rank feval)
|
||||
- ``tac_qlib.contrib.strategy.optimal_stop`` -> OptimalStopControl
|
||||
|
||||
The strategy smoke test runs a real (tiny) daily backtest through qlib's
|
||||
executor against the TradeAC lake. The model smoke test checks data
|
||||
preparation (per-day query groups) + the rank feval without a full fit.
|
||||
|
||||
Run::
|
||||
|
||||
TAC_LAKE_DIR=/home/data/lake .venv/bin/python tests/test_contrib.py
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(__file__), ".."))
|
||||
|
||||
LAKE_ROOT = os.environ["TAC_LAKE_DIR"]
|
||||
CODES = ["AAPL", "MSFT", "NVDA", "GOOGL", "AMZN", "META"]
|
||||
START, END = "2026-06-01", "2026-07-31"
|
||||
|
||||
|
||||
def _signal(close: pd.DataFrame) -> pd.Series:
|
||||
"""3-day momentum score indexed (datetime, instrument) covering [START, END]."""
|
||||
mom = close.pct_change(3).stack()
|
||||
mom.index = mom.index.set_names(["datetime", "instrument"])
|
||||
return mom.dropna()
|
||||
|
||||
|
||||
def main():
|
||||
from tac_qlib.qlib_init import qlib_init
|
||||
|
||||
qlib_init(provider_uri=LAKE_ROOT, market="US", freq="day", log_level="WARN")
|
||||
from qlib.data import D
|
||||
|
||||
close = D.features(CODES, ["$close"], START, END, freq="day")["$close"]
|
||||
close = close.unstack("instrument")
|
||||
sig = _signal(close)
|
||||
assert len(sig) > 0, "empty synthetic signal"
|
||||
print(f"[ok] synthetic signal: {len(sig)} rows, {sig.index.get_level_values(0).nunique()} days")
|
||||
|
||||
# ---- OptimalStopControl end-to-end ------------------------------------
|
||||
from tac_qlib.contrib.strategy.optimal_stop import OptimalStopControl
|
||||
from qlib.contrib.evaluate import backtest_daily
|
||||
|
||||
strat = OptimalStopControl(
|
||||
signal=sig, topk=2, entry_pct=0.8, exit_pct=0.5,
|
||||
max_hold_days=5, min_hold_days=1, sl=-0.05, notional=10_000.0,
|
||||
)
|
||||
report, positions = backtest_daily(
|
||||
start_time=START, end_time=END, strategy=strat, account=1_000_000,
|
||||
benchmark=None,
|
||||
exchange_kwargs={"codes": CODES, "deal_price": "$close", "freq": "day",
|
||||
"open_cost": 0.0005, "close_cost": 0.0015, "min_cost": 5.0},
|
||||
)
|
||||
assert isinstance(report, pd.DataFrame) and "return" in report and len(report) >= 5
|
||||
assert not report["return"].isna().all()
|
||||
print(f"[ok] OptimalStopControl backtest: {len(report)} days, "
|
||||
f"end equity {float(report['return'].add(1).cumprod().iloc[-1]):.4f}")
|
||||
|
||||
# ---- RankICLGBModel: instantiate + _prepare_data (per-day groups) ------
|
||||
from tac_qlib.contrib.data.handler import TACHandler
|
||||
from qlib.data.dataset import DatasetH
|
||||
|
||||
h = TACHandler(
|
||||
instruments=CODES, start_time=START, end_time=END,
|
||||
fit_start_time=START, fit_end_time="2026-06-30", freq="day",
|
||||
lake_root=LAKE_ROOT, market="US",
|
||||
label="Ref($close,-6)/Ref($close,-1)-1",
|
||||
)
|
||||
ds = DatasetH(
|
||||
handler=h,
|
||||
segments={"train": (START, "2026-06-30"), "valid": ("2026-07-01", END)},
|
||||
)
|
||||
from tac_qlib.contrib.model.rank_gbdt import RankICLGBModel, rankic_feval
|
||||
|
||||
model = RankICLGBModel(loss="mse", learning_rate=0.05, num_leaves=7, n_estimators=50)
|
||||
data = model._prepare_data(ds)
|
||||
lgb_ds, names = list(zip(*data))
|
||||
assert names == ("train", "valid")
|
||||
groups = lgb_ds[0].get_group()
|
||||
assert groups is not None and len(groups) >= 5, f"per-day query groups missing: {groups}"
|
||||
# every group size == number of instruments that day
|
||||
assert set(groups) <= {len(CODES), len(CODES) - 1}, f"unexpected group sizes {groups}"
|
||||
print(f"[ok] RankICLGBModel._prepare_data: groups={groups[:5]}... (n_days={len(groups)})")
|
||||
|
||||
# rank feval on a hand-built lgb.Dataset
|
||||
import lightgbm as lgb
|
||||
|
||||
y = np.array([1.0, 2.0, 3.0, 3.0, 2.0, 1.0])
|
||||
preds = np.array([1.0, 2.0, 3.0, 3.0, 2.0, 1.0])
|
||||
dv = lgb.Dataset(np.zeros((6, 2)), label=y, group=np.array([3, 3]))
|
||||
name, value, higher = rankic_feval(preds, dv)
|
||||
assert name == "rankic" and higher is True and abs(value - 1.0) < 1e-9
|
||||
print(f"[ok] rankic_feval: {name}={value:.4f} (higher_is_better={higher})")
|
||||
|
||||
print("\nALL CONTRIB CHECKS PASSED")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,106 @@
|
||||
"""Smoke tests: qlib against the TradeAC lake (plain asserts, no pytest needed).
|
||||
|
||||
Run::
|
||||
|
||||
.venv/bin/python tests/test_lake_providers.py
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(__file__), ".."))
|
||||
|
||||
# TAC_LAKE_DIR is mandatory (no default fallback). Fail fast if it is missing.
|
||||
LAKE_ROOT = os.environ["TAC_LAKE_DIR"]
|
||||
|
||||
|
||||
def main():
|
||||
from tac_qlib.qlib_init import qlib_init
|
||||
|
||||
qlib_init(provider_uri=LAKE_ROOT, market="US", freq="day", log_level="WARN")
|
||||
|
||||
from qlib.data import D
|
||||
from qlib.data.data import Cal, ExpressionD, Inst, DatasetD
|
||||
|
||||
# ---- calendar ---------------------------------------------------------
|
||||
cal = Cal.calendar(freq="day")
|
||||
assert isinstance(cal, (list, np.ndarray)) and len(cal) >= 100, f"calendar too small: {len(cal)}"
|
||||
print(f"[ok] calendar: {len(cal)} trading days, {pd.Timestamp(cal[0]).date()} -> {pd.Timestamp(cal[-1]).date()}")
|
||||
|
||||
# ---- instruments ------------------------------------------------------
|
||||
inst = Inst.list_instruments({"market": "all"}, start_time=cal[0], end_time=cal[-1], freq="day")
|
||||
assert len(inst) >= 5, f"expected >=5 instruments, got {inst}"
|
||||
print(f"[ok] instruments: {sorted(inst)}")
|
||||
|
||||
# ---- raw features -----------------------------------------------------
|
||||
start, end = "2026-03-01", "2026-06-30"
|
||||
df = D.features(sorted(inst)[:4], ["$close", "$volume", "$vwap"], start, end, freq="day")
|
||||
assert not df.empty
|
||||
assert df.columns.tolist() == ["$close", "$volume", "$vwap"]
|
||||
assert not df["$close"].isna().all()
|
||||
# index must be the (datetime, instrument) MultiIndex, sorted
|
||||
assert isinstance(df.index, pd.MultiIndex)
|
||||
assert df.index.names == [df.index.names[0], df.index.names[1]]
|
||||
n_rows = len(df)
|
||||
print(f"[ok] D.features: {len(df)} rows x {len(df.columns)} cols; close sample:\n{df['$close'].head(3)}")
|
||||
|
||||
# NaN for fields the lake does not store
|
||||
df_unk = D.features(sorted(inst)[:2], ["$factor", "$change"], start, end, freq="day")
|
||||
assert df_unk["$factor"].isna().all() and df_unk["$change"].isna().all()
|
||||
print("[ok] unknown fields ($factor/$change) are all-NaN")
|
||||
|
||||
# ---- expression engine (Option A / qlib defaults) ---------------------
|
||||
exp = "Ref($close,-2)/$close-1" # same default label as Alpha158
|
||||
sym = sorted(inst)[0] # use a symbol that is actually in the lake
|
||||
s = ExpressionD.expression(sym, exp, start_time=start, end_time=end, freq="day")
|
||||
assert isinstance(s, pd.Series) and len(s) > 0
|
||||
assert s.notna().any()
|
||||
print(f"[ok] ExpressionD.expression: {len(s)} values, sample:\n{s.head(3)}")
|
||||
|
||||
# a full dataset can be materialised through the expression engine
|
||||
df_ds = DatasetD.dataset(sorted(inst), [exp], start, end, freq="day")
|
||||
assert isinstance(df_ds, pd.DataFrame) and len(df_ds) > 0
|
||||
print(f"[ok] DatasetD.dataset: {df_ds.shape}")
|
||||
|
||||
# ---- TACHandler: feature discovery + DropAllNaN ------------------------
|
||||
from tac_qlib.contrib.data.handler import TACHandler
|
||||
|
||||
h = TACHandler(
|
||||
instruments=sorted(inst)[:6],
|
||||
start_time="2026-03-01",
|
||||
end_time="2026-08-06",
|
||||
fit_start_time="2026-03-01",
|
||||
fit_end_time="2026-05-31",
|
||||
freq="day",
|
||||
lake_root=LAKE_ROOT,
|
||||
market="US",
|
||||
)
|
||||
# the lake's stoch_* columns are fully NaN -> they must be dropped by DropAllNaN
|
||||
assert not any("stoch" in str(c) for c in h._infer.columns), h._infer.columns.tolist()
|
||||
# train/valid/test must expose identical feature columns
|
||||
from qlib.data.dataset import DatasetH
|
||||
from qlib.data.dataset.handler import DataHandlerLP
|
||||
|
||||
ds = DatasetH(
|
||||
handler=h,
|
||||
segments={
|
||||
"train": ("2026-03-01", "2026-05-31"),
|
||||
"valid": ("2026-06-01", "2026-06-30"),
|
||||
"test": ("2026-07-01", "2026-08-06"),
|
||||
},
|
||||
)
|
||||
cols = {
|
||||
seg: ds.prepare(segments=seg, col_set="feature", data_key=DataHandlerLP.DK_I).columns.tolist()
|
||||
for seg in ("train", "valid", "test")
|
||||
}
|
||||
assert cols["train"] == cols["valid"] == cols["test"], cols
|
||||
print(f"[ok] TACHandler: {len(cols['train'])} features, stoch dropped, segments aligned")
|
||||
|
||||
print("\nALL LAKE PROVIDER CHECKS PASSED")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,109 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# Basic LightGBM qrun workflow on the TradeAC lake -- short window sanity run.
|
||||
# Train on ~3 months, early-stop on 1 month valid, predict+backtest on ~1 month.
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///mlruns.db"
|
||||
default_exp_name: "tac-basic-short"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.05
|
||||
num_leaves: 15
|
||||
n_estimators: 200
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.01
|
||||
reg_lambda: 0.01
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: all
|
||||
start_time: 2026-03-01
|
||||
end_time: 2026-08-14
|
||||
fit_start_time: 2026-03-01
|
||||
fit_end_time: 2026-05-31
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
segments:
|
||||
train: [2026-03-01, 2026-05-31]
|
||||
valid: [2026-06-01, 2026-06-30]
|
||||
test: [2026-07-01, 2026-08-14]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 2
|
||||
n_drop: 1
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-07-01
|
||||
end_time: 2026-08-14
|
||||
account: 1000000
|
||||
benchmark: QQQ
|
||||
exchange_kwargs:
|
||||
codes: all
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,129 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# Tune run 1: wider, longer-horizon, de-duplicated universe.
|
||||
#
|
||||
# Baseline (exp 1 / run f29f5446): IC 0.071 / ICIR 0.17, Rank IC ~0.014;
|
||||
# strategy +4.9% ann (raw) vs benchmark ~+89% ann; excess return w/ cost
|
||||
# -0.94 ann, IR -2.23, excess max drawdown -18.9%. topk=2 with 24 trades over
|
||||
# 27 days on a universe of correlated ETFs + leveraged hedges (VXX/USO/SLV)
|
||||
# produced high turnover and a portfolio that trailed AAPL badly.
|
||||
#
|
||||
# Changes:
|
||||
# - universe: drop leveraged/noisy names (VXX, USO, SLV, BIL) and near-
|
||||
# duplicate index baskets (GPIQ, QQQE, KTEC); keep 10 liquid core names.
|
||||
# - label: 5-day forward return (Ref($close,-6)/Ref($close,-1)-1) to cut
|
||||
# single-day noise and match the intended holding horizon.
|
||||
# - topk 2 -> 5, n_drop 1: more diversification, lower turnover per name.
|
||||
# - benchmark AAPL -> QQQ (a real index ETF the universe tracks).
|
||||
# - model: learning_rate 0.03, 300 estimators (slower, deeper fit).
|
||||
#
|
||||
# Trigger:
|
||||
# rd_run_workflow config_path=tac-qlib/workflows/tune_run1_wider_5d.yaml \
|
||||
# experiment_name=tac-rd-tune
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///mlruns.db"
|
||||
default_exp_name: "tac-rd-tune"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.03
|
||||
num_leaves: 15
|
||||
n_estimators: 300
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.01
|
||||
reg_lambda: 0.01
|
||||
seed: 2026
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
|
||||
start_time: 2000-01-03
|
||||
end_time: 2026-08-06
|
||||
fit_start_time: 2026-03-01
|
||||
fit_end_time: 2026-05-31
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
segments:
|
||||
train: [2026-03-01, 2026-05-31]
|
||||
valid: [2026-06-01, 2026-06-30]
|
||||
test: [2026-07-01, 2026-08-06]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 5
|
||||
n_drop: 1
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-07-01
|
||||
end_time: 2026-08-06
|
||||
account: 1000000
|
||||
benchmark: QQQ
|
||||
exchange_kwargs:
|
||||
codes: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,127 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# Tune run 2: same-day signal, strongly regularized model, 3x rotating book.
|
||||
#
|
||||
# Baseline (exp 1 / run f29f5446): IC 0.071 / ICIR 0.17, Rank IC ~0.014;
|
||||
# excess return w/ cost -0.94 ann, IR -2.23. The 1-day signal was noisy
|
||||
# (Rank IC ~ 0) and the topk=2 book turned over 24 times in 27 days, paying
|
||||
# ~1.1% of the $1M account in costs.
|
||||
#
|
||||
# Changes (isolates model/backtest effects; universe + label same as baseline):
|
||||
# - model: stronger regularization (reg_alpha 0.5, reg_lambda 5.0,
|
||||
# subsample 0.7, colsample 0.6) to combat the unstable Rank IC.
|
||||
# - topk 2 -> 3, n_drop 1 -> 2: rotate out losers faster (lower cost drag,
|
||||
# higher turnover on only the worst names).
|
||||
# - benchmark AAPL -> QQQ.
|
||||
# - universe: drop leveraged/duplicate names (VXX, USO, SLV, BIL, GPIQ,
|
||||
# QQQE, KTEC) for a cleaner cross-section; keeps baseline 1-day label.
|
||||
#
|
||||
# Trigger:
|
||||
# rd_run_workflow config_path=tac-qlib/workflows/tune_run2_regularized.yaml \
|
||||
# experiment_name=tac-rd-tune
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///mlruns.db"
|
||||
default_exp_name: "tac-rd-tune"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.05
|
||||
num_leaves: 15
|
||||
n_estimators: 250
|
||||
colsample_bytree: 0.6
|
||||
subsample: 0.7
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.5
|
||||
reg_lambda: 5.0
|
||||
seed: 2026
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
|
||||
start_time: 2000-01-03
|
||||
end_time: 2026-08-06
|
||||
fit_start_time: 2026-03-01
|
||||
fit_end_time: 2026-05-31
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
segments:
|
||||
train: [2026-03-01, 2026-05-31]
|
||||
valid: [2026-06-01, 2026-06-30]
|
||||
test: [2026-07-01, 2026-08-06]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 3
|
||||
n_drop: 2
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-07-01
|
||||
end_time: 2026-08-06
|
||||
account: 1000000
|
||||
benchmark: QQQ
|
||||
exchange_kwargs:
|
||||
codes: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,146 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# Tune run 3 (NEXT run): 5-day label + clean 10-name universe.
|
||||
#
|
||||
# Baseline (exp 1 / run f29f5446):
|
||||
# IC 0.071, ICIR 0.17, Rank IC 0.014, Rank ICIR 0.03 -> ranking ~ coin flip
|
||||
# valid l2 best at round 0 and never improved (early-stopped ~50 rounds, overfit)
|
||||
# backtest: strategy +4.9% ann (raw) vs equal-weight universe +89.2% ann
|
||||
# (benchmark was unset -> qlib used equal-weight), excess w/ cost -94.0% ann,
|
||||
# IR -2.23, max DD -18.9%. topk=2, 24 trades/27 days, $11.1k cost (1.1% of $1M),
|
||||
# ending book ~97.5% in AAPL+IBIT (two names, both ~49%).
|
||||
#
|
||||
# PRIMARY LEVER (change one thing, everything else held at baseline):
|
||||
# label: 1-day next return -> 5-day forward return
|
||||
# "Ref($close,-6)/Ref($close,-1)-1".
|
||||
# Rationale: Rank ICIR 0.03 is the binding constraint - a topk book's return
|
||||
# is bounded by ranking quality, and no backtest tuning fixes a non-existent
|
||||
# ranking. The retained TA features (rsi_14, macd_hist, ema_20, volume,
|
||||
# stoch, aroon) are momentum/mean-reversion proxies that predict multi-day
|
||||
# drift, not overnight noise; and the avg holding in the baseline book was
|
||||
# several days, so a 1-day label mismatches the holding horizon.
|
||||
#
|
||||
# SUPPORTING (kept minimal, flagged for attribution):
|
||||
# - universe 17 -> 10: drop leveraged/vol/cash names (VXX, USO, SLV, BIL)
|
||||
# and near-duplicate index baskets (GPIQ, QQQE, KTEC). 17 names were really
|
||||
# ~8 independent betas (QQQ/QQQE/IVV/SMH/AIQ overlap heavily).
|
||||
# - topk 2 -> 5, n_drop 1 -> 2: stop the 2-name lottery, cut per-name turnover.
|
||||
# - benchmark: unset -> QQQ (a real index ETF the universe tracks; the
|
||||
# "excess return" vs equal-weight of a 17-name universe is misleading).
|
||||
# - model: explicitly num_boost_round 1000 + early_stopping_rounds 50 so the
|
||||
# round count is actually controlled (baseline's n_estimators: 200 was a
|
||||
# no-op, swallowed into lgb params; rounds were the 1000 default).
|
||||
# Hyperparameters otherwise identical to baseline (lr 0.05, num_leaves 15,
|
||||
# reg 0.01/0.01) for a clean label A/B.
|
||||
#
|
||||
# Trigger into a NEW experiment (do not pollute exp 1):
|
||||
# rd_run_workflow config_path=tac-qlib/workflows/tune_run3_label5d_clean_universe.yaml \
|
||||
# experiment_name=tac-rd-tune
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-tune"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.05
|
||||
num_leaves: 15
|
||||
num_boost_round: 1000
|
||||
early_stopping_rounds: 50
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.01
|
||||
reg_lambda: 0.01
|
||||
seed: 2026
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
|
||||
start_time: 2000-01-03
|
||||
end_time: 2026-08-06
|
||||
fit_start_time: 2026-03-01
|
||||
fit_end_time: 2026-05-31
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
segments:
|
||||
train: [2026-03-01, 2026-05-31]
|
||||
valid: [2026-06-01, 2026-06-30]
|
||||
test: [2026-07-01, 2026-08-06]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 5
|
||||
n_drop: 2
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-07-01
|
||||
end_time: 2026-08-06
|
||||
account: 1000000
|
||||
benchmark: QQQ
|
||||
exchange_kwargs:
|
||||
codes: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,121 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# Run 94736d89 (exp-4 tac-rd-tune2) follow-up -- single lever: WIDER UNIVERSE.
|
||||
#
|
||||
# Baseline (run 94736d89): 10 correlated tech/growth names -> weak cross-section
|
||||
# (IC 0.038 / ICIR 0.10), topk=5 book all-correlated, 295 trades / 152d and
|
||||
# $58k cost drag (5.8% of $1M) -> excess ann -18.8% vs QQQ.
|
||||
#
|
||||
# This run holds EVERYTHING else fixed (windows, 5-day label, LGB hyperparams,
|
||||
# topk=5/n_drop=2, benchmark QQQ) and only widens the universe 10 -> 17 with the
|
||||
# full lake set, adding genuinely uncorrelated assets (BIL cash, USO oil, SLV
|
||||
# silver, VXX vol, KTEC/QQQE/GPIQ factor sleeves) to de-correlate the cross-section,
|
||||
# stabilize the top-5 ranking and cut the churn/cost drag.
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///mlruns.db"
|
||||
default_exp_name: "tac-rd-tune3"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.05
|
||||
num_leaves: 15
|
||||
num_boost_round: 1000
|
||||
early_stopping_rounds: 50
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.01
|
||||
reg_lambda: 0.01
|
||||
seed: 2026
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ,BIL,GPIQ,KTEC,QQQE,SLV,USO,VXX
|
||||
start_time: 2000-01-03
|
||||
end_time: 2026-08-01
|
||||
fit_start_time: 2024-06-03
|
||||
fit_end_time: 2025-11-28
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
segments:
|
||||
train: [2024-06-03, 2025-11-28]
|
||||
valid: [2025-12-01, 2025-12-31]
|
||||
test: [2026-01-01, 2026-08-01]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 5
|
||||
n_drop: 2
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-01-01
|
||||
end_time: 2026-08-01
|
||||
account: 1000000
|
||||
benchmark: QQQ
|
||||
exchange_kwargs:
|
||||
codes: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ,BIL,GPIQ,KTEC,QQQE,SLV,USO,VXX
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,144 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# Tune run 4 (NEXT run): fix the universe bug + extend the train window.
|
||||
#
|
||||
# Previous (exp 3 / run 1e170e7f): IC -0.025 / ICIR -0.071 / RankIC -0.022 /
|
||||
# RankICIR -0.073 (noise), Long-Short -27% ann; excess +15.8% ann w/ cost
|
||||
# (IR 1.35) vs QQQ; $1M -> $971.7k (-2.8%); 65 trades/27d, $12.1k cost.
|
||||
# l2.train 0.35 vs l2.valid 0.95 -> gross overfit (valid best at round 0,
|
||||
# early-stopped at 16 trees).
|
||||
#
|
||||
# CRITICAL BUG in that run: the 10-name universe was silently IGNORED.
|
||||
# TACHandler passes `instruments` as a comma-separated STRING; qlib wraps it
|
||||
# as {"market": "<comma string>", "filter_pipe": []}; LakeInstrumentProvider
|
||||
# ._resolve_symbols() only handles list/tuple/ndarray and falls through to
|
||||
# load_symbols() = the ENTIRE 17-symbol lake. So the model trained/traded on
|
||||
# VXX, USO, SLV, BIL, GPIQ, QQQE, KTEC too - exactly the leveraged/hedge
|
||||
# names the "clean 10-name universe" hypothesis meant to drop. The universe
|
||||
# A/B is UNTESTED.
|
||||
# Fix (providers.py:100 _resolve_symbols): split comma-separated strings.
|
||||
#
|
||||
# PRIMARY LEVER (this run, ONE hypothesis):
|
||||
# universe = the intended 10-name dedup pool (AAPL,MSFT,TSLA,QQQ,IVV,SMH,
|
||||
# TLT,IBIT,MCHI,AIQ), now actually enforced, + train window 3 months -> 2
|
||||
# years. The 3-month window (~1000 rows for a 21-feature GBDT) is the hard
|
||||
# ceiling on signal; features span 2000-2026 so more data is free.
|
||||
# Everything else held at run-1e170e7f for a clean A/B: 5-day label,
|
||||
# LGB baseline hyperparams, topk 5 / n_drop 2, benchmark QQQ.
|
||||
#
|
||||
# SUPPORTING (flagged, NOT changed this run to keep attribution clean):
|
||||
# - if valid loss still rises monotonically after 2y of data, next step is
|
||||
# regularization (reg_alpha/lambda 0.01 -> ~0.5, num_leaves 15 -> 10,
|
||||
# lr 0.05 -> 0.02) rather than label/topk changes.
|
||||
#
|
||||
# Trigger into a NEW experiment (do not pollute exp 1/3):
|
||||
# rd_run_workflow config_path=tac-qlib/workflows/tune_run4_fix_universe_longtrain.yaml \
|
||||
# experiment_name=tac-rd-tune2
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-tune2"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.05
|
||||
num_leaves: 15
|
||||
num_boost_round: 1000
|
||||
early_stopping_rounds: 50
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.01
|
||||
reg_lambda: 0.01
|
||||
seed: 2026
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
|
||||
start_time: 2000-01-03
|
||||
end_time: 2026-08-06
|
||||
fit_start_time: 2024-06-03
|
||||
fit_end_time: 2026-05-31
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
segments:
|
||||
train: [2024-06-03, 2026-05-31]
|
||||
valid: [2026-06-01, 2026-06-30]
|
||||
test: [2026-07-01, 2026-08-06]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 5
|
||||
n_drop: 2
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-07-01
|
||||
end_time: 2026-08-06
|
||||
account: 1000000
|
||||
benchmark: QQQ
|
||||
exchange_kwargs:
|
||||
codes: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,133 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# Tune run 5: longer backtest window (2026-01-01 -> 2026-08-01).
|
||||
#
|
||||
# Purpose: test the fixed universe provider (_resolve_symbols now honors the
|
||||
# comma-separated 10-name instruments) and the fixed artifact pinning
|
||||
# (mlruns/<exp_id>/<run_id>/) over a 7-month out-of-sample window instead of
|
||||
# the single month (Jul) of run 47e9e369 / tune_run4.
|
||||
#
|
||||
# Changes vs tune_run4_fix_universe_longtrain.yaml:
|
||||
# - test/backtest window 2026-07-01..08-06 -> 2026-01-01..2026-08-01
|
||||
# - train/valid moved back so they stay strictly before test (no leakage):
|
||||
# train: 2024-06-03 .. 2025-11-28 (~18 months, ~4500 rows x 10 names)
|
||||
# valid: 2025-12-01 .. 2025-12-31 (1 month, right before test)
|
||||
# test : 2026-01-01 .. 2026-08-01 (7 months)
|
||||
# - everything else held fixed: 5-day label, LGB baseline hyperparams,
|
||||
# topk 5 / n_drop 2, benchmark QQQ, universe 10 names.
|
||||
#
|
||||
# NOTE: requires the providers.py fix so the universe is actually 10 names
|
||||
# (not silently expanded to all 17 lake symbols).
|
||||
#
|
||||
# Trigger (existing experiment, exp id 4 -> artifacts under
|
||||
# $TAC_LAKE_DIR/mlruns/4/<run_id>/ ):
|
||||
# rd_run_workflow config_path=tac-qlib/workflows/tune_run5_longtest.yaml \
|
||||
# experiment_name=tac-rd-tune2
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-tune2"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.05
|
||||
num_leaves: 15
|
||||
num_boost_round: 1000
|
||||
early_stopping_rounds: 50
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.01
|
||||
reg_lambda: 0.01
|
||||
seed: 2026
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
|
||||
start_time: 2000-01-03
|
||||
end_time: 2026-08-01
|
||||
fit_start_time: 2024-06-03
|
||||
fit_end_time: 2025-11-28
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
segments:
|
||||
train: [2024-06-03, 2025-11-28]
|
||||
valid: [2025-12-01, 2025-12-31]
|
||||
test: [2026-01-01, 2026-08-01]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 5
|
||||
n_drop: 2
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-01-01
|
||||
end_time: 2026-08-01
|
||||
account: 1000000
|
||||
benchmark: QQQ
|
||||
exchange_kwargs:
|
||||
codes: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,148 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# Tune run 6 (NEXT run): wider 10-name universe A/B vs run f744455056 (exp 1).
|
||||
#
|
||||
# Baseline (exp 1 / run f744455056 — this run):
|
||||
# Input : universe AAPL,MSFT,QQQ,IVV,SMH,TLT (6 names, 5 of them the same
|
||||
# tech beta); 21 features (OHLCV + TA); label 1-day next return;
|
||||
# LGB lr 0.05 / 15 leaves / 200 trees / reg 0.01,0.01;
|
||||
# train 03-01..05-31 / valid 06-01..06-30 / test 07-01..08-06.
|
||||
# Output: IC 0.048, ICIR 0.09, Rank IC 0.065, Rank ICIR 0.13 -> noise-level
|
||||
# (per-day n=6, IC swings -0.89..+0.74 with many null days).
|
||||
# Backtest had NO benchmark (benchmark null) -> the "+180% ann, IR 6.4"
|
||||
# headline is raw strategy return, not excess. Strategy +16.5% over 27
|
||||
# days, but ~half the P&L came from ONE day (2026-07-30 MSFT +14% sell,
|
||||
# +$72k realized). 30 trades/27 days, $15.3k cost (1.5% of $1M),
|
||||
# ending book 46.6% SMH + 50.8% TLT (2-name lottery).
|
||||
#
|
||||
# PRIMARY LEVER (change one thing, everything else held at baseline):
|
||||
# universe: 6 -> 10 names (AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ).
|
||||
# Rationale: with 6 near-collinear names there is nothing to rank — ICIR 0.09
|
||||
# is cross-sectional noise and the topk book just re-buys tech momentum on
|
||||
# correlated bets. Widening to ~10 independent-ish betas (mega tech, semis,
|
||||
# S&P, Nasdaq, bonds, BTC, EM, robotics) gives the cross-section real breadth,
|
||||
# stabilizes IC, and makes a diversified topk book possible.
|
||||
#
|
||||
# SUPPORTING (kept minimal, flagged for attribution):
|
||||
# - topk 2 -> 4, n_drop 1 -> 2: kill the 2-name lottery, cut per-name churn.
|
||||
# - benchmark: unset -> QQQ: the baseline "excess return" was raw strategy
|
||||
# return because no benchmark was wired; QQQ is the index the tech-heavy
|
||||
# universe tracks.
|
||||
# - model: explicit num_boost_round 1000 + early_stopping_rounds 50 so round
|
||||
# count is controlled (baseline's n_estimators: 200 was swallowed into lgb
|
||||
# params and valid l2 rose monotonically -> overfit). Hyperparameters
|
||||
# otherwise identical to baseline for a clean universe A/B.
|
||||
# - label: KEPT at 1-day next return so this run isolates the universe lever;
|
||||
# a 5-day horizon is the natural NEXT experiment (see tune_run3).
|
||||
#
|
||||
# Trigger into a NEW experiment (do not pollute exp 1); evolved_from = f744455056:
|
||||
# rd_run_workflow config_path=tac-qlib/workflows/tune_run6_wider_universe_ab.yaml \
|
||||
# experiment_name=tac-rd-tune
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-tune"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.05
|
||||
num_leaves: 15
|
||||
num_boost_round: 1000
|
||||
early_stopping_rounds: 50
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.01
|
||||
reg_lambda: 0.01
|
||||
seed: 2026
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
|
||||
start_time: 2000-01-03
|
||||
end_time: 2026-08-06
|
||||
fit_start_time: 2026-03-01
|
||||
fit_end_time: 2026-05-31
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-2)/Ref($close,-1)-1"
|
||||
segments:
|
||||
train: [2026-03-01, 2026-05-31]
|
||||
valid: [2026-06-01, 2026-06-30]
|
||||
test: [2026-07-01, 2026-08-06]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 4
|
||||
n_drop: 2
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-07-01
|
||||
end_time: 2026-08-06
|
||||
account: 1000000
|
||||
benchmark: QQQ
|
||||
exchange_kwargs:
|
||||
codes: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,113 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# Improved RankIC workflow: 300+ stock universe, proven RankICLGBModel params,
|
||||
# extended 12-month validation, full SP feature set (40 features).
|
||||
#
|
||||
# Changes from repro run:
|
||||
# 1. Single RankICLGBModel (not ensemble) — proven config from skill
|
||||
# 2. num_leaves=15 (not 31) — the verified value
|
||||
# 3. Universe expanded from 50 ETFs to 300+ single stocks + ETFs
|
||||
# 4. Validation extended to 12 months (2025-01 to 2026-01)
|
||||
# 5. Full 40 SP features (no leakage confirmed)
|
||||
# 6. Early stopping still at 200 (proven)
|
||||
#
|
||||
# Run:
|
||||
# rd_run_workflow config_path=tac-qlib/workflows/workflow_lgb_300sp_rankic.yaml \
|
||||
# experiment_name=tac-rd-300sp-rankic
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-300sp-rankic" }
|
||||
|
||||
task:
|
||||
model:
|
||||
# Single RankICLGBModel — proven config from tac-qlib-custom skill.
|
||||
# Per-day query groups + feval=rankic + metric='None' so early-stopping
|
||||
# tracks mean per-day Spearman instead of l2.
|
||||
class: RankICLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 15
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l1: 0.0
|
||||
lambda_l2: 0.5
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
seed: 2026
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
# Expanded universe: all lake symbols (instruments: "all" = every symbol with bars in the lake)
|
||||
instruments: "all"
|
||||
start_time: "2015-01-03"
|
||||
end_time: "2026-08-14"
|
||||
fit_start_time: "2016-01-04"
|
||||
fit_end_time: "2025-01-01"
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
# Full 40 SP features + 6 OHLCV = 46 features
|
||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_logp,sp_hurst_exponent,sp_ou_half_life,sp_ou_revert,sp_ou_zscore,sp_hmm_state,sp_hmm_p_regime1,sp_jump_flag,sp_jump_ratio,sp_jump_tail,sp_max_move,sp_max_up,sp_max_down,sp_rv1,sp_rv5,sp_rv22,sp_rv_ac1,sp_rv_cv_22,sp_vol_ratio_1_22,sp_vol_ratio_5_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_rskew_5,sp_rskew_22,sp_rkurt_5,sp_rkurt_22,sp_dsv_1,sp_dsv_5,sp_dsv_22,sp_dsv_ratio_1,sp_dsv_ratio_5,sp_dsv_ratio_22,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead,sp_sig_level2_lead_lag_5,sp_sig_level2_lag_lead_5"
|
||||
infer_processors:
|
||||
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-01-01" } }
|
||||
- { class: ProcessInf, kwargs: {} }
|
||||
- { class: CSRankNorm, kwargs: {} }
|
||||
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-01-01" } }
|
||||
- { class: Fillna, kwargs: {} }
|
||||
segments:
|
||||
train: ["2016-01-04", "2024-12-31"]
|
||||
valid: ["2025-01-02", "2026-01-02"]
|
||||
test: ["2026-01-04", "2026-08-14"]
|
||||
|
||||
record:
|
||||
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
|
||||
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs: { signal: "<PRED>", topk: 10, n_drop: 2, only_tradable: true, risk_degree: 0.95 }
|
||||
backtest:
|
||||
start_time: "2026-01-04"
|
||||
end_time: "2026-08-14"
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
codes: ""
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,145 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# CANONICAL: SP-5d LightGBM with the stochastic-control OptimalStopControl
|
||||
# strategy (entry-gated by signal percentile, optimal-stopping exits by
|
||||
# percentile / time stop / stop-loss, equal-weight control sizing).
|
||||
#
|
||||
# This is the stochastic-optimal-stopping strategy ported from the experiments:
|
||||
# - entry: a symbol opens only when its cross-sectional signal percentile
|
||||
# >= entry_pct and fewer than `topk` positions are open
|
||||
# - exit: percentile < exit_pct (continuation value too low), or
|
||||
# max_hold_days (finite-horizon time stop), or P&L <= sl
|
||||
# (loss control) after min_hold_days
|
||||
# - sizing: equal-weight control (risk_degree fraction of total value split
|
||||
# across targets)
|
||||
#
|
||||
# Strategy class: tac_qlib.contrib.strategy.optimal_stop.OptimalStopControl
|
||||
# Calibrate entry_pct / exit_pct / max_hold_days on the VALID window only
|
||||
# (the experiments showed valid-window calibration overfits; prefer robust
|
||||
# defaults: entry 0.85 / exit 0.7 / hold 10 / sl -0.08).
|
||||
#
|
||||
# Run:
|
||||
# rd_run_workflow config_path=tac-qlib/workflows/workflow_lgb_sp5d_optstop.yaml \
|
||||
# experiment_name=tac-rd-optstop
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-optstop"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.03
|
||||
num_leaves: 31
|
||||
n_estimators: 500
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seed: 42
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: "{{ UNIVERSE }}"
|
||||
start_time: 2015-01-03
|
||||
end_time: 2026-08-10
|
||||
fit_start_time: 2015-01-03
|
||||
fit_end_time: 2025-09-01
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||
infer_processors:
|
||||
- class: DropAllNaN
|
||||
kwargs: {}
|
||||
- class: ProcessInf
|
||||
kwargs: {}
|
||||
- class: CSRankNorm
|
||||
kwargs: {}
|
||||
- class: ZScoreNorm
|
||||
kwargs: {}
|
||||
- class: Fillna
|
||||
kwargs: {}
|
||||
segments:
|
||||
train: [2015-01-03, 2025-09-01]
|
||||
valid: [2025-09-03, 2026-01-03]
|
||||
test: [2026-01-04, 2026-08-10]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: OptimalStopControl
|
||||
module_path: tac_qlib.contrib.strategy.optimal_stop
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 10
|
||||
entry_pct: 0.85
|
||||
exit_pct: 0.7
|
||||
max_hold_days: 10
|
||||
min_hold_days: 2
|
||||
sl: -0.08
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-01-04
|
||||
end_time: 2026-08-10
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
codes: "{{ UNIVERSE }}"
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,142 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# CANONICAL: LightGBM with RankIC early-stopping on the 50-ETF SP-5d panel.
|
||||
#
|
||||
# Uses the tac-qlib contrib stack so no reinvention is needed:
|
||||
# - model: RankICLGBModel (tac_qlib.contrib.model.rank_gbdt) — early-stops
|
||||
# on per-day cross-sectional RankIC, not l2. The measured lever:
|
||||
# RankIC 0.047 -> 0.075 on the SP-5d signal, and with the tuned
|
||||
# budget the first config that beat SPY net of costs.
|
||||
# - handler: TACHandler (tac_qlib.contrib.data.handler) — lake features
|
||||
# - records: SignalRecord + SigAnaRecord + PortAnaRecord (TopkDropout)
|
||||
#
|
||||
# Feature columns are the 24 sp_* columns computed by the Rust get_lake_sp tool
|
||||
# (7 stochastic-process families: ou,hmm,jump,har,trend,hurst,signature). Any
|
||||
# other column present in the lake features parquet can be listed instead.
|
||||
#
|
||||
# Run:
|
||||
# rd_run_workflow config_path=tac-qlib/workflows/workflow_lgb_sp5d_rankic.yaml \
|
||||
# experiment_name=tac-rd-rankic
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-rankic"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: RankICLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 31
|
||||
n_estimators: 3000
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l2: 0.5
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seed: 42
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: "{{ UNIVERSE }}"
|
||||
start_time: 2015-01-03
|
||||
end_time: 2026-08-10
|
||||
fit_start_time: 2015-01-03
|
||||
fit_end_time: 2025-09-01
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||
infer_processors:
|
||||
- class: DropAllNaN
|
||||
kwargs: {}
|
||||
- class: ProcessInf
|
||||
kwargs: {}
|
||||
- class: CSRankNorm
|
||||
kwargs: {}
|
||||
- class: ZScoreNorm
|
||||
kwargs: {}
|
||||
- class: Fillna
|
||||
kwargs: {}
|
||||
segments:
|
||||
train: [2015-01-03, 2025-09-01]
|
||||
valid: [2025-09-03, 2026-01-03]
|
||||
test: [2026-01-04, 2026-08-10]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 10
|
||||
n_drop: 2
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-01-04
|
||||
end_time: 2026-08-10
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
codes: "{{ UNIVERSE }}"
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,137 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# Seed ensemble of the RankIC-early-stopping LightGBM on the 50-ETF SP-5d panel.
|
||||
#
|
||||
# Same canonical setup as workflow_lgb_sp5d_rankic.yaml but with
|
||||
# RankICEnsembleLGBModel (tac_qlib.contrib.model.rank_ensemble): 5 sub-models,
|
||||
# one per seed, identical hyper-parameters; predictions are the seed average.
|
||||
# The seeds train in a thread pool (parallel: 5), so this is ~2x faster than
|
||||
# the same 5 models serially on a 6-physical-core host.
|
||||
#
|
||||
# Run:
|
||||
# rd_run_workflow config_path=tac-qlib/workflows/workflow_lgb_sp5d_rankic_ensemble.yaml \
|
||||
# experiment_name=tac-rd-rankic-ensemble
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-rankic-ensemble"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: RankICEnsembleLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 31
|
||||
n_estimators: 3000
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l2: 0.5
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seeds: "42,7,2026,99,123"
|
||||
parallel: 5
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: "{{ UNIVERSE }}"
|
||||
start_time: 2015-01-03
|
||||
end_time: 2026-08-10
|
||||
fit_start_time: 2015-01-03
|
||||
fit_end_time: 2025-09-01
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||
infer_processors:
|
||||
- class: DropAllNaN
|
||||
kwargs: {}
|
||||
- class: ProcessInf
|
||||
kwargs: {}
|
||||
- class: CSRankNorm
|
||||
kwargs: {}
|
||||
- class: ZScoreNorm
|
||||
kwargs: {}
|
||||
- class: Fillna
|
||||
kwargs: {}
|
||||
segments:
|
||||
train: [2015-01-03, 2025-09-01]
|
||||
valid: [2025-09-03, 2026-01-03]
|
||||
test: [2026-01-04, 2026-08-10]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 10
|
||||
n_drop: 2
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-01-04
|
||||
end_time: 2026-08-10
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
codes: "{{ UNIVERSE }}"
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,103 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# Reproduction run of the RankIC-early-stopping LightGBM ensemble on 50-ETF SP-5d.
|
||||
# Matches the canonical ensemble but with trimmed SP features (no OU/HMM) and
|
||||
# fit_start_time shifted to 2016-01-04 to avoid warm-up NaN rows.
|
||||
#
|
||||
# Run:
|
||||
# rd_run_workflow config_path=tac-qlib/workflows/workflow_lgb_sp5d_rankic_ensemble_repro.yaml \
|
||||
# experiment_name=tac-rd-rank-ensemble-repro
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-rank-ensemble-repro" }
|
||||
|
||||
task:
|
||||
model:
|
||||
class: RankICEnsembleLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 31
|
||||
n_estimators: 3000
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l2: 0.5
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seeds: "42,7,2026,99,123"
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM"
|
||||
start_time: "2015-01-03"
|
||||
end_time: "2026-08-14"
|
||||
fit_start_time: "2016-01-04"
|
||||
fit_end_time: "2025-09-01"
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead"
|
||||
infer_processors:
|
||||
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
|
||||
- { class: ProcessInf, kwargs: {} }
|
||||
- { class: CSRankNorm, kwargs: {} }
|
||||
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
|
||||
- { class: Fillna, kwargs: {} }
|
||||
segments:
|
||||
train: ["2016-01-04", "2025-09-01"]
|
||||
valid: ["2025-09-03", "2026-01-03"]
|
||||
test: ["2026-01-04", "2026-08-10"]
|
||||
|
||||
record:
|
||||
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
|
||||
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs: { signal: "<PRED>", topk: 10, n_drop: 2, only_tradable: true, risk_degree: 0.95 }
|
||||
backtest:
|
||||
start_time: "2026-01-04"
|
||||
end_time: "2026-08-10"
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
codes: "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM"
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,129 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# LightGBM on the TradeAC lake -- qrun workflow (train -> signal -> backtest).
|
||||
#
|
||||
# Run it like a stock qlib project:
|
||||
#
|
||||
# cd tac-qlib
|
||||
# qrun workflows/workflow_lgb_taclake.yaml \
|
||||
# --experiment_name tac-lake-lgb --uri_folder mlruns
|
||||
#
|
||||
# Or with a custom lake root:
|
||||
#
|
||||
# TAC_LAKE_DIR=/path/to/lake qrun workflows/workflow_lgb_taclake.yaml \
|
||||
# --experiment_name tac-lake-lgb
|
||||
#
|
||||
# The lake providers (calendar/instrument/feature) are wired in `qlib_init`; the
|
||||
# expression engine and backtest Exchange stay upstream qlib. The TACHandler reads
|
||||
# OHLCV + ta-lib features straight from the parquet lake.
|
||||
#
|
||||
# Segment split (the lake holds 1d bars since 2026-02-09):
|
||||
# train 2026-03-01..2026-05-31 / valid 2026-06-01..2026-06-30 / test 2026-07-01..2026-08-06
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
# --- lake-backed providers (see tac_qlib.data.providers) -----------------
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
# sqlite backend avoids mlflow's filesystem-backend maintenance-mode opt-out
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///mlruns.db"
|
||||
default_exp_name: "tac-lake-demo"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.05
|
||||
num_leaves: 15
|
||||
n_estimators: 200
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.01
|
||||
reg_lambda: 0.01
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: all
|
||||
start_time: 2026-03-01
|
||||
end_time: 2026-08-06
|
||||
fit_start_time: 2026-03-01
|
||||
fit_end_time: 2026-05-31
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
segments:
|
||||
train: [2026-03-01, 2026-05-31]
|
||||
valid: [2026-06-01, 2026-06-30]
|
||||
test: [2026-07-01, 2026-08-06]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 2
|
||||
n_drop: 1
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-07-01
|
||||
end_time: 2026-08-06
|
||||
account: 1000000
|
||||
# any symbol the lake holds works; the lake has no index quotes yet
|
||||
benchmark: AAPL
|
||||
exchange_kwargs:
|
||||
codes: all
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,55 @@
|
||||
/**
|
||||
* tls-server.cjs — production HTTPS entrypoint for TradeAC.
|
||||
*
|
||||
* Serves the compiled Next.js app (tac-app/.next) over HTTPS with the
|
||||
* self-signed certificate generated by the container entrypoint
|
||||
* (entrypoint.sh) when SERVER_TLS=true.
|
||||
*
|
||||
* It reuses Next's normal production server (next({ dev: false }) +
|
||||
* getRequestHandler()), so middleware, server actions, and all routes behave
|
||||
* exactly like `next start` — just over TLS.
|
||||
*/
|
||||
"use strict";
|
||||
|
||||
const fs = require("node:fs");
|
||||
const https = require("node:https");
|
||||
const path = require("node:path");
|
||||
const next = require("next");
|
||||
|
||||
const port = Number(process.env.PORT || 3000);
|
||||
const host = process.env.HOSTNAME || "0.0.0.0";
|
||||
const tlsKeyPath = process.env.TLS_KEY || "/tmp/tls/key.pem";
|
||||
const tlsCertPath = process.env.TLS_CERT || "/tmp/tls/cert.pem";
|
||||
const tlsHost = process.env.TLS_HOST || "localhost";
|
||||
|
||||
async function main() {
|
||||
const app = next({
|
||||
dev: false,
|
||||
dir: path.join(__dirname, "tac-app"),
|
||||
hostname: host,
|
||||
port,
|
||||
});
|
||||
const handle = app.getRequestHandler();
|
||||
|
||||
await app.prepare();
|
||||
|
||||
const key = fs.readFileSync(tlsKeyPath);
|
||||
const cert = fs.readFileSync(tlsCertPath);
|
||||
|
||||
const server = https.createServer({ key, cert }, (req, res) => handle(req, res));
|
||||
server.listen(port, host, () => {
|
||||
console.log(`> TradeAC HTTPS (self-signed) ready on https://${tlsHost}:${port}`);
|
||||
});
|
||||
|
||||
const shutdown = () => {
|
||||
server.close(() => process.exit(0));
|
||||
setTimeout(() => process.exit(0), 2000).unref();
|
||||
};
|
||||
process.on("SIGTERM", shutdown);
|
||||
process.on("SIGINT", shutdown);
|
||||
}
|
||||
|
||||
main().catch((err) => {
|
||||
console.error("Failed to start HTTPS server:", err);
|
||||
process.exit(1);
|
||||
});
|
||||
Reference in New Issue
Block a user