book: scaffold + ch00 (execution trail as spine) — evidence exp 8-31, round 3

This commit is contained in:
TradeAC Book Agent
2026-08-18 22:35:23 +00:00
commit c93424e76c
83 changed files with 17676 additions and 0 deletions
+109
View File
@@ -0,0 +1,109 @@
# -----------------------------------------------------------------------------
# Basic LightGBM qrun workflow on the TradeAC lake -- short window sanity run.
# Train on ~3 months, early-stop on 1 month valid, predict+backtest on ~1 month.
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///mlruns.db"
default_exp_name: "tac-basic-short"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
kwargs:
loss: mse
learning_rate: 0.05
num_leaves: 15
n_estimators: 200
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.01
reg_lambda: 0.01
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: all
start_time: 2026-03-01
end_time: 2026-08-14
fit_start_time: 2026-03-01
fit_end_time: 2026-05-31
freq: day
lake_root: "{{ LAKE }}"
market: US
segments:
train: [2026-03-01, 2026-05-31]
valid: [2026-06-01, 2026-06-30]
test: [2026-07-01, 2026-08-14]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 2
n_drop: 1
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-07-01
end_time: 2026-08-14
account: 1000000
benchmark: QQQ
exchange_kwargs:
codes: all
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
+129
View File
@@ -0,0 +1,129 @@
# -----------------------------------------------------------------------------
# Tune run 1: wider, longer-horizon, de-duplicated universe.
#
# Baseline (exp 1 / run f29f5446): IC 0.071 / ICIR 0.17, Rank IC ~0.014;
# strategy +4.9% ann (raw) vs benchmark ~+89% ann; excess return w/ cost
# -0.94 ann, IR -2.23, excess max drawdown -18.9%. topk=2 with 24 trades over
# 27 days on a universe of correlated ETFs + leveraged hedges (VXX/USO/SLV)
# produced high turnover and a portfolio that trailed AAPL badly.
#
# Changes:
# - universe: drop leveraged/noisy names (VXX, USO, SLV, BIL) and near-
# duplicate index baskets (GPIQ, QQQE, KTEC); keep 10 liquid core names.
# - label: 5-day forward return (Ref($close,-6)/Ref($close,-1)-1) to cut
# single-day noise and match the intended holding horizon.
# - topk 2 -> 5, n_drop 1: more diversification, lower turnover per name.
# - benchmark AAPL -> QQQ (a real index ETF the universe tracks).
# - model: learning_rate 0.03, 300 estimators (slower, deeper fit).
#
# Trigger:
# rd_run_workflow config_path=tac-qlib/workflows/tune_run1_wider_5d.yaml \
# experiment_name=tac-rd-tune
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///mlruns.db"
default_exp_name: "tac-rd-tune"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
kwargs:
loss: mse
learning_rate: 0.03
num_leaves: 15
n_estimators: 300
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.01
reg_lambda: 0.01
seed: 2026
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
start_time: 2000-01-03
end_time: 2026-08-06
fit_start_time: 2026-03-01
fit_end_time: 2026-05-31
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
segments:
train: [2026-03-01, 2026-05-31]
valid: [2026-06-01, 2026-06-30]
test: [2026-07-01, 2026-08-06]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 5
n_drop: 1
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-07-01
end_time: 2026-08-06
account: 1000000
benchmark: QQQ
exchange_kwargs:
codes: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -0,0 +1,127 @@
# -----------------------------------------------------------------------------
# Tune run 2: same-day signal, strongly regularized model, 3x rotating book.
#
# Baseline (exp 1 / run f29f5446): IC 0.071 / ICIR 0.17, Rank IC ~0.014;
# excess return w/ cost -0.94 ann, IR -2.23. The 1-day signal was noisy
# (Rank IC ~ 0) and the topk=2 book turned over 24 times in 27 days, paying
# ~1.1% of the $1M account in costs.
#
# Changes (isolates model/backtest effects; universe + label same as baseline):
# - model: stronger regularization (reg_alpha 0.5, reg_lambda 5.0,
# subsample 0.7, colsample 0.6) to combat the unstable Rank IC.
# - topk 2 -> 3, n_drop 1 -> 2: rotate out losers faster (lower cost drag,
# higher turnover on only the worst names).
# - benchmark AAPL -> QQQ.
# - universe: drop leveraged/duplicate names (VXX, USO, SLV, BIL, GPIQ,
# QQQE, KTEC) for a cleaner cross-section; keeps baseline 1-day label.
#
# Trigger:
# rd_run_workflow config_path=tac-qlib/workflows/tune_run2_regularized.yaml \
# experiment_name=tac-rd-tune
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///mlruns.db"
default_exp_name: "tac-rd-tune"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
kwargs:
loss: mse
learning_rate: 0.05
num_leaves: 15
n_estimators: 250
colsample_bytree: 0.6
subsample: 0.7
subsample_freq: 1
reg_alpha: 0.5
reg_lambda: 5.0
seed: 2026
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
start_time: 2000-01-03
end_time: 2026-08-06
fit_start_time: 2026-03-01
fit_end_time: 2026-05-31
freq: day
lake_root: "{{ LAKE }}"
market: US
segments:
train: [2026-03-01, 2026-05-31]
valid: [2026-06-01, 2026-06-30]
test: [2026-07-01, 2026-08-06]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 3
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-07-01
end_time: 2026-08-06
account: 1000000
benchmark: QQQ
exchange_kwargs:
codes: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -0,0 +1,146 @@
# -----------------------------------------------------------------------------
# Tune run 3 (NEXT run): 5-day label + clean 10-name universe.
#
# Baseline (exp 1 / run f29f5446):
# IC 0.071, ICIR 0.17, Rank IC 0.014, Rank ICIR 0.03 -> ranking ~ coin flip
# valid l2 best at round 0 and never improved (early-stopped ~50 rounds, overfit)
# backtest: strategy +4.9% ann (raw) vs equal-weight universe +89.2% ann
# (benchmark was unset -> qlib used equal-weight), excess w/ cost -94.0% ann,
# IR -2.23, max DD -18.9%. topk=2, 24 trades/27 days, $11.1k cost (1.1% of $1M),
# ending book ~97.5% in AAPL+IBIT (two names, both ~49%).
#
# PRIMARY LEVER (change one thing, everything else held at baseline):
# label: 1-day next return -> 5-day forward return
# "Ref($close,-6)/Ref($close,-1)-1".
# Rationale: Rank ICIR 0.03 is the binding constraint - a topk book's return
# is bounded by ranking quality, and no backtest tuning fixes a non-existent
# ranking. The retained TA features (rsi_14, macd_hist, ema_20, volume,
# stoch, aroon) are momentum/mean-reversion proxies that predict multi-day
# drift, not overnight noise; and the avg holding in the baseline book was
# several days, so a 1-day label mismatches the holding horizon.
#
# SUPPORTING (kept minimal, flagged for attribution):
# - universe 17 -> 10: drop leveraged/vol/cash names (VXX, USO, SLV, BIL)
# and near-duplicate index baskets (GPIQ, QQQE, KTEC). 17 names were really
# ~8 independent betas (QQQ/QQQE/IVV/SMH/AIQ overlap heavily).
# - topk 2 -> 5, n_drop 1 -> 2: stop the 2-name lottery, cut per-name turnover.
# - benchmark: unset -> QQQ (a real index ETF the universe tracks; the
# "excess return" vs equal-weight of a 17-name universe is misleading).
# - model: explicitly num_boost_round 1000 + early_stopping_rounds 50 so the
# round count is actually controlled (baseline's n_estimators: 200 was a
# no-op, swallowed into lgb params; rounds were the 1000 default).
# Hyperparameters otherwise identical to baseline (lr 0.05, num_leaves 15,
# reg 0.01/0.01) for a clean label A/B.
#
# Trigger into a NEW experiment (do not pollute exp 1):
# rd_run_workflow config_path=tac-qlib/workflows/tune_run3_label5d_clean_universe.yaml \
# experiment_name=tac-rd-tune
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-tune"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
kwargs:
loss: mse
learning_rate: 0.05
num_leaves: 15
num_boost_round: 1000
early_stopping_rounds: 50
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.01
reg_lambda: 0.01
seed: 2026
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
start_time: 2000-01-03
end_time: 2026-08-06
fit_start_time: 2026-03-01
fit_end_time: 2026-05-31
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
segments:
train: [2026-03-01, 2026-05-31]
valid: [2026-06-01, 2026-06-30]
test: [2026-07-01, 2026-08-06]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 5
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-07-01
end_time: 2026-08-06
account: 1000000
benchmark: QQQ
exchange_kwargs:
codes: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -0,0 +1,121 @@
# -----------------------------------------------------------------------------
# Run 94736d89 (exp-4 tac-rd-tune2) follow-up -- single lever: WIDER UNIVERSE.
#
# Baseline (run 94736d89): 10 correlated tech/growth names -> weak cross-section
# (IC 0.038 / ICIR 0.10), topk=5 book all-correlated, 295 trades / 152d and
# $58k cost drag (5.8% of $1M) -> excess ann -18.8% vs QQQ.
#
# This run holds EVERYTHING else fixed (windows, 5-day label, LGB hyperparams,
# topk=5/n_drop=2, benchmark QQQ) and only widens the universe 10 -> 17 with the
# full lake set, adding genuinely uncorrelated assets (BIL cash, USO oil, SLV
# silver, VXX vol, KTEC/QQQE/GPIQ factor sleeves) to de-correlate the cross-section,
# stabilize the top-5 ranking and cut the churn/cost drag.
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///mlruns.db"
default_exp_name: "tac-rd-tune3"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
kwargs:
loss: mse
learning_rate: 0.05
num_leaves: 15
num_boost_round: 1000
early_stopping_rounds: 50
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.01
reg_lambda: 0.01
seed: 2026
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ,BIL,GPIQ,KTEC,QQQE,SLV,USO,VXX
start_time: 2000-01-03
end_time: 2026-08-01
fit_start_time: 2024-06-03
fit_end_time: 2025-11-28
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
segments:
train: [2024-06-03, 2025-11-28]
valid: [2025-12-01, 2025-12-31]
test: [2026-01-01, 2026-08-01]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 5
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-01-01
end_time: 2026-08-01
account: 1000000
benchmark: QQQ
exchange_kwargs:
codes: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ,BIL,GPIQ,KTEC,QQQE,SLV,USO,VXX
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -0,0 +1,144 @@
# -----------------------------------------------------------------------------
# Tune run 4 (NEXT run): fix the universe bug + extend the train window.
#
# Previous (exp 3 / run 1e170e7f): IC -0.025 / ICIR -0.071 / RankIC -0.022 /
# RankICIR -0.073 (noise), Long-Short -27% ann; excess +15.8% ann w/ cost
# (IR 1.35) vs QQQ; $1M -> $971.7k (-2.8%); 65 trades/27d, $12.1k cost.
# l2.train 0.35 vs l2.valid 0.95 -> gross overfit (valid best at round 0,
# early-stopped at 16 trees).
#
# CRITICAL BUG in that run: the 10-name universe was silently IGNORED.
# TACHandler passes `instruments` as a comma-separated STRING; qlib wraps it
# as {"market": "<comma string>", "filter_pipe": []}; LakeInstrumentProvider
# ._resolve_symbols() only handles list/tuple/ndarray and falls through to
# load_symbols() = the ENTIRE 17-symbol lake. So the model trained/traded on
# VXX, USO, SLV, BIL, GPIQ, QQQE, KTEC too - exactly the leveraged/hedge
# names the "clean 10-name universe" hypothesis meant to drop. The universe
# A/B is UNTESTED.
# Fix (providers.py:100 _resolve_symbols): split comma-separated strings.
#
# PRIMARY LEVER (this run, ONE hypothesis):
# universe = the intended 10-name dedup pool (AAPL,MSFT,TSLA,QQQ,IVV,SMH,
# TLT,IBIT,MCHI,AIQ), now actually enforced, + train window 3 months -> 2
# years. The 3-month window (~1000 rows for a 21-feature GBDT) is the hard
# ceiling on signal; features span 2000-2026 so more data is free.
# Everything else held at run-1e170e7f for a clean A/B: 5-day label,
# LGB baseline hyperparams, topk 5 / n_drop 2, benchmark QQQ.
#
# SUPPORTING (flagged, NOT changed this run to keep attribution clean):
# - if valid loss still rises monotonically after 2y of data, next step is
# regularization (reg_alpha/lambda 0.01 -> ~0.5, num_leaves 15 -> 10,
# lr 0.05 -> 0.02) rather than label/topk changes.
#
# Trigger into a NEW experiment (do not pollute exp 1/3):
# rd_run_workflow config_path=tac-qlib/workflows/tune_run4_fix_universe_longtrain.yaml \
# experiment_name=tac-rd-tune2
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-tune2"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
kwargs:
loss: mse
learning_rate: 0.05
num_leaves: 15
num_boost_round: 1000
early_stopping_rounds: 50
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.01
reg_lambda: 0.01
seed: 2026
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
start_time: 2000-01-03
end_time: 2026-08-06
fit_start_time: 2024-06-03
fit_end_time: 2026-05-31
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
segments:
train: [2024-06-03, 2026-05-31]
valid: [2026-06-01, 2026-06-30]
test: [2026-07-01, 2026-08-06]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 5
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-07-01
end_time: 2026-08-06
account: 1000000
benchmark: QQQ
exchange_kwargs:
codes: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
+133
View File
@@ -0,0 +1,133 @@
# -----------------------------------------------------------------------------
# Tune run 5: longer backtest window (2026-01-01 -> 2026-08-01).
#
# Purpose: test the fixed universe provider (_resolve_symbols now honors the
# comma-separated 10-name instruments) and the fixed artifact pinning
# (mlruns/<exp_id>/<run_id>/) over a 7-month out-of-sample window instead of
# the single month (Jul) of run 47e9e369 / tune_run4.
#
# Changes vs tune_run4_fix_universe_longtrain.yaml:
# - test/backtest window 2026-07-01..08-06 -> 2026-01-01..2026-08-01
# - train/valid moved back so they stay strictly before test (no leakage):
# train: 2024-06-03 .. 2025-11-28 (~18 months, ~4500 rows x 10 names)
# valid: 2025-12-01 .. 2025-12-31 (1 month, right before test)
# test : 2026-01-01 .. 2026-08-01 (7 months)
# - everything else held fixed: 5-day label, LGB baseline hyperparams,
# topk 5 / n_drop 2, benchmark QQQ, universe 10 names.
#
# NOTE: requires the providers.py fix so the universe is actually 10 names
# (not silently expanded to all 17 lake symbols).
#
# Trigger (existing experiment, exp id 4 -> artifacts under
# $TAC_LAKE_DIR/mlruns/4/<run_id>/ ):
# rd_run_workflow config_path=tac-qlib/workflows/tune_run5_longtest.yaml \
# experiment_name=tac-rd-tune2
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-tune2"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
kwargs:
loss: mse
learning_rate: 0.05
num_leaves: 15
num_boost_round: 1000
early_stopping_rounds: 50
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.01
reg_lambda: 0.01
seed: 2026
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
start_time: 2000-01-03
end_time: 2026-08-01
fit_start_time: 2024-06-03
fit_end_time: 2025-11-28
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
segments:
train: [2024-06-03, 2025-11-28]
valid: [2025-12-01, 2025-12-31]
test: [2026-01-01, 2026-08-01]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 5
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-01-01
end_time: 2026-08-01
account: 1000000
benchmark: QQQ
exchange_kwargs:
codes: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -0,0 +1,148 @@
# -----------------------------------------------------------------------------
# Tune run 6 (NEXT run): wider 10-name universe A/B vs run f744455056 (exp 1).
#
# Baseline (exp 1 / run f744455056 — this run):
# Input : universe AAPL,MSFT,QQQ,IVV,SMH,TLT (6 names, 5 of them the same
# tech beta); 21 features (OHLCV + TA); label 1-day next return;
# LGB lr 0.05 / 15 leaves / 200 trees / reg 0.01,0.01;
# train 03-01..05-31 / valid 06-01..06-30 / test 07-01..08-06.
# Output: IC 0.048, ICIR 0.09, Rank IC 0.065, Rank ICIR 0.13 -> noise-level
# (per-day n=6, IC swings -0.89..+0.74 with many null days).
# Backtest had NO benchmark (benchmark null) -> the "+180% ann, IR 6.4"
# headline is raw strategy return, not excess. Strategy +16.5% over 27
# days, but ~half the P&L came from ONE day (2026-07-30 MSFT +14% sell,
# +$72k realized). 30 trades/27 days, $15.3k cost (1.5% of $1M),
# ending book 46.6% SMH + 50.8% TLT (2-name lottery).
#
# PRIMARY LEVER (change one thing, everything else held at baseline):
# universe: 6 -> 10 names (AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ).
# Rationale: with 6 near-collinear names there is nothing to rank — ICIR 0.09
# is cross-sectional noise and the topk book just re-buys tech momentum on
# correlated bets. Widening to ~10 independent-ish betas (mega tech, semis,
# S&P, Nasdaq, bonds, BTC, EM, robotics) gives the cross-section real breadth,
# stabilizes IC, and makes a diversified topk book possible.
#
# SUPPORTING (kept minimal, flagged for attribution):
# - topk 2 -> 4, n_drop 1 -> 2: kill the 2-name lottery, cut per-name churn.
# - benchmark: unset -> QQQ: the baseline "excess return" was raw strategy
# return because no benchmark was wired; QQQ is the index the tech-heavy
# universe tracks.
# - model: explicit num_boost_round 1000 + early_stopping_rounds 50 so round
# count is controlled (baseline's n_estimators: 200 was swallowed into lgb
# params and valid l2 rose monotonically -> overfit). Hyperparameters
# otherwise identical to baseline for a clean universe A/B.
# - label: KEPT at 1-day next return so this run isolates the universe lever;
# a 5-day horizon is the natural NEXT experiment (see tune_run3).
#
# Trigger into a NEW experiment (do not pollute exp 1); evolved_from = f744455056:
# rd_run_workflow config_path=tac-qlib/workflows/tune_run6_wider_universe_ab.yaml \
# experiment_name=tac-rd-tune
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-tune"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
kwargs:
loss: mse
learning_rate: 0.05
num_leaves: 15
num_boost_round: 1000
early_stopping_rounds: 50
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.01
reg_lambda: 0.01
seed: 2026
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
start_time: 2000-01-03
end_time: 2026-08-06
fit_start_time: 2026-03-01
fit_end_time: 2026-05-31
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-2)/Ref($close,-1)-1"
segments:
train: [2026-03-01, 2026-05-31]
valid: [2026-06-01, 2026-06-30]
test: [2026-07-01, 2026-08-06]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 4
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-07-01
end_time: 2026-08-06
account: 1000000
benchmark: QQQ
exchange_kwargs:
codes: AAPL,MSFT,TSLA,QQQ,IVV,SMH,TLT,IBIT,MCHI,AIQ
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -0,0 +1,113 @@
# -----------------------------------------------------------------------------
# Improved RankIC workflow: 300+ stock universe, proven RankICLGBModel params,
# extended 12-month validation, full SP feature set (40 features).
#
# Changes from repro run:
# 1. Single RankICLGBModel (not ensemble) — proven config from skill
# 2. num_leaves=15 (not 31) — the verified value
# 3. Universe expanded from 50 ETFs to 300+ single stocks + ETFs
# 4. Validation extended to 12 months (2025-01 to 2026-01)
# 5. Full 40 SP features (no leakage confirmed)
# 6. Early stopping still at 200 (proven)
#
# Run:
# rd_run_workflow config_path=tac-qlib/workflows/workflow_lgb_300sp_rankic.yaml \
# experiment_name=tac-rd-300sp-rankic
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-300sp-rankic" }
task:
model:
# Single RankICLGBModel — proven config from tac-qlib-custom skill.
# Per-day query groups + feval=rankic + metric='None' so early-stopping
# tracks mean per-day Spearman instead of l2.
class: RankICLGBModel
module_path: tac_qlib.contrib.model.rank_gbdt
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 15
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l1: 0.0
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
seed: 2026
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
# Expanded universe: all lake symbols (instruments: "all" = every symbol with bars in the lake)
instruments: "all"
start_time: "2015-01-03"
end_time: "2026-08-14"
fit_start_time: "2016-01-04"
fit_end_time: "2025-01-01"
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
# Full 40 SP features + 6 OHLCV = 46 features
feature_fields: "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_logp,sp_hurst_exponent,sp_ou_half_life,sp_ou_revert,sp_ou_zscore,sp_hmm_state,sp_hmm_p_regime1,sp_jump_flag,sp_jump_ratio,sp_jump_tail,sp_max_move,sp_max_up,sp_max_down,sp_rv1,sp_rv5,sp_rv22,sp_rv_ac1,sp_rv_cv_22,sp_vol_ratio_1_22,sp_vol_ratio_5_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_rskew_5,sp_rskew_22,sp_rkurt_5,sp_rkurt_22,sp_dsv_1,sp_dsv_5,sp_dsv_22,sp_dsv_ratio_1,sp_dsv_ratio_5,sp_dsv_ratio_22,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead,sp_sig_level2_lead_lag_5,sp_sig_level2_lag_lead_5"
infer_processors:
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-01-01" } }
- { class: ProcessInf, kwargs: {} }
- { class: CSRankNorm, kwargs: {} }
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-01-01" } }
- { class: Fillna, kwargs: {} }
segments:
train: ["2016-01-04", "2024-12-31"]
valid: ["2025-01-02", "2026-01-02"]
test: ["2026-01-04", "2026-08-14"]
record:
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs: { signal: "<PRED>", topk: 10, n_drop: 2, only_tradable: true, risk_degree: 0.95 }
backtest:
start_time: "2026-01-04"
end_time: "2026-08-14"
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: ""
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -0,0 +1,145 @@
# -----------------------------------------------------------------------------
# CANONICAL: SP-5d LightGBM with the stochastic-control OptimalStopControl
# strategy (entry-gated by signal percentile, optimal-stopping exits by
# percentile / time stop / stop-loss, equal-weight control sizing).
#
# This is the stochastic-optimal-stopping strategy ported from the experiments:
# - entry: a symbol opens only when its cross-sectional signal percentile
# >= entry_pct and fewer than `topk` positions are open
# - exit: percentile < exit_pct (continuation value too low), or
# max_hold_days (finite-horizon time stop), or P&L <= sl
# (loss control) after min_hold_days
# - sizing: equal-weight control (risk_degree fraction of total value split
# across targets)
#
# Strategy class: tac_qlib.contrib.strategy.optimal_stop.OptimalStopControl
# Calibrate entry_pct / exit_pct / max_hold_days on the VALID window only
# (the experiments showed valid-window calibration overfits; prefer robust
# defaults: entry 0.85 / exit 0.7 / hold 10 / sl -0.08).
#
# Run:
# rd_run_workflow config_path=tac-qlib/workflows/workflow_lgb_sp5d_optstop.yaml \
# experiment_name=tac-rd-optstop
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-optstop"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
kwargs:
loss: mse
learning_rate: 0.03
num_leaves: 31
n_estimators: 500
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seed: 42
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-10
fit_start_time: 2015-01-03
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
infer_processors:
- class: DropAllNaN
kwargs: {}
- class: ProcessInf
kwargs: {}
- class: CSRankNorm
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments:
train: [2015-01-03, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: OptimalStopControl
module_path: tac_qlib.contrib.strategy.optimal_stop
kwargs:
signal: "<PRED>"
topk: 10
entry_pct: 0.85
exit_pct: 0.7
max_hold_days: 10
min_hold_days: 2
sl: -0.08
risk_degree: 0.95
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -0,0 +1,142 @@
# -----------------------------------------------------------------------------
# CANONICAL: LightGBM with RankIC early-stopping on the 50-ETF SP-5d panel.
#
# Uses the tac-qlib contrib stack so no reinvention is needed:
# - model: RankICLGBModel (tac_qlib.contrib.model.rank_gbdt) — early-stops
# on per-day cross-sectional RankIC, not l2. The measured lever:
# RankIC 0.047 -> 0.075 on the SP-5d signal, and with the tuned
# budget the first config that beat SPY net of costs.
# - handler: TACHandler (tac_qlib.contrib.data.handler) — lake features
# - records: SignalRecord + SigAnaRecord + PortAnaRecord (TopkDropout)
#
# Feature columns are the 24 sp_* columns computed by the Rust get_lake_sp tool
# (7 stochastic-process families: ou,hmm,jump,har,trend,hurst,signature). Any
# other column present in the lake features parquet can be listed instead.
#
# Run:
# rd_run_workflow config_path=tac-qlib/workflows/workflow_lgb_sp5d_rankic.yaml \
# experiment_name=tac-rd-rankic
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-rankic"
task:
model:
class: RankICLGBModel
module_path: tac_qlib.contrib.model.rank_gbdt
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seed: 42
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-10
fit_start_time: 2015-01-03
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
infer_processors:
- class: DropAllNaN
kwargs: {}
- class: ProcessInf
kwargs: {}
- class: CSRankNorm
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments:
train: [2015-01-03, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 10
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -0,0 +1,137 @@
# -----------------------------------------------------------------------------
# Seed ensemble of the RankIC-early-stopping LightGBM on the 50-ETF SP-5d panel.
#
# Same canonical setup as workflow_lgb_sp5d_rankic.yaml but with
# RankICEnsembleLGBModel (tac_qlib.contrib.model.rank_ensemble): 5 sub-models,
# one per seed, identical hyper-parameters; predictions are the seed average.
# The seeds train in a thread pool (parallel: 5), so this is ~2x faster than
# the same 5 models serially on a 6-physical-core host.
#
# Run:
# rd_run_workflow config_path=tac-qlib/workflows/workflow_lgb_sp5d_rankic_ensemble.yaml \
# experiment_name=tac-rd-rankic-ensemble
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-rankic-ensemble"
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
parallel: 5
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-10
fit_start_time: 2015-01-03
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
infer_processors:
- class: DropAllNaN
kwargs: {}
- class: ProcessInf
kwargs: {}
- class: CSRankNorm
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments:
train: [2015-01-03, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 10
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -0,0 +1,103 @@
# -----------------------------------------------------------------------------
# Reproduction run of the RankIC-early-stopping LightGBM ensemble on 50-ETF SP-5d.
# Matches the canonical ensemble but with trimmed SP features (no OU/HMM) and
# fit_start_time shifted to 2016-01-04 to avoid warm-up NaN rows.
#
# Run:
# rd_run_workflow config_path=tac-qlib/workflows/workflow_lgb_sp5d_rankic_ensemble_repro.yaml \
# experiment_name=tac-rd-rank-ensemble-repro
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-rank-ensemble-repro" }
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM"
start_time: "2015-01-03"
end_time: "2026-08-14"
fit_start_time: "2016-01-04"
fit_end_time: "2025-09-01"
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead"
infer_processors:
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: ProcessInf, kwargs: {} }
- { class: CSRankNorm, kwargs: {} }
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: Fillna, kwargs: {} }
segments:
train: ["2016-01-04", "2025-09-01"]
valid: ["2025-09-03", "2026-01-03"]
test: ["2026-01-04", "2026-08-10"]
record:
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs: { signal: "<PRED>", topk: 10, n_drop: 2, only_tradable: true, risk_degree: 0.95 }
backtest:
start_time: "2026-01-04"
end_time: "2026-08-10"
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -0,0 +1,129 @@
# -----------------------------------------------------------------------------
# LightGBM on the TradeAC lake -- qrun workflow (train -> signal -> backtest).
#
# Run it like a stock qlib project:
#
# cd tac-qlib
# qrun workflows/workflow_lgb_taclake.yaml \
# --experiment_name tac-lake-lgb --uri_folder mlruns
#
# Or with a custom lake root:
#
# TAC_LAKE_DIR=/path/to/lake qrun workflows/workflow_lgb_taclake.yaml \
# --experiment_name tac-lake-lgb
#
# The lake providers (calendar/instrument/feature) are wired in `qlib_init`; the
# expression engine and backtest Exchange stay upstream qlib. The TACHandler reads
# OHLCV + ta-lib features straight from the parquet lake.
#
# Segment split (the lake holds 1d bars since 2026-02-09):
# train 2026-03-01..2026-05-31 / valid 2026-06-01..2026-06-30 / test 2026-07-01..2026-08-06
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
# --- lake-backed providers (see tac_qlib.data.providers) -----------------
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
# sqlite backend avoids mlflow's filesystem-backend maintenance-mode opt-out
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///mlruns.db"
default_exp_name: "tac-lake-demo"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
kwargs:
loss: mse
learning_rate: 0.05
num_leaves: 15
n_estimators: 200
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.01
reg_lambda: 0.01
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: all
start_time: 2026-03-01
end_time: 2026-08-06
fit_start_time: 2026-03-01
fit_end_time: 2026-05-31
freq: day
lake_root: "{{ LAKE }}"
market: US
segments:
train: [2026-03-01, 2026-05-31]
valid: [2026-06-01, 2026-06-30]
test: [2026-07-01, 2026-08-06]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 2
n_drop: 1
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-07-01
end_time: 2026-08-06
account: 1000000
# any symbol the lake holds works; the lake has no index quotes yet
benchmark: AAPL
exchange_kwargs:
codes: all
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d