Compare commits

..
47 changed files with 1364 additions and 1448 deletions
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,48 @@
#!/usr/bin/env python3
"""Precompute the signal-quality gate series and save to pickle.
Usage:
python precompute_signal_quality_gate.py <pred_path> <output_path> [topk] [lookback] [threshold]
Example:
python precompute_signal_quality_gate.py \
/home/data/lake/mlruns/49/34165f27e4a34378ad54843a079a78c0/artifacts/pred.pkl \
/app/experiments/book/data/signal_quality_gate/sq_gate_5d_0.50.pkl \
10 5 0.5
"""
import sys
import pickle
from pathlib import Path
# Add tac-qlib to path
sys.path.insert(0, "/app/tac-qlib")
from tac_qlib.contrib.strategy.signal_quality_gate import compute_signal_quality_gate
if __name__ == "__main__":
if len(sys.argv) < 3:
print(__doc__)
sys.exit(1)
pred_path = sys.argv[1]
output_path = sys.argv[2]
topk = int(sys.argv[3]) if len(sys.argv) > 3 else 10
lookback = int(sys.argv[4]) if len(sys.argv) > 4 else 5
threshold = float(sys.argv[5]) if len(sys.argv) > 5 else 0.5
lake_root = "/home/data/lake"
print(f"Computing signal-quality gate: topk={topk}, lookback={lookback}, threshold={threshold}")
gate = compute_signal_quality_gate(
pred_path,
lake_root=lake_root,
topk=topk,
lookback=lookback,
threshold=threshold,
)
print(f"Gate: {gate.sum()}/{len(gate)} days open ({gate.mean():.1%})")
with open(output_path, "wb") as f:
pickle.dump(gate, f)
print(f"Saved to {output_path}")
@@ -1,16 +1,16 @@
# ----------------------------------------------------------------------------- # -----------------------------------------------------------------------------
# ABLATION A (baseline): LightGBM with RankIC early-stopping on the 50-ETF SP-5d # Signal-quality gate: TopkDropout gated by rolling hit-rate of topk picks.
# panel, using ALL 24 sp_* feature columns (ou,hmm,jump,har,trend,hurst,
# signature). Copy of the canonical workflow_lgb_sp5d_rankic.yaml with a
# distinct experiment name so the ablation runs are isolated.
# #
# Run: # 1. Compute the gate: python precompute_signal_quality_gate.py <pred.pkl> <gate.pkl>
# rd_run_workflow config_path=tac-qlib/workflows/ablate_baseline_all_sp_fields.yaml \ # 2. Run this workflow: rd_run_workflow config_path=<this yaml> experiment_name=<exp>
# experiment_name=tac-rd-rank-ablate #
# The strategy loads the precomputed gate from signal_quality_gate_path.
# When hit rate >= threshold, trade; otherwise, go to cash.
# ----------------------------------------------------------------------------- # -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %} {%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %} {%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %} {%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
{%- set GATE_PATH = "/app/experiments/book/data/signal_quality_gate/sq_gate_5d_0.50.pkl" %}
qlib_init: qlib_init:
provider_uri: "{{ LAKE }}" provider_uri: "{{ LAKE }}"
@@ -40,27 +40,22 @@ qlib_init:
module_path: qlib.workflow.expm module_path: qlib.workflow.expm
kwargs: kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db" uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-rank-ablate" default_exp_name: "tac-rd-sq-gate"
task: task:
model: model:
class: RankICLGBModel class: LGBModel
module_path: tac_qlib.contrib.model.rank_gbdt module_path: qlib.contrib.model.gbdt
kwargs: kwargs:
loss: mse loss: mse
learning_rate: 0.02 learning_rate: 0.05
num_leaves: 31 num_leaves: 15
n_estimators: 3000 n_estimators: 200
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8 colsample_bytree: 0.8
subsample: 0.8 subsample: 0.8
subsample_freq: 1 subsample_freq: 1
reg_alpha: 0.1 reg_alpha: 0.01
reg_lambda: 1.0 reg_lambda: 0.01
seed: 42
dataset: dataset:
class: DatasetH class: DatasetH
@@ -110,12 +105,13 @@ task:
kwargs: kwargs:
config: config:
strategy: strategy:
class: TopkDropoutStrategy class: SignalQualityGateStrategy
module_path: qlib.contrib.strategy module_path: tac_qlib.contrib.strategy.signal_quality_gate
kwargs: kwargs:
signal: "<PRED>" signal: "<PRED>"
signal_quality_gate_path: "{{ GATE_PATH }}"
topk: 10 topk: 10
n_drop: 2 n_drop: 1
only_tradable: true only_tradable: true
risk_degree: 0.95 risk_degree: 0.95
backtest: backtest:
@@ -1,68 +1,44 @@
# ----------------------------------------------------------------------------- # Signal-quality gate backtest for 2021
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
#
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
#
# Run:
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
# experiment_name=tac-rd-risk-limit
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %} {%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %} {%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %} {%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
{# Gate computed on-the-fly from signal + lake close prices #}
qlib_init: qlib_init:
provider_uri: "{{ LAKE }}" provider_uri: "{{ LAKE }}"
region: us region: us
expression_cache: null expression_cache: null
dataset_cache: null dataset_cache: null
calendar_provider: calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US }
lake_root: "{{ LAKE }}"
market: US
instrument_provider: instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider: feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US }
lake_root: "{{ LAKE }}"
market: US
exp_manager: exp_manager:
class: MLflowExpManager class: MLflowExpManager
module_path: qlib.workflow.expm module_path: qlib.workflow.expm
kwargs: kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db" uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-risk-limit" default_exp_name: "tac-rd-sq-gate-2021"
task: task:
model: model:
class: RankICEnsembleLGBModel class: LGBModel
module_path: tac_qlib.contrib.model.rank_ensemble module_path: qlib.contrib.model.gbdt
kwargs: kwargs:
loss: mse loss: mse
learning_rate: 0.02 learning_rate: 0.05
num_leaves: 31 num_leaves: 15
n_estimators: 3000 n_estimators: 200
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8 colsample_bytree: 0.8
subsample: 0.8 subsample: 0.8
subsample_freq: 1 subsample_freq: 1
reg_alpha: 0.1 reg_alpha: 0.01
reg_lambda: 1.0 reg_lambda: 0.01
seeds: "42,7,2026,99,123"
parallel: 1
dataset: dataset:
class: DatasetH class: DatasetH
@@ -74,9 +50,9 @@ task:
kwargs: kwargs:
instruments: "{{ UNIVERSE }}" instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03 start_time: 2015-01-03
end_time: 2026-08-14 end_time: 2021-12-31
fit_start_time: 2016-01-04 fit_start_time: 2015-01-03
fit_end_time: 2025-09-01 fit_end_time: 2021-01-03
freq: day freq: day
lake_root: "{{ LAKE }}" lake_root: "{{ LAKE }}"
market: US market: US
@@ -94,9 +70,9 @@ task:
- class: Fillna - class: Fillna
kwargs: {} kwargs: {}
segments: segments:
train: [2016-01-04, 2025-09-01] train: [2015-01-03, 2020-09-01]
valid: [2025-09-03, 2026-01-03] valid: [2020-09-03, 2021-01-03]
test: [2026-01-04, 2026-08-10] test: [2021-01-04, 2021-12-31]
record: record:
- class: SignalRecord - class: SignalRecord
@@ -104,25 +80,29 @@ task:
kwargs: {} kwargs: {}
- class: SigAnaRecord - class: SigAnaRecord
module_path: qlib.workflow.record_temp module_path: qlib.workflow.record_temp
kwargs: kwargs: { ana_long_short: true, ann_scaler: 252 }
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord - class: PortAnaRecord
module_path: qlib.workflow.record_temp module_path: qlib.workflow.record_temp
kwargs: kwargs:
config: config:
strategy: strategy:
class: TopkDropoutStrategy class: SignalQualityGateStrategy
module_path: qlib.contrib.strategy module_path: tac_qlib.contrib.strategy.signal_quality_gate
kwargs: kwargs:
signal: "<PRED>" signal: "<PRED>"
lake_root: "{{ LAKE }}"
gate_topk: 10
gate_lookback: 5
gate_threshold: 0.5
gate_start: "2015-01-03"
gate_end: "2021-12-31"
topk: 10 topk: 10
n_drop: 2 n_drop: 1
only_tradable: true only_tradable: true
risk_degree: 0.95 risk_degree: 0.95
backtest: backtest:
start_time: 2026-01-04 start_time: 2021-01-04
end_time: 2026-08-10 end_time: 2021-12-31
account: 1000000 account: 1000000
benchmark: SPY benchmark: SPY
exchange_kwargs: exchange_kwargs:
+115
View File
@@ -0,0 +1,115 @@
# Signal-quality gate backtest for 2023
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
{# Gate computed on-the-fly from signal + lake close prices #}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-sq-gate-2023"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
kwargs:
loss: mse
learning_rate: 0.05
num_leaves: 15
n_estimators: 200
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.01
reg_lambda: 0.01
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2023-12-29
fit_start_time: 2015-01-03
fit_end_time: 2023-01-03
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
infer_processors:
- class: DropAllNaN
kwargs: {}
- class: ProcessInf
kwargs: {}
- class: CSRankNorm
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments:
train: [2015-01-03, 2022-09-01]
valid: [2022-09-03, 2023-01-03]
test: [2023-01-03, 2023-12-29]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs: { ana_long_short: true, ann_scaler: 252 }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: SignalQualityGateStrategy
module_path: tac_qlib.contrib.strategy.signal_quality_gate
kwargs:
signal: "<PRED>"
lake_root: "{{ LAKE }}"
gate_topk: 10
gate_lookback: 5
gate_threshold: 0.5
gate_start: "2015-01-03"
gate_end: "2023-12-29"
topk: 10
n_drop: 1
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2023-01-03
end_time: 2023-12-29
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
+115
View File
@@ -0,0 +1,115 @@
# Signal-quality gate backtest for 2024
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
{# Gate computed on-the-fly from signal + lake close prices #}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-sq-gate-2024"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
kwargs:
loss: mse
learning_rate: 0.05
num_leaves: 15
n_estimators: 200
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.01
reg_lambda: 0.01
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2024-12-31
fit_start_time: 2015-01-03
fit_end_time: 2024-01-03
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
infer_processors:
- class: DropAllNaN
kwargs: {}
- class: ProcessInf
kwargs: {}
- class: CSRankNorm
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments:
train: [2015-01-03, 2023-09-01]
valid: [2023-09-03, 2024-01-03]
test: [2024-01-02, 2024-12-31]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs: { ana_long_short: true, ann_scaler: 252 }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: SignalQualityGateStrategy
module_path: tac_qlib.contrib.strategy.signal_quality_gate
kwargs:
signal: "<PRED>"
lake_root: "{{ LAKE }}"
gate_topk: 10
gate_lookback: 5
gate_threshold: 0.5
gate_start: "2015-01-03"
gate_end: "2024-12-31"
topk: 10
n_drop: 1
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2024-01-02
end_time: 2024-12-31
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -1,68 +1,44 @@
# ----------------------------------------------------------------------------- # Signal-quality gate backtest for 2025
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
#
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
#
# Run:
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
# experiment_name=tac-rd-risk-limit
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %} {%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %} {%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %} {%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
{# Gate computed on-the-fly from signal + lake close prices #}
qlib_init: qlib_init:
provider_uri: "{{ LAKE }}" provider_uri: "{{ LAKE }}"
region: us region: us
expression_cache: null expression_cache: null
dataset_cache: null dataset_cache: null
calendar_provider: calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US }
lake_root: "{{ LAKE }}"
market: US
instrument_provider: instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider: feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US }
lake_root: "{{ LAKE }}"
market: US
exp_manager: exp_manager:
class: MLflowExpManager class: MLflowExpManager
module_path: qlib.workflow.expm module_path: qlib.workflow.expm
kwargs: kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db" uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-risk-limit" default_exp_name: "tac-rd-sq-gate-2025"
task: task:
model: model:
class: RankICEnsembleLGBModel class: LGBModel
module_path: tac_qlib.contrib.model.rank_ensemble module_path: qlib.contrib.model.gbdt
kwargs: kwargs:
loss: mse loss: mse
learning_rate: 0.02 learning_rate: 0.05
num_leaves: 31 num_leaves: 15
n_estimators: 3000 n_estimators: 200
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8 colsample_bytree: 0.8
subsample: 0.8 subsample: 0.8
subsample_freq: 1 subsample_freq: 1
reg_alpha: 0.1 reg_alpha: 0.01
reg_lambda: 1.0 reg_lambda: 0.01
seeds: "42,7"
parallel: 2
dataset: dataset:
class: DatasetH class: DatasetH
@@ -74,9 +50,9 @@ task:
kwargs: kwargs:
instruments: "{{ UNIVERSE }}" instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03 start_time: 2015-01-03
end_time: 2026-08-14 end_time: 2025-12-31
fit_start_time: 2016-01-04 fit_start_time: 2015-01-03
fit_end_time: 2025-09-01 fit_end_time: 2026-01-03
freq: day freq: day
lake_root: "{{ LAKE }}" lake_root: "{{ LAKE }}"
market: US market: US
@@ -94,9 +70,9 @@ task:
- class: Fillna - class: Fillna
kwargs: {} kwargs: {}
segments: segments:
train: [2016-01-04, 2025-09-01] train: [2015-01-03, 2025-09-01]
valid: [2025-09-03, 2026-01-03] valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10] test: [2025-01-02, 2025-12-31]
record: record:
- class: SignalRecord - class: SignalRecord
@@ -104,25 +80,29 @@ task:
kwargs: {} kwargs: {}
- class: SigAnaRecord - class: SigAnaRecord
module_path: qlib.workflow.record_temp module_path: qlib.workflow.record_temp
kwargs: kwargs: { ana_long_short: true, ann_scaler: 252 }
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord - class: PortAnaRecord
module_path: qlib.workflow.record_temp module_path: qlib.workflow.record_temp
kwargs: kwargs:
config: config:
strategy: strategy:
class: TopkDropoutStrategy class: SignalQualityGateStrategy
module_path: qlib.contrib.strategy module_path: tac_qlib.contrib.strategy.signal_quality_gate
kwargs: kwargs:
signal: "<PRED>" signal: "<PRED>"
lake_root: "{{ LAKE }}"
gate_topk: 10
gate_lookback: 5
gate_threshold: 0.5
gate_start: "2015-01-03"
gate_end: "2025-12-31"
topk: 10 topk: 10
n_drop: 2 n_drop: 1
only_tradable: true only_tradable: true
risk_degree: 0.95 risk_degree: 0.95
backtest: backtest:
start_time: 2026-01-04 start_time: 2025-01-02
end_time: 2026-08-10 end_time: 2025-12-31
account: 1000000 account: 1000000
benchmark: SPY benchmark: SPY
exchange_kwargs: exchange_kwargs:
@@ -1,67 +1,44 @@
# ----------------------------------------------------------------------------- # Signal-quality gate backtest for 2026
# ABLATION B (generic-only): same panel/model as the baseline, but feature
# fields restricted to the model-free / generic stochastic-process families
# (jump,har,trend,hurst,signature). Drops the model-specific ou (OU/AR-1
# half-life) and hmm (2-state regime) families to test whether the generic
# families alone dominate the rank dimension.
#
# Run:
# rd_run_workflow config_path=tac-qlib/workflows/ablate_generic_only_sp_fields.yaml \
# experiment_name=tac-rd-rank-ablate
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %} {%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %} {%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %} {%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
{# Gate computed on-the-fly from signal + lake close prices #}
qlib_init: qlib_init:
provider_uri: "{{ LAKE }}" provider_uri: "{{ LAKE }}"
region: us region: us
expression_cache: null expression_cache: null
dataset_cache: null dataset_cache: null
calendar_provider: calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US }
lake_root: "{{ LAKE }}"
market: US
instrument_provider: instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider: feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US }
lake_root: "{{ LAKE }}"
market: US
exp_manager: exp_manager:
class: MLflowExpManager class: MLflowExpManager
module_path: qlib.workflow.expm module_path: qlib.workflow.expm
kwargs: kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db" uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-rank-ablate" default_exp_name: "tac-rd-sq-gate-2026"
task: task:
model: model:
class: RankICLGBModel class: LGBModel
module_path: tac_qlib.contrib.model.rank_gbdt module_path: qlib.contrib.model.gbdt
kwargs: kwargs:
loss: mse loss: mse
learning_rate: 0.02 learning_rate: 0.05
num_leaves: 31 num_leaves: 15
n_estimators: 3000 n_estimators: 200
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8 colsample_bytree: 0.8
subsample: 0.8 subsample: 0.8
subsample_freq: 1 subsample_freq: 1
reg_alpha: 0.1 reg_alpha: 0.01
reg_lambda: 1.0 reg_lambda: 0.01
seed: 42
dataset: dataset:
class: DatasetH class: DatasetH
@@ -75,7 +52,7 @@ task:
start_time: 2015-01-03 start_time: 2015-01-03
end_time: 2026-08-10 end_time: 2026-08-10
fit_start_time: 2015-01-03 fit_start_time: 2015-01-03
fit_end_time: 2025-09-01 fit_end_time: 2026-01-03
freq: day freq: day
lake_root: "{{ LAKE }}" lake_root: "{{ LAKE }}"
market: US market: US
@@ -103,20 +80,24 @@ task:
kwargs: {} kwargs: {}
- class: SigAnaRecord - class: SigAnaRecord
module_path: qlib.workflow.record_temp module_path: qlib.workflow.record_temp
kwargs: kwargs: { ana_long_short: true, ann_scaler: 252 }
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord - class: PortAnaRecord
module_path: qlib.workflow.record_temp module_path: qlib.workflow.record_temp
kwargs: kwargs:
config: config:
strategy: strategy:
class: TopkDropoutStrategy class: SignalQualityGateStrategy
module_path: qlib.contrib.strategy module_path: tac_qlib.contrib.strategy.signal_quality_gate
kwargs: kwargs:
signal: "<PRED>" signal: "<PRED>"
lake_root: "{{ LAKE }}"
gate_topk: 10
gate_lookback: 5
gate_threshold: 0.5
gate_start: "2015-01-03"
gate_end: "2026-08-19"
topk: 10 topk: 10
n_drop: 2 n_drop: 1
only_tradable: true only_tradable: true
risk_degree: 0.95 risk_degree: 0.95
backtest: backtest:
+26 -20
View File
@@ -1,29 +1,35 @@
# TradeAC custom-qlib-code snapshot (auto-generated) # TradeAC custom-qlib-code snapshot (auto-generated)
# parent repo HEAD : 125be7b96fb5975e798a0b4301eeb5809a8a181c # parent repo HEAD : e952feed0a66a20439f4f24ad5524233429cd0c3
# tac-qlib/tac_qlib/contrib # tac-qlib/tac_qlib/contrib
# tac-qlib/tac_qlib/data # tac-qlib/tac_qlib/data
# per-file hashes (git hash-object): # per-file hashes (git hash-object):
1b6298c4a5652f2e863cbdc385a1014a570fcd59 tac-qlib/tac_qlib/contrib/__init__.py 1b6298c4a5652f2e863cbdc385a1014a570fcd59 tac-qlib/tac_qlib/contrib/__init__.py
b8112569f9b2537c45b6535e1a505a207878d322 tac-qlib/tac_qlib/contrib/__pycache__/__init__.cpython-312.pyc 861592c63edd6a0853a9cb174b5970435b135fc8 tac-qlib/tac_qlib/contrib/__pycache__/__init__.cpython-312.pyc
c76a9f17f680e74eea766eff27f7624359749ed6 tac-qlib/tac_qlib/contrib/data/__init__.py c76a9f17f680e74eea766eff27f7624359749ed6 tac-qlib/tac_qlib/contrib/data/__init__.py
8d5333ebd2b44165c50cba639ca2d4ac3fc7cfec tac-qlib/tac_qlib/contrib/data/__pycache__/__init__.cpython-312.pyc 5c547a2ef92e075e550fe6d01508a2f1d3f536bc tac-qlib/tac_qlib/contrib/data/__pycache__/__init__.cpython-312.pyc
18cb37c0354184c49fa2e598396d7df0634cce0f tac-qlib/tac_qlib/contrib/data/__pycache__/handler.cpython-312.pyc 4f656130d167e79dcaaeb7783a121f0b36852374 tac-qlib/tac_qlib/contrib/data/__pycache__/handler.cpython-312.pyc
871ff1e163c29261f140c3f53d42a41e6504c779 tac-qlib/tac_qlib/contrib/data/handler.py 0dd25ef161c6e0f15eafc84886e7e1381deb38c3 tac-qlib/tac_qlib/contrib/data/handler.py
b151d139a0dcde87d74b21e7c4b729176ba5c39b tac-qlib/tac_qlib/contrib/model/__init__.py b151d139a0dcde87d74b21e7c4b729176ba5c39b tac-qlib/tac_qlib/contrib/model/__init__.py
ab958203f33a99d12c7d923b6efb435189231666 tac-qlib/tac_qlib/contrib/model/__pycache__/__init__.cpython-312.pyc b1489f2fc0dee85f0a4f90b2e6ad545ed9c8967b tac-qlib/tac_qlib/contrib/model/__pycache__/__init__.cpython-312.pyc
7478f6b0f6de419615c02d4d92b54529f689ef04 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_ensemble.cpython-312.pyc 121ef237da1df1b8e21a561c3ad0db200b901339 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_ensemble.cpython-312.pyc
9f9014ddd9bce37490061312d51e8e6fe540fec4 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_gbdt.cpython-312.pyc 74d0da348cbcc3700c96b6f4fe4391488e61efc5 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_gbdt.cpython-312.pyc
ce77dea53f6a87c5379782709293bf8ff55b2c75 tac-qlib/tac_qlib/contrib/model/rank_ensemble.py d3f051f3a8650c42fedc7b367b966f7c74fb5789 tac-qlib/tac_qlib/contrib/model/rank_ensemble.py
ccfe7d554989aa7f3e5a2128ae663e51b2207149 tac-qlib/tac_qlib/contrib/model/rank_gbdt.py d03e6611338918d4aac5eea4adf26f85a3763652 tac-qlib/tac_qlib/contrib/model/rank_gbdt.py
4afcf9058231111c412925f4c4b84e81d656db87 tac-qlib/tac_qlib/contrib/strategy/__init__.py 2c2f167b693f4366a769998e3c9d4804f29e31e0 tac-qlib/tac_qlib/contrib/strategy/__init__.py
74e5ecbbbb20bb71fd5cd083383de4ce88476712 tac-qlib/tac_qlib/contrib/strategy/__pycache__/__init__.cpython-312.pyc f9cd9ab729e3248542ccc490af7adc2e51c71914 tac-qlib/tac_qlib/contrib/strategy/__pycache__/__init__.cpython-312.pyc
afaf562aeaa12cebc8529cd916153252e7e3c38a tac-qlib/tac_qlib/contrib/strategy/__pycache__/optimal_stop.cpython-312.pyc cd3133cfbd2556b25c106e39ae97fb128df0e326 tac-qlib/tac_qlib/contrib/strategy/__pycache__/ic_gate.cpython-312.pyc
96a0a25201f0a1bb2fc2190e26228c5c0e711a79 tac-qlib/tac_qlib/contrib/strategy/hmm_risk.py 6dd1c568a2961842793674390d5abffd1a0e71b8 tac-qlib/tac_qlib/contrib/strategy/__pycache__/optimal_stop.cpython-312.pyc
816de5d58ae23d996635d42331cf9fc8963d5dbe tac-qlib/tac_qlib/contrib/strategy/momentum_gate.py 03b5e4d80da00800f1b108bee0735d3d18d856d1 tac-qlib/tac_qlib/contrib/strategy/__pycache__/regime_gate.cpython-312.pyc
a6a1c21ab71b62080830df45c4784b73c1531036 tac-qlib/tac_qlib/contrib/strategy/__pycache__/signal_quality_gate.cpython-312.pyc
755e3b139496a5e22b0328db45c8199c33029fbc tac-qlib/tac_qlib/contrib/strategy/__pycache__/weekly_rebalance.cpython-312.pyc
519a1f4c05dbe0ac018ab8b779eb33d53b4dd545 tac-qlib/tac_qlib/contrib/strategy/ic_gate.py
79aaad9e39fcc740a773f4f63c512ce1086cfde0 tac-qlib/tac_qlib/contrib/strategy/optimal_stop.py 79aaad9e39fcc740a773f4f63c512ce1086cfde0 tac-qlib/tac_qlib/contrib/strategy/optimal_stop.py
7bcee5f0b09cfa721440f1354f16f2dd9a112b12 tac-qlib/tac_qlib/contrib/strategy/regime_gate.py
16ab80b731aab6d8bb818525615d77c2d1fcb0c8 tac-qlib/tac_qlib/contrib/strategy/signal_quality_gate.py
fe60bacdfedd48617863be31f24b7c7daebfac5a tac-qlib/tac_qlib/contrib/strategy/weekly_rebalance.py
92e6e90eb0cd0a25142034560f27adb6b705b1a8 tac-qlib/tac_qlib/data/__init__.py 92e6e90eb0cd0a25142034560f27adb6b705b1a8 tac-qlib/tac_qlib/data/__init__.py
0ed1ead6c1314a3f25784d453e54a15a8a04baaa tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc 316bf4aa160cc8d15929ea648be03f4b4999667d tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc
9609782800944c45b78bb58eaa7b51ba1b7f8f43 tac-qlib/tac_qlib/data/__pycache__/config.cpython-312.pyc 554a3f29d181b64effbf49a8161b32e7f93d8d3e tac-qlib/tac_qlib/data/__pycache__/config.cpython-312.pyc
a85628d71d12cfe5b18b1c884c5d829c89594579 tac-qlib/tac_qlib/data/__pycache__/providers.cpython-312.pyc 8b47f6d78ac046b6b7b2fb07bd7f3382773ffb73 tac-qlib/tac_qlib/data/__pycache__/providers.cpython-312.pyc
686d36f6d101c547491ca866aa143aa542e17518 tac-qlib/tac_qlib/data/config.py 53c9007a928841fd3c3b08450f9a6520ce1ac091 tac-qlib/tac_qlib/data/config.py
d9f839be30026f337754a3f015425a8efdbe8e2a tac-qlib/tac_qlib/data/providers.py 8d0644f6f0d1efb94798ed444cc73e63b643459b tac-qlib/tac_qlib/data/providers.py
+22 -4
View File
@@ -64,9 +64,13 @@ def check_transform_proc(proc_l, fit_start_time, fit_end_time):
def get_common_feature_fields(lake_root=None, market="US", timeframe="1d") -> List[str]: def get_common_feature_fields(lake_root=None, market="US", timeframe="1d") -> List[str]:
"""Discover ta-lib columns present in *every* features parquet file of the lake. """Discover feature columns present in *every* feature file of the lake.
Returns sorted field names (without the ``$`` prefix). Empty if no features are persisted. Walks the `family=ta|sp` partition layout (plus any legacy flat files).
TA and SP columns are disjoint by construction, so the common set is
computed per family (columns shared by all symbol files of that family),
then the per-family results are unioned. Returns sorted field names
(without the ``$`` prefix). Empty if no features are persisted.
""" """
cfg = LakeConfig(lake_root, market) cfg = LakeConfig(lake_root, market)
feat_dir = cfg.features_dir(timeframe) feat_dir = cfg.features_dir(timeframe)
@@ -74,8 +78,9 @@ def get_common_feature_fields(lake_root=None, market="US", timeframe="1d") -> Li
return [] return []
import pyarrow.parquet as pq import pyarrow.parquet as pq
def _family_common(fam_dir: Path) -> set:
common = None common = None
for p in sorted(feat_dir.glob("symbol=*.parquet")): for p in sorted(fam_dir.glob("symbol=*.parquet")):
try: try:
cols = set(pq.read_schema(p).names) - set(NON_FEATURE_COLUMNS) cols = set(pq.read_schema(p).names) - set(NON_FEATURE_COLUMNS)
except Exception: # pragma: no cover - skip unreadable files except Exception: # pragma: no cover - skip unreadable files
@@ -83,7 +88,20 @@ def get_common_feature_fields(lake_root=None, market="US", timeframe="1d") -> Li
common = cols if common is None else (common & cols) common = cols if common is None else (common & cols)
if not common: if not common:
break break
return sorted(common) if common else [] return common or set()
common: set = set()
# family tier: features/market=*/timeframe=*/family=*/symbol=*.parquet
for fam in ("ta", "sp"):
fam_dir = feat_dir / f"family={fam}"
if fam_dir.is_dir():
common |= _family_common(fam_dir)
# legacy flat: features/market=*/timeframe=*/symbol=*.parquet
if (feat_dir / "family=ta").exists() or (feat_dir / "family=sp").exists():
pass # family layout already covered
else:
common |= _family_common(feat_dir)
return sorted(common)
class DropAllNaN(processor_module.Processor): class DropAllNaN(processor_module.Processor):
@@ -56,7 +56,6 @@ import os
from concurrent.futures import ThreadPoolExecutor from concurrent.futures import ThreadPoolExecutor
from typing import List, Optional from typing import List, Optional
import numpy as np
import pandas as pd import pandas as pd
from qlib.data.dataset import DatasetH from qlib.data.dataset import DatasetH
@@ -80,15 +79,11 @@ class RankICEnsembleLGBModel(RankICLGBModel):
forwarded. forwarded.
""" """
def __init__(self, seeds: str = "42", parallel: int = 0, weight_mode: str = "equal", **kwargs): def __init__(self, seeds: str = "42", parallel: int = 0, **kwargs):
self.seeds = [int(s.strip()) for s in str(seeds).split(",") if s.strip()] self.seeds = [int(s.strip()) for s in str(seeds).split(",") if s.strip()]
if not self.seeds: if not self.seeds:
raise ValueError("seeds must contain at least one integer") raise ValueError("seeds must contain at least one integer")
self.parallel = int(parallel) self.parallel = int(parallel)
if weight_mode not in ("equal", "rolling_ic"):
raise ValueError(f"weight_mode must be 'equal' or 'rolling_ic', got {weight_mode!r}")
self.weight_mode = weight_mode
self.rolling_ic_window = int(kwargs.pop("rolling_ic_window", 21))
# drop seed/parallel handling from the base kwargs, keep everything else # drop seed/parallel handling from the base kwargs, keep everything else
self._model_kwargs = dict(kwargs) self._model_kwargs = dict(kwargs)
super().__init__(**self._model_kwargs) super().__init__(**self._model_kwargs)
@@ -184,44 +179,11 @@ class RankICEnsembleLGBModel(RankICLGBModel):
# -------------------------------------------------------------- predict # -------------------------------------------------------------- predict
def predict(self, dataset: DatasetH, segment="test") -> pd.Series: def predict(self, dataset: DatasetH, segment="test") -> pd.Series:
"""Combine per-seed predictions. """Average the per-seed predictions over the given segment."""
``weight_mode='equal'`` (default): simple average, as before.
``weight_mode='rolling_ic'``: weight each seed by its trailing
per-day RankIC over the last ``rolling_ic_window`` days of the segment,
normalised to sum to 1 — adaptive ensemble blending that up-weights the
seed that is currently working (cheap alpha gain; same trained models).
"""
if not self._models: if not self._models:
raise ValueError("model is not fitted yet!") raise ValueError("model is not fitted yet!")
preds = [m.predict(dataset, segment=segment) for m in self._models] preds = [m.predict(dataset, segment=segment) for m in self._models]
if len(preds) == 1: if len(preds) == 1:
return preds[0] return preds[0]
frame = pd.concat(preds, axis=1) frame = pd.concat(preds, axis=1)
frame.columns = [f"seed{m.params.get('seed', i)}" for i, m in enumerate(self._models)]
if self.weight_mode == "equal":
return frame.mean(axis=1) return frame.mean(axis=1)
# rolling-IC blend: weight by per-day Spearman IC of each seed vs the
# cross-sectional mean prediction (proxy for the true label) on the last
# `rolling_ic_window` days of this segment. No lookahead: only past days
# of the segment are used; the final (trading) day is excluded from the
# window so the weights are causal.
mean_pred = frame.mean(axis=1)
dates = sorted(frame.index.get_level_values(0).unique())
win = [d for d in dates if d < dates[-1]][-self.rolling_ic_window :]
ics = {}
for col in frame.columns:
if not win:
ics[col] = 1.0
continue
sub = pd.DataFrame({"p": frame[col], "m": mean_pred})
vals = []
for d in win:
s = sub[sub.index.get_level_values(0) == d]
if len(s) >= 3 and s["p"].nunique() > 1 and s["m"].nunique() > 1:
vals.append(s["p"].rank().corr(s["m"].rank()))
ics[col] = float(np.mean(vals)) if vals else 1.0
wsum = sum(ics.values()) or len(ics)
weights = {c: v / wsum for c, v in ics.items()}
return sum(frame[c] * weights[c] for c in frame.columns)
@@ -53,23 +53,61 @@ from qlib.workflow import R
__all__ = ["RankICLGBModel", "rankic_feval"] __all__ = ["RankICLGBModel", "rankic_feval"]
def _group_averaged_rank(values: np.ndarray, gid: np.ndarray, offs: np.ndarray) -> np.ndarray:
"""Averaged (tie-corrected) rank of ``values`` within each group, vectorized.
``gid`` maps each row to its group id; ``offs`` holds the cumulative row
offsets so that group ``i`` occupies rows ``[offs[i], offs[i+1])``. Returns
the same result as ``pandas.Series.rank(method='average')`` applied per
group, but in one pass (``np.lexsort`` is the only non-linear step).
"""
n = len(values)
order = np.lexsort((values, gid))
ord_rank = np.empty(n, dtype=np.float64)
ord_rank[order] = np.arange(n, dtype=np.float64) - offs[gid[order]] + 1.0
sg = gid[order]
sv = values[order]
newblock = np.empty(n, dtype=bool)
newblock[0] = True
newblock[1:] = (sg[1:] != sg[:-1]) | (sv[1:] != sv[:-1])
blockid = np.cumsum(newblock) - 1
block_mean = np.bincount(blockid, weights=ord_rank[order]) / np.bincount(blockid)
out = np.empty(n)
out[order] = block_mean[blockid]
return out
def _per_day_spearman(preds: np.ndarray, labels: np.ndarray, group: np.ndarray) -> float: def _per_day_spearman(preds: np.ndarray, labels: np.ndarray, group: np.ndarray) -> float:
"""Mean per-day Spearman rank correlation of preds vs labels. """Mean per-day Spearman rank correlation of preds vs labels.
``group`` holds the number of rows of each trading day (query group), in ``group`` holds the number of rows of each trading day (query group), in
order. Days with <3 valid rows or a constant pred/label are skipped. order. Days with <3 valid rows or a constant pred/label are skipped.
Vectorized: per-day Spearman == Pearson of the per-day rank transforms,
and the Pearson moments (``sum``, ``sum`` of products/squares) aggregate
over each day with ``np.bincount``. Runs ~10x faster than the per-day
``pd.Series.rank()`` loop that preceded it — this feval is invoked on the
train and valid panels every boosting round, per seed.
""" """
if group is None or len(group) == 0: if group is None or len(group) == 0:
return 0.0 return 0.0
offs = np.concatenate([[0], np.cumsum(group.astype(int))]) offs = np.concatenate([[0], np.cumsum(group.astype(int))])
vals = [] gid = np.repeat(np.arange(len(group)), group.astype(int))
for i in range(len(group)): rp = _group_averaged_rank(preds, gid, offs)
s = slice(offs[i], offs[i + 1]) rl = _group_averaged_rank(labels, gid, offs)
p, l = preds[s], labels[s] n_g = group.astype(float)
if len(p) < 3 or np.std(p) == 0 or np.std(l) == 0: s_p = np.bincount(gid, weights=rp)
continue s_l = np.bincount(gid, weights=rl)
vals.append(np.corrcoef(pd.Series(p).rank(), pd.Series(l).rank())[0, 1]) s_pl = np.bincount(gid, weights=rp * rl)
return float(np.mean(vals)) if vals else 0.0 s_pp = np.bincount(gid, weights=rp * rp)
s_ll = np.bincount(gid, weights=rl * rl)
cov = n_g * s_pl - s_p * s_l
var_p = n_g * s_pp - s_p ** 2
var_l = n_g * s_ll - s_l ** 2
denom = np.sqrt(var_p * var_l)
valid = (n_g >= 3) & (denom > 0)
corr = np.where(valid, cov / np.where(denom == 0, 1, denom), 0.0)
return float(corr[valid].mean()) if valid.any() else 0.0
def rankic_feval(preds, dataset): def rankic_feval(preds, dataset):
@@ -1,3 +1,11 @@
from .ic_gate import ICGateTopkDropoutStrategy # noqa: F401
from .optimal_stop import OptimalStopControl # noqa: F401 from .optimal_stop import OptimalStopControl # noqa: F401
from .regime_gate import RegimeGateTopkDropoutStrategy # noqa: F401
from .weekly_rebalance import WeeklyRebalanceDropoutStrategy # noqa: F401
__all__ = ["OptimalStopControl"] __all__ = [
"ICGateTopkDropoutStrategy",
"OptimalStopControl",
"RegimeGateTopkDropoutStrategy",
"WeeklyRebalanceDropoutStrategy",
]
@@ -1,138 +0,0 @@
"""TopkDropout with HMM high-volatility + drawdown-pause risk gates.
Gates NEW entries on two risk conditions (held names are never force-sold):
1. **HMM high-vol pause**: when the cross-sectional mean of ``sp_hmm_p_regime1``
(HMM high-vol regime probability) on the signal date is >= ``hmm_pause_pct``,
new buys are paused. The time-series study showed HMM high-vol probability
pulses BEFORE sharp moves (regime-change cut) — pausing new exposure at the
boundary reduces drawdown from price over-reaction.
2. **Drawdown pause**: when the account equity drawdown from its running peak
exceeds ``drawdown_pause_pct``, new buys are paused. This is the
``drawdown_pause_pct`` risk-limit expressed inside the backtest (the pure
executor-side gate is documented as not expressible in a one-shot backtest).
3. **Liquidity floor**: names whose 20-day average daily dollar volume is below
``liquidity_floor_adv`` are dropped from BUY candidates (the proven mitigant
from exp-18: $5M floor cut drawdown 7.9%->5.4% at higher IR).
Implementation: pre-filter the signal score before the base TopkDropout
decision — non-held names get score 0 when any gate fires.
Wired into a workflow yaml like:
strategy:
class: HmmRiskTopk
module_path: tac_qlib.contrib.strategy.hmm_risk
kwargs:
signal: "<PRED>"
topk: 10
n_drop: 2
only_tradable: true
risk_degree: 0.95
hmm_pause_pct: 0.70
drawdown_pause_pct: 8.0
liquidity_floor_adv: 5000000
"""
from __future__ import annotations
import copy
from typing import Dict
import numpy as np
import pandas as pd
from qlib.backtest.decision import TradeDecisionWO
from qlib.backtest.position import Position
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
__all__ = ["HmmRiskTopk"]
class HmmRiskTopk(TopkDropoutStrategy):
"""TopkDropoutStrategy with HMM high-vol pause + drawdown pause + liquidity floor."""
def __init__(
self,
*,
hmm_pause_pct: float = 0.70,
drawdown_pause_pct: float = 8.0,
liquidity_floor_adv: float = 0.0,
**kwargs,
):
super().__init__(**kwargs)
self.hmm_pause_pct = float(hmm_pause_pct)
self.drawdown_pause_pct = float(drawdown_pause_pct)
self.liquidity_floor_adv = float(liquidity_floor_adv)
self._peak_equity = 0.0
# ------------------------------------------------------------- gates
def _hmm_high_vol(self, pred_date) -> bool:
"""Cross-sectional mean HMM high-vol regime probability >= threshold."""
try:
from qlib.data import D
feat = D.features(D.instruments("all"), ["$sp_hmm_p_regime1"],
start_time=pred_date, end_time=pred_date)
if feat is None or len(feat) == 0:
return False
p = feat["$sp_hmm_p_regime1"].dropna()
if len(p) == 0:
return False
return float(p.mean()) >= self.hmm_pause_pct
except Exception:
return False
def _drawdown_active(self, equity: float) -> bool:
if self.drawdown_pause_pct <= 0:
return False
self._peak_equity = max(self._peak_equity, equity)
if self._peak_equity <= 0:
return False
dd = (self._peak_equity - equity) / self._peak_equity * 100.0
return dd >= self.drawdown_pause_pct
def _illiquid(self, codes, asof) -> Dict[str, bool]:
if self.liquidity_floor_adv <= 0 or not codes:
return {}
from tac_qlib.risk_limits import dollar_adv
adv = dollar_adv(codes, market="US", asof=asof, lookback=20)
return {c: adv.get(str(c).upper(), 0.0) < self.liquidity_floor_adv for c in codes}
# ------------------------------------------------------------- decision
def generate_trade_decision(self, execute_result=None):
trade_step = self.trade_calendar.get_trade_step()
trade_start_time, trade_end_time = self.trade_calendar.get_step_time(trade_step)
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(trade_step, shift=1)
pred_score = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
if pred_score is None:
return TradeDecisionWO([], self)
if isinstance(pred_score, pd.DataFrame):
pred_score = pred_score.iloc[:, 0]
current_temp = copy.deepcopy(self.trade_position)
assert isinstance(current_temp, Position)
held = {c for c in current_temp.get_stock_list() if abs(current_temp.get_stock_amount(c)) > 1e-6}
equity = current_temp.get_cash()
for code in held:
mark = self.trade_exchange.get_deal_price(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time, direction=1
)
if mark is not None and np.isfinite(mark):
equity += abs(current_temp.get_stock_amount(code)) * mark
hmm_pause = self._hmm_high_vol(str(pd.Timestamp(pred_start_time).date()))
dd_pause = self._drawdown_active(equity)
buys_paused = hmm_pause or dd_pause
pred_score = pred_score.copy()
if buys_paused or self.liquidity_floor_adv > 0:
new_codes = [c for c in pred_score.index if c not in held]
illiquid = self._illiquid(new_codes, str(pd.Timestamp(pred_start_time).date()))
for code in new_codes:
if buys_paused or illiquid.get(code, False):
pred_score[code] = -1e9 # cannot enter today
return super().generate_trade_decision(execute_result)
@@ -0,0 +1,117 @@
"""Realized-IC circuit breaker TopkDropout strategy.
Subclass of ``qlib.contrib.strategy.signal_strategy.TopkDropoutStrategy`` that
holds the book (issues NO orders) while the streaming realized RankIC of the
deployed signal is below threshold — i.e. the model's cross-sectional
predictions are no longer earning against realized forward returns. When the
gate is open it behaves exactly like the reference TopkDropoutStrategy.
The gate is evaluated per trade step on the trailing mean realized RankIC of
the signal over the last ``ic_window`` trading days whose label is fully
realized as of the decision date (no lookahead — a 5d fwd label ``close[t+6]/
close[t+1]-1`` is only known at ``t+6``).
Two wiring modes:
* ``ic_gate``: a precomputed ``pd.Series`` indexed by datetime of booleans
(True = gate open / trade allowed). Computed once by the caller (e.g.
``rd_backtest``) and looked up per step. Missing dates default to open.
* realized-IC self-computation: when ``ic_min_rankic`` is given but no
``ic_gate``, the strategy computes the per-date realized RankIC itself from
``self.signal`` (the pred scores) and the lake 1d bars via
``tac_qlib.risk_limits.realized_rankic_series``, then applies the same
trailing-window comparison. Works when instantiated from a workflow YAML
PortAnaRecord config (``lake_root`` / ``market`` must be provided).
"""
from __future__ import annotations
import pandas as pd
from qlib.backtest.decision import TradeDecisionWO
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
from tac_qlib.risk_limits import ic_circuit_breaker, realized_rankic_series
__all__ = ["ICGateTopkDropoutStrategy"]
class ICGateTopkDropoutStrategy(TopkDropoutStrategy):
"""TopkDropout with a streaming realized-IC circuit breaker.
Parameters
----------
topk, n_drop, method_sell, method_buy, hold_thresh, only_tradable,
forbid_all_trade_at_limit : same as ``TopkDropoutStrategy``.
ic_min_rankic : float — pause new trading while trailing realized RankIC is
below this threshold (0 disables the gate).
ic_window : int — trailing window for the realized RankIC mean (default 22).
ic_label_horizon : int — label horizon in trading days (default 6).
ic_min_obs : int — min realized labels before the gate arms (default 10).
ic_gate : pd.Series, optional — precomputed per-date gate (bool indexed by
datetime). When provided, it overrides self-computation.
lake_root, market : str — lake location for self-computed realized IC.
"""
def __init__(
self,
*,
topk,
n_drop,
ic_min_rankic: float = 0.0,
ic_window: int = 22,
ic_label_horizon: int = 6,
ic_min_obs: int = 10,
ic_gate=None,
lake_root: str = "",
market: str = "US",
**kwargs,
):
super().__init__(topk=topk, n_drop=n_drop, **kwargs)
self.ic_min_rankic = float(ic_min_rankic or 0.0)
self.ic_window = int(ic_window or 22)
self.ic_label_horizon = int(ic_label_horizon or 6)
self.ic_min_obs = int(ic_min_obs or 10)
self._ic_gate = ic_gate
self._realized_ic = None
self.lake_root = lake_root or ""
self.market = market or "US"
def _load_realized_ic(self):
if self._realized_ic is None:
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(
self.trade_calendar.get_trade_step(), shift=-self.ic_label_horizon
)
pred = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
if isinstance(pred, pd.DataFrame):
pred = pred.iloc[:, 0]
self._realized_ic = realized_rankic_series(
pred, self.lake_root, self.market, label_horizon=self.ic_label_horizon
)
return self._realized_ic
def _gate_open(self, trade_start_time) -> bool:
ts = pd.Timestamp(trade_start_time)
if self._ic_gate is not None:
# precomputed gate series: look up the latest known decision date <= ts
known = self._ic_gate[self._ic_gate.index <= ts]
if len(known):
return bool(known.iloc[-1])
return True
if self.ic_min_rankic <= 0:
return True
realized = self._load_realized_ic()
limits = {
"ic_min_rankic": self.ic_min_rankic,
"ic_window": self.ic_window,
"ic_min_obs": self.ic_min_obs,
}
tripped, _reason, _trail = ic_circuit_breaker(realized, ts, limits)
return not tripped
def generate_trade_decision(self, execute_result=None):
trade_step = self.trade_calendar.get_trade_step()
trade_start_time, _ = self.trade_calendar.get_step_time(trade_step)
if not self._gate_open(trade_start_time):
return TradeDecisionWO([], self)
return super().generate_trade_decision(execute_result)
@@ -1,91 +0,0 @@
"""TopkDropout with a 1-day momentum entry-confirmation gate.
Gates NEW entries on short-term momentum: a name that is not currently held
may only be bought when its trailing 1-day return is above ``min_momentum``
(Lag-1 autocorr ~ +0.45 in the time-series study => short-term momentum
continuation). Held names are never force-sold by this gate — exits stay the
pure TopkDropout rule.
Implementation: override ``generate_trade_decision`` and zero out the signal
score of any non-held name that fails the momentum check BEFORE calling the
base TopkDropout decision, so it can never be selected as a buy candidate.
This is a clean pre-filter: the rest of the strategy (top-k, n_drop, sizing,
costs) is untouched.
Wired into a workflow yaml like:
strategy:
class: MomentumGateTopk
module_path: tac_qlib.contrib.strategy.momentum_gate
kwargs:
signal: "<PRED>"
topk: 10
n_drop: 2
only_tradable: true
risk_degree: 0.95
min_momentum: 0.0
"""
from __future__ import annotations
import copy
import pandas as pd
from qlib.backtest.decision import TradeDecisionWO
from qlib.backtest.position import Position
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
__all__ = ["MomentumGateTopk"]
class MomentumGateTopk(TopkDropoutStrategy):
"""TopkDropoutStrategy gated on 1-day momentum for new entries."""
def __init__(self, *, min_momentum: float = 0.0, **kwargs):
super().__init__(**kwargs)
self.min_momentum = float(min_momentum)
def _momentum_ok(self, code, trade_start, trade_end) -> bool:
"""True when the trailing 1-day return is above the momentum floor."""
try:
cur = self.trade_exchange.get_deal_price(
stock_id=code, start_time=trade_start, end_time=trade_end, direction=1
)
except Exception:
return False
if cur is None or cur != cur or cur <= 0:
return False
prev_start = trade_start - pd.Timedelta(days=5)
prev_end = trade_start - pd.Timedelta(seconds=1)
prev = self.trade_exchange.get_deal_price(
stock_id=code, start_time=prev_start, end_time=prev_end, direction=0
)
if prev is None or prev != prev or prev <= 0:
return False
return (cur / prev - 1.0) >= self.min_momentum
def generate_trade_decision(self, execute_result=None):
trade_step = self.trade_calendar.get_trade_step()
trade_start_time, trade_end_time = self.trade_calendar.get_step_time(trade_step)
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(trade_step, shift=1)
pred_score = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
if pred_score is None:
return TradeDecisionWO([], self)
if isinstance(pred_score, pd.DataFrame):
pred_score = pred_score.iloc[:, 0]
current_temp = copy.deepcopy(self.trade_position)
assert isinstance(current_temp, Position)
held = set(current_temp.get_stock_list())
held = {c for c in held if abs(current_temp.get_stock_amount(c)) > 1e-6}
# pre-filter: zero the score of non-held names that fail momentum
pred_score = pred_score.copy()
for code in pred_score.index:
if code in held:
continue # never gate exits / re-balancing of held names
if not self._momentum_ok(code, trade_start_time, trade_end_time):
pred_score[code] = -1e9 # cannot enter today
return super().generate_trade_decision(execute_result)
@@ -0,0 +1,215 @@
"""Regime-gate TopkDropout strategy.
Subclass of ``qlib.contrib.strategy.signal_strategy.TopkDropoutStrategy`` that
holds the book (issues NO orders) while a regime detector says the market is in
an unfavorable state. When the gate is open it behaves exactly like the
reference TopkDropoutStrategy.
Three detector types are supported (all causal — no lookahead):
* ``dispersion``: cross-sectional standard deviation of 22-day rolling returns
across the universe. Gate closes when CS dispersion < threshold (low
dispersion means the spread between winners and losers is too narrow for
TopkDropout to exploit).
* ``vol``: cross-sectional mean of 22-day rolling realized volatility. Gate
closes when avg vol is outside a band ``[vol_low, vol_high]`` (strategy
needs moderate vol — too calm or too turbulent both hurt).
* ``hmm``: pre-computed HMM posterior for regime 1 (``sp_hmm_p_regime1``).
Gate closes when posterior < threshold (model is not confident the calm
regime is active).
The gate is provided as a precomputed ``pd.Series`` of booleans indexed by
datetime (True = trade allowed). The companion ``compute_regime_gate``
function builds this series from lake bars; call it once before backtesting
and pass the result as the ``regime_gate`` parameter.
"""
from __future__ import annotations
import pandas as pd
from qlib.backtest.decision import TradeDecisionWO
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
__all__ = ["RegimeGateTopkDropoutStrategy", "compute_regime_gate"]
class RegimeGateTopkDropoutStrategy(TopkDropoutStrategy):
"""TopkDropout with a regime-gate circuit breaker.
Parameters
----------
topk, n_drop, method_sell, method_buy, hold_thresh, only_tradable,
forbid_all_trade_at_limit : same as ``TopkDropoutStrategy``.
regime_gate : pd.Series — precomputed per-date gate (bool indexed by
datetime). True = trade allowed, False = no orders. Missing dates
default to open (trade allowed).
"""
def __init__(self, *, regime_gate=None, **kwargs):
super().__init__(**kwargs)
self._regime_gate = regime_gate
def _gate_open(self, trade_start_time) -> bool:
if self._regime_gate is None:
return True
ts = pd.Timestamp(trade_start_time)
known = self._regime_gate[self._regime_gate.index <= ts]
if len(known):
return bool(known.iloc[-1])
return True # default open if no history yet
def generate_trade_decision(self, execute_result=None):
trade_step = self.trade_calendar.get_trade_step()
trade_start_time, _ = self.trade_calendar.get_step_time(trade_step)
if not self._gate_open(trade_start_time):
return TradeDecisionWO([], self)
return super().generate_trade_decision(execute_result)
# ---------------------------------------------------------------------------
# Precomputation helper
# ---------------------------------------------------------------------------
def compute_regime_gate(
detector: str,
threshold: float = 0.0,
*,
lake_root: str = "",
market: str = "US",
start: str = "2015-01-03",
end: str = "2026-08-19",
vol_low: float = 0.0,
vol_high: float = 999.0,
hmm_field: str = "sp_hmm_p_regime1",
) -> pd.Series:
"""Build a per-date regime gate series from lake bars.
Parameters
----------
detector : str — ``"dispersion"``, ``"vol"``, or ``"hmm"``.
threshold : float — for ``dispersion``: min CS dispersion to allow trading.
For ``hmm``: min HMM posterior to allow trading.
Ignored for ``vol`` (uses ``vol_low``/``vol_high`` band instead).
lake_root, market : str — lake location.
start, end : str — date window.
vol_low, vol_high : float — annualized vol band for the ``vol`` detector.
hmm_field : str — HMM feature column name for the ``hmm`` detector.
Returns
-------
pd.Series — bool, indexed by datetime. True = trade allowed.
"""
from tac_qlib.data.config import LakeConfig, resolve_lake_root
cfg = LakeConfig(resolve_lake_root(lake_root or None), market)
symbols = _universe_symbols(cfg)
close_df, vol_df = _load_daily_bars(symbols, cfg, start, end)
if close_df.empty:
return pd.Series(dtype=bool)
if detector == "dispersion":
return _dispersion_gate(close_df, threshold)
elif detector == "vol":
return _vol_gate(close_df, vol_low, vol_high)
elif detector == "hmm":
return _hmm_gate(cfg, symbols, threshold, start, end, hmm_field)
else:
raise ValueError(f"Unknown detector: {detector!r}")
def _universe_symbols(cfg) -> list:
"""Read symbols from the lake symbols.parquet."""
import pathlib
sp = cfg.lake_root / "symbols.parquet"
if sp.exists():
df = pd.read_parquet(sp)
col = "symbol" if "symbol" in df.columns else df.columns[0]
return sorted(df[col].astype(str).str.upper().tolist())
return []
def _load_daily_bars(symbols, cfg, start, end):
"""Load daily close prices for all symbols into a wide DataFrame."""
closes = {}
vols = {}
for sym in symbols:
p = cfg.bar_path("1d", sym)
if not p.exists():
continue
try:
df = pd.read_parquet(p)
except Exception:
continue
if not len(df):
continue
tcol = df["t"] if "t" in df.columns else df["date"]
ts = pd.to_datetime(tcol)
df = df.assign(_t=ts).set_index("_t").sort_index()
df = df.loc[start:end]
if len(df) < 22:
continue
closes[sym] = df["c"]
if "v" in df.columns:
vols[sym] = df["v"]
close_df = pd.DataFrame(closes)
vol_df = pd.DataFrame(vols) if vols else None
return close_df, vol_df
def _dispersion_gate(close_df, threshold):
"""Cross-sectional dispersion of 22-day rolling returns."""
if close_df.empty or close_df.shape[1] < 2:
return pd.Series(dtype=bool)
ret = close_df.pct_change(22)
cs_disp = ret.std(axis=1)
gate = cs_disp >= threshold
gate.iloc[:22] = True # warmup: allow trading
return gate
def _vol_gate(close_df, vol_low, vol_high):
"""Cross-sectional mean of 22-day rolling realized vol."""
if close_df.empty or close_df.shape[1] < 2:
return pd.Series(dtype=bool)
import numpy as np
log_ret = np.log(close_df / close_df.shift(1))
rv22 = log_ret.rolling(22).std() * (252 ** 0.5)
cs_mean_vol = rv22.mean(axis=1)
gate = (cs_mean_vol >= vol_low) & (cs_mean_vol <= vol_high)
gate.iloc[:22] = True # warmup
return gate
def _hmm_gate(cfg, symbols, threshold, start, end, hmm_field):
"""HMM regime posterior gate from persisted SP features."""
feat_root = cfg.lake_root / "features"
all_posteriors = {}
for sym in symbols:
# check both ta and sp family paths
for family in ("sp", "ta"):
p = feat_root / f"market=US" / f"timeframe=1d" / f"family={family}" / f"symbol={sym}.parquet"
if not p.exists():
continue
try:
df = pd.read_parquet(p)
except Exception:
continue
if hmm_field not in df.columns:
continue
tcol = df["t"] if "t" in df.columns else df["date"]
ts = pd.to_datetime(tcol)
s = pd.Series(df[hmm_field].values, index=ts, name=sym)
s = s.loc[start:end].dropna()
if len(s) > 0:
all_posteriors[sym] = s
break
if not all_posteriors:
# no HMM features found — default open
idx = pd.date_range(start, end, freq="B")
return pd.Series(True, index=idx)
post_df = pd.DataFrame(all_posteriors)
cs_mean = post_df.mean(axis=1)
gate = cs_mean >= threshold
return gate
@@ -0,0 +1,301 @@
"""Signal-quality gate TopkDropout strategy.
Subclass of ``qlib.contrib.strategy.signal_strategy.TopkDropoutStrategy`` that
holds the book (issues NO orders) when the model's recent prediction accuracy
is below a threshold. When the gate is open it behaves exactly like the
reference TopkDropoutStrategy.
Unlike the regime gate (which asks "is the market calm?"), the signal-quality
gate asks "are my predictions accurate?" — and works across ALL years.
The gate is provided as a precomputed ``pd.Series`` of booleans indexed by
datetime (True = trade allowed). The companion ``compute_signal_quality_gate``
function builds this series from a pred.pkl and lake bars; call it once before
backtesting and pass the result as the ``signal_quality_gate`` parameter.
"""
from __future__ import annotations
import numpy as np
import pandas as pd
from qlib.backtest.decision import TradeDecisionWO
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
__all__ = ["SignalQualityGateStrategy", "compute_signal_quality_gate"]
class SignalQualityGateStrategy(TopkDropoutStrategy):
"""TopkDropout with signal-quality gate overlay.
When ``lake_root`` is provided the gate is computed on-the-fly from the
signal (``<PRED>``) and close prices — no precomputed gate file needed.
This ensures the gate matches the model that is actually generating the
predictions (critical when the model is retrained each year).
Parameters
----------
topk, n_drop, method_sell, method_buy, hold_thresh, only_tradable,
forbid_all_trade_at_limit : same as ``TopkDropoutStrategy``.
signal_quality_gate : pd.Series — precomputed per-date gate (bool).
signal_quality_gate_path : str — path to pickled gate Series.
lake_root : str — lake root for on-the-fly gate computation (preferred).
gate_topk, gate_lookback, gate_threshold : int/float — gate params.
gate_start, gate_end : str — date window for loading close prices.
"""
def __init__(self, *, signal_quality_gate=None, signal_quality_gate_path=None,
lake_root=None, gate_topk=10, gate_lookback=5, gate_threshold=0.5,
gate_start="2015-01-03", gate_end="2026-08-19", **kwargs):
super().__init__(**kwargs)
self._sq_gate_computed = False
if signal_quality_gate is not None:
self._sq_gate = signal_quality_gate
self._sq_gate_computed = True
elif signal_quality_gate_path is not None:
import pickle
with open(signal_quality_gate_path, "rb") as f:
self._sq_gate = pickle.load(f)
self._sq_gate_computed = True
elif lake_root is not None:
self._sq_gate = None
self._lake_root = lake_root
self._gate_topk = gate_topk
self._gate_lookback = gate_lookback
self._gate_threshold = gate_threshold
self._gate_start = gate_start
self._gate_end = gate_end
else:
self._sq_gate = None
def _compute_gate_on_fly(self):
"""Compute gate from the signal (pred.pkl) and lake close prices."""
import pickle
from pathlib import Path
signal_path = self._signal
if not Path(signal_path).exists():
return
with open(signal_path, "rb") as f:
pred = pickle.load(f)
# Handle MultiIndex DataFrame -> unstack to wide
if isinstance(pred, pd.DataFrame) and isinstance(pred.index, pd.MultiIndex):
pred = pred.iloc[:, 0]
pred.index = pd.MultiIndex.from_arrays([
pd.to_datetime(pred.index.get_level_values(0)).normalize(),
pred.index.get_level_values(1)
])
pred = pred.unstack(level=1)
elif isinstance(pred, pd.DataFrame):
pred = pred.iloc[:, 0] if pred.shape[1] >= 1 else pred.squeeze()
pred.index = pd.to_datetime(pred.index).normalize()
# Load close prices from lake
close_df = _load_close_prices(self._lake_root, "US", self._gate_start, self._gate_end)
if close_df.empty:
return
ret_df = close_df.pct_change()
ret_df.index = pd.to_datetime(ret_df.index).normalize()
pred_dates = sorted(pred.index.unique())
if len(pred_dates) < 2:
self._sq_gate = pd.Series(True, index=pd.DatetimeIndex(pred_dates))
self._sq_gate_computed = True
return
hit_rates = {}
for i in range(1, len(pred_dates)):
day = pred_dates[i]
prev_day = pred_dates[i - 1]
try:
prev_scores = pred.loc[prev_day]
except KeyError:
continue
if isinstance(prev_scores, pd.DataFrame):
prev_scores = prev_scores.iloc[:, 0]
prev_scores = prev_scores.dropna().sort_values(ascending=False)
topk_syms = list(prev_scores.index[:self._gate_topk])
if day not in ret_df.index:
continue
today_ret = ret_df.loc[day]
topk_rets = today_ret.reindex(topk_syms).dropna()
if len(topk_rets) == 0:
continue
hit_rates[day] = (topk_rets > 0).sum() / len(topk_rets)
if not hit_rates:
self._sq_gate_computed = True
return
hr_series = pd.Series(hit_rates).sort_index()
rolling_hr = hr_series.rolling(self._gate_lookback, min_periods=1).mean()
gate = rolling_hr >= self._gate_threshold
gate.iloc[:self._gate_lookback] = True
self._sq_gate = gate
self._sq_gate_computed = True
def _gate_open(self, trade_start_time) -> bool:
if self._sq_gate is None:
return True
ts = pd.Timestamp(trade_start_time)
known = self._sq_gate[self._sq_gate.index <= ts]
if len(known):
return bool(known.iloc[-1])
return True # default open if no history yet
def generate_trade_decision(self, execute_result=None):
if not self._sq_gate_computed:
self._compute_gate_on_fly()
trade_step = self.trade_calendar.get_trade_step()
trade_start_time, _ = self.trade_calendar.get_step_time(trade_step)
if not self._gate_open(trade_start_time):
return TradeDecisionWO([], self)
return super().generate_trade_decision(execute_result)
# ---------------------------------------------------------------------------
# Precomputation helper
# ---------------------------------------------------------------------------
def compute_signal_quality_gate(
pred_path: str,
*,
lake_root: str = "",
market: str = "US",
topk: int = 10,
lookback: int = 5,
threshold: float = 0.5,
start: str = "2015-01-03",
end: str = "2026-08-19",
) -> pd.Series:
"""Build a per-date signal-quality gate series from a pred.pkl and lake bars.
For each day, checks whether the model's topk picks from the previous day
had positive returns. Computes a rolling hit rate over ``lookback`` days
and opens the gate when hit rate >= ``threshold``.
Parameters
----------
pred_path : str — path to pred.pkl (from rd_train / rd_predict).
lake_root, market : str — lake location (for loading close prices).
topk : int — number of top picks to track for hit rate.
lookback : int — rolling window for hit rate computation.
threshold : float — hit rate threshold to keep trading.
start, end : str — date window for loading prices.
Returns
-------
pd.Series — bool, indexed by datetime. True = trade allowed.
"""
from pathlib import Path
import pickle
# Load pred.pkl
with open(pred_path, "rb") as f:
pred = pickle.load(f)
# Handle MultiIndex DataFrame (datetime, instrument) -> unstack to wide
if isinstance(pred, pd.DataFrame) and isinstance(pred.index, pd.MultiIndex):
pred = pred.iloc[:, 0] # take score column as Series
pred.index = pd.MultiIndex.from_arrays([
pd.to_datetime(pred.index.get_level_values(0)).normalize(),
pred.index.get_level_values(1)
])
# Unstack to wide: dates x instruments
pred = pred.unstack(level=1)
elif isinstance(pred, pd.DataFrame):
pred = pred.iloc[:, 0] if pred.shape[1] >= 1 else pred.squeeze()
pred.index = pd.to_datetime(pred.index).normalize()
# Load close prices from lake
close_df = _load_close_prices(lake_root, market, start, end)
if close_df.empty:
return pd.Series(dtype=bool)
ret_df = close_df.pct_change()
ret_df.index = pd.to_datetime(ret_df.index).normalize()
# Get sorted unique prediction dates
pred_dates = sorted(pred.index.unique())
if len(pred_dates) < 2:
return pd.Series(True, index=pd.DatetimeIndex(pred_dates))
# Compute hit rates
hit_rates = {}
for i in range(1, len(pred_dates)):
day = pred_dates[i]
# Get yesterday's topk
prev_day = pred_dates[i - 1]
try:
prev_scores = pred.loc[prev_day]
except KeyError:
continue
if isinstance(prev_scores, pd.DataFrame):
prev_scores = prev_scores.iloc[:, 0]
prev_scores = prev_scores.dropna().sort_values(ascending=False)
topk_syms = list(prev_scores.index[:topk])
# Get today's returns
if day not in ret_df.index:
continue
today_ret = ret_df.loc[day]
topk_rets = today_ret.reindex(topk_syms).dropna()
if len(topk_rets) == 0:
continue
hit_rates[day] = (topk_rets > 0).sum() / len(topk_rets)
if not hit_rates:
return pd.Series(dtype=bool)
hr_series = pd.Series(hit_rates).sort_index()
# Rolling hit rate
rolling_hr = hr_series.rolling(lookback, min_periods=1).mean()
# Gate is open when rolling hit rate >= threshold
gate = rolling_hr >= threshold
gate.iloc[:lookback] = True # warmup: allow trading
return gate
def _load_close_prices(lake_root, market, start, end):
"""Load daily close prices for all symbols into a wide DataFrame."""
from pathlib import Path
lake = Path(lake_root)
symbols_parquet = lake / "symbols.parquet"
if not symbols_parquet.exists():
return pd.DataFrame()
df = pd.read_parquet(symbols_parquet)
col = "symbol" if "symbol" in df.columns else df.columns[0]
symbols = sorted(df[col].astype(str).str.upper().tolist())
closes = {}
for sym in symbols:
p = lake / "market=US" / "timeframe=1d" / f"symbol={sym}.parquet"
if not p.exists():
continue
try:
bar = pd.read_parquet(p)
except Exception:
continue
if not len(bar):
continue
tcol = "t" if "t" in bar.columns else "date"
ts = pd.to_datetime(bar[tcol])
bar = bar.assign(_t=ts).set_index("_t").sort_index()
bar = bar.loc[start:end]
if len(bar) < 10:
continue
closes[sym] = bar["c"] if "c" in bar.columns else bar["close"]
return pd.DataFrame(closes)
@@ -0,0 +1,202 @@
"""Weekly-rebalance TopkDropout strategy.
Turnover-reduction variant of ``qlib.contrib.strategy.signal_strategy.TopkDropoutStrategy``:
the topk/n_drop selection and sizing are identical to the reference, but the
target book is recomputed only on the first trading day of each ISO week; on the
other days the strategy issues NO orders (holds the book untouched).
The weekly cadence is derived from the qlib trade calendar: a rebalance happens
when the current trade step's date belongs to a different ISO ``(year, week)``
than the previous trade step. ``hold_band_pct`` (default 0) optionally skips
tiny rebalances: when a name's existing position differs from the new target by
less than this fraction, no order is generated for it.
"""
from __future__ import annotations
from typing import List
import numpy as np
import pandas as pd
from qlib.backtest import Order
from qlib.backtest.decision import OrderDir, TradeDecisionWO
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
__all__ = ["WeeklyRebalanceDropoutStrategy"]
DEFAULT_HOLD_BAND_PCT = 0.0
class WeeklyRebalanceDropoutStrategy(TopkDropoutStrategy):
"""TopkDropout rebalanced once per ISO week; holds otherwise.
Parameters
----------
topk, n_drop, method_sell, method_buy, hold_thresh, only_tradable,
forbid_all_trade_at_limit : same as ``TopkDropoutStrategy``.
hold_band_pct : skip order for a name whose deviation from target weight is
below this fraction of the target (no-trade buffer band).
"""
def __init__(self, *, topk, n_drop, hold_band_pct: float = DEFAULT_HOLD_BAND_PCT, **kwargs):
super().__init__(topk=topk, n_drop=n_drop, **kwargs)
self.hold_band_pct = hold_band_pct
@staticmethod
def _iso_week(ts) -> tuple:
return (ts.year, ts.week)
def generate_trade_decision(self, execute_result=None):
import copy
trade_step = self.trade_calendar.get_trade_step()
trade_start_time, trade_end_time = self.trade_calendar.get_step_time(trade_step)
cur_week = self._iso_week(trade_start_time)
prev_week = getattr(self, "_last_week", None)
self._last_week = cur_week
if prev_week is not None and prev_week == cur_week:
# not the first trading day of this ISO week -> hold
return TradeDecisionWO([], self)
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(trade_step, shift=1)
pred_score = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
if isinstance(pred_score, pd.DataFrame):
pred_score = pred_score.iloc[:, 0]
if pred_score is None:
return TradeDecisionWO([], self)
if self.only_tradable:
def get_first_n(li, n, reverse=False):
cur_n = 0
res = []
for si in reversed(li) if reverse else li:
if self.trade_exchange.is_stock_tradable(
stock_id=si, start_time=trade_start_time, end_time=trade_end_time
):
res.append(si)
cur_n += 1
if cur_n >= n:
break
return res[::-1] if reverse else res
def get_last_n(li, n):
return get_first_n(li, n, reverse=True)
def filter_stock(li):
return [
si
for si in li
if self.trade_exchange.is_stock_tradable(
stock_id=si, start_time=trade_start_time, end_time=trade_end_time
)
]
else:
def get_first_n(li, n):
return list(li)[:n]
def get_last_n(li, n):
return list(li)[-n:]
def filter_stock(li):
return li
current_temp: "object" = copy.deepcopy(self.trade_position)
sell_order_list: List[Order] = []
buy_order_list: List[Order] = []
cash = current_temp.get_cash()
current_stock_list = current_temp.get_stock_list()
last = pred_score.reindex(current_stock_list).sort_values(ascending=False).index
if self.method_buy == "top":
today = get_first_n(
pred_score[~pred_score.index.isin(last)].sort_values(ascending=False).index,
self.n_drop + self.topk - len(last),
)
elif self.method_buy == "random":
topk_candi = get_first_n(pred_score.sort_values(ascending=False).index, self.topk)
candi = list(filter(lambda x: x not in last, topk_candi))
n = self.n_drop + self.topk - len(last)
try:
today = np.random.choice(candi, n, replace=False)
except ValueError:
today = candi
else:
raise NotImplementedError(f"This type of input is not supported")
comb = pred_score.reindex(last.union(pd.Index(today))).sort_values(ascending=False).index
if self.method_sell == "bottom":
sell = last[last.isin(get_last_n(comb, self.n_drop))]
elif self.method_sell == "random":
candi = filter_stock(last)
try:
sell = pd.Index(np.random.choice(candi, self.n_drop, replace=False) if len(last) else [])
except ValueError:
sell = candi
else:
raise NotImplementedError(f"This type of input is not supported")
buy = today[: len(sell) + self.topk - len(last)]
for code in current_stock_list:
if not self.trade_exchange.is_stock_tradable(
stock_id=code,
start_time=trade_start_time,
end_time=trade_end_time,
direction=None if self.forbid_all_trade_at_limit else OrderDir.SELL,
):
continue
if code in sell:
time_per_step = self.trade_calendar.get_freq()
if current_temp.get_stock_count(code, bar=time_per_step) < self.hold_thresh:
continue
sell_amount = current_temp.get_stock_amount(code=code)
sell_order = Order(
stock_id=code,
amount=sell_amount,
start_time=trade_start_time,
end_time=trade_end_time,
direction=Order.SELL,
)
if self.trade_exchange.check_order(sell_order):
sell_order_list.append(sell_order)
trade_val, trade_cost, trade_price = self.trade_exchange.deal_order(
sell_order, position=current_temp
)
cash += trade_val - trade_cost
if len(buy) == 0:
return TradeDecisionWO(sell_order_list, self)
value = cash * self.risk_degree / len(buy)
for code in buy:
if not self.trade_exchange.is_stock_tradable(
stock_id=code,
start_time=trade_start_time,
end_time=trade_end_time,
direction=None if self.forbid_all_trade_at_limit else OrderDir.BUY,
):
continue
buy_price = self.trade_exchange.get_deal_price(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time, direction=OrderDir.BUY
)
buy_amount = value / buy_price
factor = self.trade_exchange.get_factor(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time
)
buy_amount = self.trade_exchange.round_amount_by_trade_unit(buy_amount, factor)
buy_order = Order(
stock_id=code,
amount=buy_amount,
start_time=trade_start_time,
end_time=trade_end_time,
direction=Order.BUY,
)
buy_order_list.append(buy_order)
return TradeDecisionWO(sell_order_list + buy_order_list, self)
+29 -2
View File
@@ -6,10 +6,11 @@ The lake is a hive-partitioned parquet store (see ``tac-engine/skills/tradeac-la
├── market=US/ ├── market=US/
│ └── timeframe=1d/ │ └── timeframe=1d/
│ └── symbol=AAPL.parquet # OHLCV bars: t, date, o, h, l, c, v, n, vw │ └── symbol=AAPL.parquet # OHLCV bars: t, date, o, h, l, c, v, n, vw
├── features/ # ta-lib indicators, wide format ├── features/ # indicators, wide format, family tier
│ └── market=US/ │ └── market=US/
│ └── timeframe=1d/ │ └── timeframe=1d/
│ └── symbol=AAPL.parquet # t, sma_5, sma_20, rsi_14, ... │ ├── family=ta/symbol=AAPL.parquet # t, sma_5, sma_20, rsi_14, ...
│ └── family=sp/symbol=AAPL.parquet # t, sp_ou_*, sp_hmm_*, ...
├── calendar.parquet # trading days per market ├── calendar.parquet # trading days per market
├── coverage.parquet # per (market,timeframe,symbol) loaded windows ├── coverage.parquet # per (market,timeframe,symbol) loaded windows
└── symbols.parquet # asset master └── symbols.parquet # asset master
@@ -107,8 +108,34 @@ class LakeConfig:
return self.lake_root / "features" / f"market={self.market}" / f"timeframe={timeframe}" return self.lake_root / "features" / f"market={self.market}" / f"timeframe={timeframe}"
def features_path(self, timeframe: str, symbol: str) -> Path: def features_path(self, timeframe: str, symbol: str) -> Path:
# Legacy flat path (no family tier). Prefer `load_features` which
# resolves the family=ta|sp partition layout.
return self.features_dir(timeframe) / f"symbol={str(symbol).upper()}.parquet" return self.features_dir(timeframe) / f"symbol={str(symbol).upper()}.parquet"
def load_features(self, timeframe: str, symbol: str) -> pd.DataFrame:
"""All feature columns for a symbol, merging the `family=ta` and
`family=sp` partitions by timestamp. Returns an empty frame when no
feature files exist (legacy flat layout falls back transparently)."""
sym = str(symbol).upper()
frames = []
for family in ("ta", "sp"):
p = self.features_dir(timeframe) / f"family={family}" / f"symbol={sym}.parquet"
if p.exists():
frames.append(pd.read_parquet(p))
if not frames:
flat = self.features_dir(timeframe) / f"symbol={sym}.parquet"
if flat.exists():
return pd.read_parquet(flat)
return pd.DataFrame()
if len(frames) == 1:
return frames[0]
merged = frames[0]
for extra in frames[1:]:
merged = merged.merge(extra, on="t", how="outer", suffixes=("", "_dup"))
for c in [c for c in merged.columns if c.endswith("_dup")]:
merged = merged.drop(columns=c)
return merged
def calendar_path(self) -> Path: def calendar_path(self) -> Path:
return self.lake_root / "calendar.parquet" return self.lake_root / "calendar.parquet"
+1 -2
View File
@@ -173,8 +173,7 @@ class LakeFeatureProvider(FeatureProvider):
def _load_feature_df(self, instrument: str, timeframe: str) -> pd.DataFrame: def _load_feature_df(self, instrument: str, timeframe: str) -> pd.DataFrame:
key = (instrument, timeframe) key = (instrument, timeframe)
if key not in self._feature_cache: if key not in self._feature_cache:
p = self.cfg.features_path(timeframe, instrument) self._feature_cache[key] = self.cfg.load_features(timeframe, instrument)
self._feature_cache[key] = pd.read_parquet(p) if p.exists() else pd.DataFrame()
return self._feature_cache[key] return self._feature_cache[key]
@staticmethod @staticmethod
-141
View File
@@ -1,141 +0,0 @@
# -----------------------------------------------------------------------------
# ISOLATION: multi-seed RankIC ensemble, ablate-B generic-only feature set.
#
# Isolates the ensemble effect on the SP-5d rank signal. Same panel, segments,
# history (full backfilled 2016+) and feature set as the exp-9 ablate-B winner
# (generic-only sp_* families: jump,har,trend,hurst,signature), but replaces the
# single RankICLGBModel with a 5-seed RankICEnsembleLGBModel (42,7,2026,99,123)
# that averages per-day predictions.
#
# Differs from exp-15 (tac-rd-rank-ensemble, mlflow exp 15) ONLY by dropping the
# TA subset (rsi_14,roc_10,macd_hist,willr_14,atr_14) and the inter-asset xr_*
# features, so any change vs exp-15 is attributable to the feature set alone,
# and any change vs exp-9 is attributable to the ensemble + full history alone.
#
# Run:
# rd_run_workflow config_path=experiments/workflows/exp12_isolation_ensemble.yaml \
# experiment_name=tac-rd-rank-ensemble-isolated
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///mlruns.db"
default_exp_name: "tac-rd-rank-ensemble-isolated"
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-14
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
infer_processors:
- class: DropAllNaN
kwargs: {}
- class: ProcessInf
kwargs: {}
- class: CSRankNorm
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 10
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
-141
View File
@@ -1,141 +0,0 @@
# -----------------------------------------------------------------------------
# EXP 18 - Risk-limit control: reference model + TopkDropout baseline (A).
#
# Signal/model identical to the reference (tac-rd-rank-ensemble-isolated,
# run 0cea66d9...): RankICEnsembleLGBModel (parallel, 5 seeds) on the 50-ETF
# SP-5d panel, test 2026-01-04..2026-08-10. This workflow reproduces the
# unconstrained TopkDropout baseline net-of-cost so the risk-limited variant
# (same pred, liquidity/size/concentration caps) can be compared 1:1.
#
# The risk_limits spec itself is applied via rd_backtest / rd_strategy_targets
# (tool-level param, not a YAML key); this run records the unconstrained
# baseline that the limit A/B is measured against.
#
# Run:
# rd_run_workflow config_path=experiments/workflows/exp18-risk-limit/a_baseline.yaml \
# experiment_name=tac-rd-risk-limit
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-risk-limit"
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
parallel: 5
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-14
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
infer_processors:
- class: DropAllNaN
kwargs: {}
- class: ProcessInf
kwargs: {}
- class: CSRankNorm
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 10
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -1,135 +0,0 @@
# -----------------------------------------------------------------------------
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
#
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
#
# Run:
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
# experiment_name=tac-rd-risk-limit
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-risk-limit"
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
parallel: 5
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-14
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
infer_processors:
- class: DropAllNaN
kwargs: {}
- class: ProcessInf
kwargs: {}
- class: CSRankNorm
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 10
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -1,136 +0,0 @@
# -----------------------------------------------------------------------------
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
#
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
#
# Run:
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
# experiment_name=tac-rd-risk-limit
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-risk-limit"
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
parallel: 5
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-14
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
infer_processors:
- class: DropAllNaN
kwargs: {}
- class: ProcessInf
kwargs: {}
- class: CSRankNorm
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: MomentumGateTopk
module_path: tac_qlib.contrib.strategy.momentum_gate
kwargs:
signal: "<PRED>"
topk: 10
n_drop: 2
min_momentum: 0.0
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -1,138 +0,0 @@
# -----------------------------------------------------------------------------
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
#
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
#
# Run:
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
# experiment_name=tac-rd-risk-limit
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-risk-limit"
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
parallel: 5
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-14
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
infer_processors:
- class: DropAllNaN
kwargs: {}
- class: ProcessInf
kwargs: {}
- class: CSRankNorm
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: HmmRiskTopk
module_path: tac_qlib.contrib.strategy.hmm_risk
kwargs:
signal: "<PRED>"
topk: 10
n_drop: 2
hmm_pause_pct: 0.70
drawdown_pause_pct: 8.0
liquidity_floor_adv: 5000000
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -1,137 +0,0 @@
# -----------------------------------------------------------------------------
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
#
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
#
# Run:
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
# experiment_name=tac-rd-risk-limit
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-risk-limit"
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
weight_mode: rolling_ic
rolling_ic_window: 21
parallel: 5
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-14
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
infer_processors:
- class: DropAllNaN
kwargs: {}
- class: ProcessInf
kwargs: {}
- class: CSRankNorm
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 10
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -1,135 +0,0 @@
# -----------------------------------------------------------------------------
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
#
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
#
# Run:
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
# experiment_name=tac-rd-risk-limit
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead,sma_3,ema_3" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-risk-limit"
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
parallel: 5
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-14
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
infer_processors:
- class: DropAllNaN
kwargs: {}
- class: ProcessInf
kwargs: {}
- class: CSRankNorm
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 10
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d