Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
002117c33e | ||
|
|
e952feed0a |
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,48 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Precompute the signal-quality gate series and save to pickle.
|
||||
|
||||
Usage:
|
||||
python precompute_signal_quality_gate.py <pred_path> <output_path> [topk] [lookback] [threshold]
|
||||
|
||||
Example:
|
||||
python precompute_signal_quality_gate.py \
|
||||
/home/data/lake/mlruns/49/34165f27e4a34378ad54843a079a78c0/artifacts/pred.pkl \
|
||||
/app/experiments/book/data/signal_quality_gate/sq_gate_5d_0.50.pkl \
|
||||
10 5 0.5
|
||||
"""
|
||||
import sys
|
||||
import pickle
|
||||
from pathlib import Path
|
||||
|
||||
# Add tac-qlib to path
|
||||
sys.path.insert(0, "/app/tac-qlib")
|
||||
|
||||
from tac_qlib.contrib.strategy.signal_quality_gate import compute_signal_quality_gate
|
||||
|
||||
if __name__ == "__main__":
|
||||
if len(sys.argv) < 3:
|
||||
print(__doc__)
|
||||
sys.exit(1)
|
||||
|
||||
pred_path = sys.argv[1]
|
||||
output_path = sys.argv[2]
|
||||
topk = int(sys.argv[3]) if len(sys.argv) > 3 else 10
|
||||
lookback = int(sys.argv[4]) if len(sys.argv) > 4 else 5
|
||||
threshold = float(sys.argv[5]) if len(sys.argv) > 5 else 0.5
|
||||
|
||||
lake_root = "/home/data/lake"
|
||||
|
||||
print(f"Computing signal-quality gate: topk={topk}, lookback={lookback}, threshold={threshold}")
|
||||
gate = compute_signal_quality_gate(
|
||||
pred_path,
|
||||
lake_root=lake_root,
|
||||
topk=topk,
|
||||
lookback=lookback,
|
||||
threshold=threshold,
|
||||
)
|
||||
|
||||
print(f"Gate: {gate.sum()}/{len(gate)} days open ({gate.mean():.1%})")
|
||||
|
||||
with open(output_path, "wb") as f:
|
||||
pickle.dump(gate, f)
|
||||
print(f"Saved to {output_path}")
|
||||
+19
-23
@@ -1,16 +1,16 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# ABLATION A (baseline): LightGBM with RankIC early-stopping on the 50-ETF SP-5d
|
||||
# panel, using ALL 24 sp_* feature columns (ou,hmm,jump,har,trend,hurst,
|
||||
# signature). Copy of the canonical workflow_lgb_sp5d_rankic.yaml with a
|
||||
# distinct experiment name so the ablation runs are isolated.
|
||||
# Signal-quality gate: TopkDropout gated by rolling hit-rate of topk picks.
|
||||
#
|
||||
# Run:
|
||||
# rd_run_workflow config_path=tac-qlib/workflows/ablate_baseline_all_sp_fields.yaml \
|
||||
# experiment_name=tac-rd-rank-ablate
|
||||
# 1. Compute the gate: python precompute_signal_quality_gate.py <pred.pkl> <gate.pkl>
|
||||
# 2. Run this workflow: rd_run_workflow config_path=<this yaml> experiment_name=<exp>
|
||||
#
|
||||
# The strategy loads the precomputed gate from signal_quality_gate_path.
|
||||
# When hit rate >= threshold, trade; otherwise, go to cash.
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
{%- set GATE_PATH = "/app/experiments/book/data/signal_quality_gate/sq_gate_5d_0.50.pkl" %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
@@ -40,27 +40,22 @@ qlib_init:
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-rank-ablate"
|
||||
default_exp_name: "tac-rd-sq-gate"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: RankICLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_gbdt
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 31
|
||||
n_estimators: 3000
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l2: 0.5
|
||||
learning_rate: 0.05
|
||||
num_leaves: 15
|
||||
n_estimators: 200
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seed: 42
|
||||
reg_alpha: 0.01
|
||||
reg_lambda: 0.01
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
@@ -110,12 +105,13 @@ task:
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
class: SignalQualityGateStrategy
|
||||
module_path: tac_qlib.contrib.strategy.signal_quality_gate
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
signal_quality_gate_path: "{{ GATE_PATH }}"
|
||||
topk: 10
|
||||
n_drop: 2
|
||||
n_drop: 1
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
+32
-52
@@ -1,68 +1,44 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
|
||||
#
|
||||
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
|
||||
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
|
||||
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
|
||||
#
|
||||
# Run:
|
||||
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
|
||||
# experiment_name=tac-rd-risk-limit
|
||||
# -----------------------------------------------------------------------------
|
||||
# Signal-quality gate backtest for 2021
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
{# Gate computed on-the-fly from signal + lake close prices #}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-risk-limit"
|
||||
default_exp_name: "tac-rd-sq-gate-2021"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: RankICEnsembleLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 31
|
||||
n_estimators: 3000
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l2: 0.5
|
||||
learning_rate: 0.05
|
||||
num_leaves: 15
|
||||
n_estimators: 200
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seeds: "42,7,2026,99,123"
|
||||
parallel: 1
|
||||
reg_alpha: 0.01
|
||||
reg_lambda: 0.01
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
@@ -74,9 +50,9 @@ task:
|
||||
kwargs:
|
||||
instruments: "{{ UNIVERSE }}"
|
||||
start_time: 2015-01-03
|
||||
end_time: 2026-08-14
|
||||
fit_start_time: 2016-01-04
|
||||
fit_end_time: 2025-09-01
|
||||
end_time: 2021-12-31
|
||||
fit_start_time: 2015-01-03
|
||||
fit_end_time: 2021-01-03
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
@@ -94,9 +70,9 @@ task:
|
||||
- class: Fillna
|
||||
kwargs: {}
|
||||
segments:
|
||||
train: [2016-01-04, 2025-09-01]
|
||||
valid: [2025-09-03, 2026-01-03]
|
||||
test: [2026-01-04, 2026-08-10]
|
||||
train: [2015-01-03, 2020-09-01]
|
||||
valid: [2020-09-03, 2021-01-03]
|
||||
test: [2021-01-04, 2021-12-31]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
@@ -104,25 +80,29 @@ task:
|
||||
kwargs: {}
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
kwargs: { ana_long_short: true, ann_scaler: 252 }
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
class: SignalQualityGateStrategy
|
||||
module_path: tac_qlib.contrib.strategy.signal_quality_gate
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
lake_root: "{{ LAKE }}"
|
||||
gate_topk: 10
|
||||
gate_lookback: 5
|
||||
gate_threshold: 0.5
|
||||
gate_start: "2015-01-03"
|
||||
gate_end: "2021-12-31"
|
||||
topk: 10
|
||||
n_drop: 2
|
||||
n_drop: 1
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-01-04
|
||||
end_time: 2026-08-10
|
||||
start_time: 2021-01-04
|
||||
end_time: 2021-12-31
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
@@ -0,0 +1,115 @@
|
||||
# Signal-quality gate backtest for 2023
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
{# Gate computed on-the-fly from signal + lake close prices #}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-sq-gate-2023"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.05
|
||||
num_leaves: 15
|
||||
n_estimators: 200
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.01
|
||||
reg_lambda: 0.01
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: "{{ UNIVERSE }}"
|
||||
start_time: 2015-01-03
|
||||
end_time: 2023-12-29
|
||||
fit_start_time: 2015-01-03
|
||||
fit_end_time: 2023-01-03
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||
infer_processors:
|
||||
- class: DropAllNaN
|
||||
kwargs: {}
|
||||
- class: ProcessInf
|
||||
kwargs: {}
|
||||
- class: CSRankNorm
|
||||
kwargs: {}
|
||||
- class: ZScoreNorm
|
||||
kwargs: {}
|
||||
- class: Fillna
|
||||
kwargs: {}
|
||||
segments:
|
||||
train: [2015-01-03, 2022-09-01]
|
||||
valid: [2022-09-03, 2023-01-03]
|
||||
test: [2023-01-03, 2023-12-29]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: { ana_long_short: true, ann_scaler: 252 }
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: SignalQualityGateStrategy
|
||||
module_path: tac_qlib.contrib.strategy.signal_quality_gate
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
lake_root: "{{ LAKE }}"
|
||||
gate_topk: 10
|
||||
gate_lookback: 5
|
||||
gate_threshold: 0.5
|
||||
gate_start: "2015-01-03"
|
||||
gate_end: "2023-12-29"
|
||||
topk: 10
|
||||
n_drop: 1
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2023-01-03
|
||||
end_time: 2023-12-29
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
codes: "{{ UNIVERSE }}"
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,115 @@
|
||||
# Signal-quality gate backtest for 2024
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
{# Gate computed on-the-fly from signal + lake close prices #}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-sq-gate-2024"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.05
|
||||
num_leaves: 15
|
||||
n_estimators: 200
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.01
|
||||
reg_lambda: 0.01
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: "{{ UNIVERSE }}"
|
||||
start_time: 2015-01-03
|
||||
end_time: 2024-12-31
|
||||
fit_start_time: 2015-01-03
|
||||
fit_end_time: 2024-01-03
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||
infer_processors:
|
||||
- class: DropAllNaN
|
||||
kwargs: {}
|
||||
- class: ProcessInf
|
||||
kwargs: {}
|
||||
- class: CSRankNorm
|
||||
kwargs: {}
|
||||
- class: ZScoreNorm
|
||||
kwargs: {}
|
||||
- class: Fillna
|
||||
kwargs: {}
|
||||
segments:
|
||||
train: [2015-01-03, 2023-09-01]
|
||||
valid: [2023-09-03, 2024-01-03]
|
||||
test: [2024-01-02, 2024-12-31]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: { ana_long_short: true, ann_scaler: 252 }
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: SignalQualityGateStrategy
|
||||
module_path: tac_qlib.contrib.strategy.signal_quality_gate
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
lake_root: "{{ LAKE }}"
|
||||
gate_topk: 10
|
||||
gate_lookback: 5
|
||||
gate_threshold: 0.5
|
||||
gate_start: "2015-01-03"
|
||||
gate_end: "2024-12-31"
|
||||
topk: 10
|
||||
n_drop: 1
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2024-01-02
|
||||
end_time: 2024-12-31
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
codes: "{{ UNIVERSE }}"
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
+31
-51
@@ -1,68 +1,44 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
|
||||
#
|
||||
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
|
||||
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
|
||||
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
|
||||
#
|
||||
# Run:
|
||||
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
|
||||
# experiment_name=tac-rd-risk-limit
|
||||
# -----------------------------------------------------------------------------
|
||||
# Signal-quality gate backtest for 2025
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
{# Gate computed on-the-fly from signal + lake close prices #}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-risk-limit"
|
||||
default_exp_name: "tac-rd-sq-gate-2025"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: RankICEnsembleLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 31
|
||||
n_estimators: 3000
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l2: 0.5
|
||||
learning_rate: 0.05
|
||||
num_leaves: 15
|
||||
n_estimators: 200
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seeds: "42,7"
|
||||
parallel: 2
|
||||
reg_alpha: 0.01
|
||||
reg_lambda: 0.01
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
@@ -74,9 +50,9 @@ task:
|
||||
kwargs:
|
||||
instruments: "{{ UNIVERSE }}"
|
||||
start_time: 2015-01-03
|
||||
end_time: 2026-08-14
|
||||
fit_start_time: 2016-01-04
|
||||
fit_end_time: 2025-09-01
|
||||
end_time: 2025-12-31
|
||||
fit_start_time: 2015-01-03
|
||||
fit_end_time: 2026-01-03
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
@@ -94,9 +70,9 @@ task:
|
||||
- class: Fillna
|
||||
kwargs: {}
|
||||
segments:
|
||||
train: [2016-01-04, 2025-09-01]
|
||||
train: [2015-01-03, 2025-09-01]
|
||||
valid: [2025-09-03, 2026-01-03]
|
||||
test: [2026-01-04, 2026-08-10]
|
||||
test: [2025-01-02, 2025-12-31]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
@@ -104,25 +80,29 @@ task:
|
||||
kwargs: {}
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
kwargs: { ana_long_short: true, ann_scaler: 252 }
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
class: SignalQualityGateStrategy
|
||||
module_path: tac_qlib.contrib.strategy.signal_quality_gate
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
lake_root: "{{ LAKE }}"
|
||||
gate_topk: 10
|
||||
gate_lookback: 5
|
||||
gate_threshold: 0.5
|
||||
gate_start: "2015-01-03"
|
||||
gate_end: "2025-12-31"
|
||||
topk: 10
|
||||
n_drop: 2
|
||||
n_drop: 1
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-01-04
|
||||
end_time: 2026-08-10
|
||||
start_time: 2025-01-02
|
||||
end_time: 2025-12-31
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
+25
-44
@@ -1,67 +1,44 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# ABLATION B (generic-only): same panel/model as the baseline, but feature
|
||||
# fields restricted to the model-free / generic stochastic-process families
|
||||
# (jump,har,trend,hurst,signature). Drops the model-specific ou (OU/AR-1
|
||||
# half-life) and hmm (2-state regime) families to test whether the generic
|
||||
# families alone dominate the rank dimension.
|
||||
#
|
||||
# Run:
|
||||
# rd_run_workflow config_path=tac-qlib/workflows/ablate_generic_only_sp_fields.yaml \
|
||||
# experiment_name=tac-rd-rank-ablate
|
||||
# -----------------------------------------------------------------------------
|
||||
# Signal-quality gate backtest for 2026
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
{# Gate computed on-the-fly from signal + lake close prices #}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-rank-ablate"
|
||||
default_exp_name: "tac-rd-sq-gate-2026"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: RankICLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_gbdt
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 31
|
||||
n_estimators: 3000
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l2: 0.5
|
||||
learning_rate: 0.05
|
||||
num_leaves: 15
|
||||
n_estimators: 200
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seed: 42
|
||||
reg_alpha: 0.01
|
||||
reg_lambda: 0.01
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
@@ -75,7 +52,7 @@ task:
|
||||
start_time: 2015-01-03
|
||||
end_time: 2026-08-10
|
||||
fit_start_time: 2015-01-03
|
||||
fit_end_time: 2025-09-01
|
||||
fit_end_time: 2026-01-03
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
@@ -103,20 +80,24 @@ task:
|
||||
kwargs: {}
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
kwargs: { ana_long_short: true, ann_scaler: 252 }
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
class: SignalQualityGateStrategy
|
||||
module_path: tac_qlib.contrib.strategy.signal_quality_gate
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
lake_root: "{{ LAKE }}"
|
||||
gate_topk: 10
|
||||
gate_lookback: 5
|
||||
gate_threshold: 0.5
|
||||
gate_start: "2015-01-03"
|
||||
gate_end: "2026-08-19"
|
||||
topk: 10
|
||||
n_drop: 2
|
||||
n_drop: 1
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
+26
-20
@@ -1,29 +1,35 @@
|
||||
# TradeAC custom-qlib-code snapshot (auto-generated)
|
||||
# parent repo HEAD : 125be7b96fb5975e798a0b4301eeb5809a8a181c
|
||||
# parent repo HEAD : e952feed0a66a20439f4f24ad5524233429cd0c3
|
||||
# tac-qlib/tac_qlib/contrib
|
||||
# tac-qlib/tac_qlib/data
|
||||
# per-file hashes (git hash-object):
|
||||
1b6298c4a5652f2e863cbdc385a1014a570fcd59 tac-qlib/tac_qlib/contrib/__init__.py
|
||||
b8112569f9b2537c45b6535e1a505a207878d322 tac-qlib/tac_qlib/contrib/__pycache__/__init__.cpython-312.pyc
|
||||
861592c63edd6a0853a9cb174b5970435b135fc8 tac-qlib/tac_qlib/contrib/__pycache__/__init__.cpython-312.pyc
|
||||
c76a9f17f680e74eea766eff27f7624359749ed6 tac-qlib/tac_qlib/contrib/data/__init__.py
|
||||
8d5333ebd2b44165c50cba639ca2d4ac3fc7cfec tac-qlib/tac_qlib/contrib/data/__pycache__/__init__.cpython-312.pyc
|
||||
18cb37c0354184c49fa2e598396d7df0634cce0f tac-qlib/tac_qlib/contrib/data/__pycache__/handler.cpython-312.pyc
|
||||
871ff1e163c29261f140c3f53d42a41e6504c779 tac-qlib/tac_qlib/contrib/data/handler.py
|
||||
5c547a2ef92e075e550fe6d01508a2f1d3f536bc tac-qlib/tac_qlib/contrib/data/__pycache__/__init__.cpython-312.pyc
|
||||
4f656130d167e79dcaaeb7783a121f0b36852374 tac-qlib/tac_qlib/contrib/data/__pycache__/handler.cpython-312.pyc
|
||||
0dd25ef161c6e0f15eafc84886e7e1381deb38c3 tac-qlib/tac_qlib/contrib/data/handler.py
|
||||
b151d139a0dcde87d74b21e7c4b729176ba5c39b tac-qlib/tac_qlib/contrib/model/__init__.py
|
||||
ab958203f33a99d12c7d923b6efb435189231666 tac-qlib/tac_qlib/contrib/model/__pycache__/__init__.cpython-312.pyc
|
||||
7478f6b0f6de419615c02d4d92b54529f689ef04 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_ensemble.cpython-312.pyc
|
||||
9f9014ddd9bce37490061312d51e8e6fe540fec4 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_gbdt.cpython-312.pyc
|
||||
ce77dea53f6a87c5379782709293bf8ff55b2c75 tac-qlib/tac_qlib/contrib/model/rank_ensemble.py
|
||||
ccfe7d554989aa7f3e5a2128ae663e51b2207149 tac-qlib/tac_qlib/contrib/model/rank_gbdt.py
|
||||
4afcf9058231111c412925f4c4b84e81d656db87 tac-qlib/tac_qlib/contrib/strategy/__init__.py
|
||||
74e5ecbbbb20bb71fd5cd083383de4ce88476712 tac-qlib/tac_qlib/contrib/strategy/__pycache__/__init__.cpython-312.pyc
|
||||
afaf562aeaa12cebc8529cd916153252e7e3c38a tac-qlib/tac_qlib/contrib/strategy/__pycache__/optimal_stop.cpython-312.pyc
|
||||
96a0a25201f0a1bb2fc2190e26228c5c0e711a79 tac-qlib/tac_qlib/contrib/strategy/hmm_risk.py
|
||||
816de5d58ae23d996635d42331cf9fc8963d5dbe tac-qlib/tac_qlib/contrib/strategy/momentum_gate.py
|
||||
b1489f2fc0dee85f0a4f90b2e6ad545ed9c8967b tac-qlib/tac_qlib/contrib/model/__pycache__/__init__.cpython-312.pyc
|
||||
121ef237da1df1b8e21a561c3ad0db200b901339 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_ensemble.cpython-312.pyc
|
||||
74d0da348cbcc3700c96b6f4fe4391488e61efc5 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_gbdt.cpython-312.pyc
|
||||
d3f051f3a8650c42fedc7b367b966f7c74fb5789 tac-qlib/tac_qlib/contrib/model/rank_ensemble.py
|
||||
d03e6611338918d4aac5eea4adf26f85a3763652 tac-qlib/tac_qlib/contrib/model/rank_gbdt.py
|
||||
2c2f167b693f4366a769998e3c9d4804f29e31e0 tac-qlib/tac_qlib/contrib/strategy/__init__.py
|
||||
f9cd9ab729e3248542ccc490af7adc2e51c71914 tac-qlib/tac_qlib/contrib/strategy/__pycache__/__init__.cpython-312.pyc
|
||||
cd3133cfbd2556b25c106e39ae97fb128df0e326 tac-qlib/tac_qlib/contrib/strategy/__pycache__/ic_gate.cpython-312.pyc
|
||||
6dd1c568a2961842793674390d5abffd1a0e71b8 tac-qlib/tac_qlib/contrib/strategy/__pycache__/optimal_stop.cpython-312.pyc
|
||||
03b5e4d80da00800f1b108bee0735d3d18d856d1 tac-qlib/tac_qlib/contrib/strategy/__pycache__/regime_gate.cpython-312.pyc
|
||||
a6a1c21ab71b62080830df45c4784b73c1531036 tac-qlib/tac_qlib/contrib/strategy/__pycache__/signal_quality_gate.cpython-312.pyc
|
||||
755e3b139496a5e22b0328db45c8199c33029fbc tac-qlib/tac_qlib/contrib/strategy/__pycache__/weekly_rebalance.cpython-312.pyc
|
||||
519a1f4c05dbe0ac018ab8b779eb33d53b4dd545 tac-qlib/tac_qlib/contrib/strategy/ic_gate.py
|
||||
79aaad9e39fcc740a773f4f63c512ce1086cfde0 tac-qlib/tac_qlib/contrib/strategy/optimal_stop.py
|
||||
7bcee5f0b09cfa721440f1354f16f2dd9a112b12 tac-qlib/tac_qlib/contrib/strategy/regime_gate.py
|
||||
16ab80b731aab6d8bb818525615d77c2d1fcb0c8 tac-qlib/tac_qlib/contrib/strategy/signal_quality_gate.py
|
||||
fe60bacdfedd48617863be31f24b7c7daebfac5a tac-qlib/tac_qlib/contrib/strategy/weekly_rebalance.py
|
||||
92e6e90eb0cd0a25142034560f27adb6b705b1a8 tac-qlib/tac_qlib/data/__init__.py
|
||||
0ed1ead6c1314a3f25784d453e54a15a8a04baaa tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc
|
||||
9609782800944c45b78bb58eaa7b51ba1b7f8f43 tac-qlib/tac_qlib/data/__pycache__/config.cpython-312.pyc
|
||||
a85628d71d12cfe5b18b1c884c5d829c89594579 tac-qlib/tac_qlib/data/__pycache__/providers.cpython-312.pyc
|
||||
686d36f6d101c547491ca866aa143aa542e17518 tac-qlib/tac_qlib/data/config.py
|
||||
d9f839be30026f337754a3f015425a8efdbe8e2a tac-qlib/tac_qlib/data/providers.py
|
||||
316bf4aa160cc8d15929ea648be03f4b4999667d tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc
|
||||
554a3f29d181b64effbf49a8161b32e7f93d8d3e tac-qlib/tac_qlib/data/__pycache__/config.cpython-312.pyc
|
||||
8b47f6d78ac046b6b7b2fb07bd7f3382773ffb73 tac-qlib/tac_qlib/data/__pycache__/providers.cpython-312.pyc
|
||||
53c9007a928841fd3c3b08450f9a6520ce1ac091 tac-qlib/tac_qlib/data/config.py
|
||||
8d0644f6f0d1efb94798ed444cc73e63b643459b tac-qlib/tac_qlib/data/providers.py
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -64,9 +64,13 @@ def check_transform_proc(proc_l, fit_start_time, fit_end_time):
|
||||
|
||||
|
||||
def get_common_feature_fields(lake_root=None, market="US", timeframe="1d") -> List[str]:
|
||||
"""Discover ta-lib columns present in *every* features parquet file of the lake.
|
||||
"""Discover feature columns present in *every* feature file of the lake.
|
||||
|
||||
Returns sorted field names (without the ``$`` prefix). Empty if no features are persisted.
|
||||
Walks the `family=ta|sp` partition layout (plus any legacy flat files).
|
||||
TA and SP columns are disjoint by construction, so the common set is
|
||||
computed per family (columns shared by all symbol files of that family),
|
||||
then the per-family results are unioned. Returns sorted field names
|
||||
(without the ``$`` prefix). Empty if no features are persisted.
|
||||
"""
|
||||
cfg = LakeConfig(lake_root, market)
|
||||
feat_dir = cfg.features_dir(timeframe)
|
||||
@@ -74,16 +78,30 @@ def get_common_feature_fields(lake_root=None, market="US", timeframe="1d") -> Li
|
||||
return []
|
||||
import pyarrow.parquet as pq
|
||||
|
||||
common = None
|
||||
for p in sorted(feat_dir.glob("symbol=*.parquet")):
|
||||
try:
|
||||
cols = set(pq.read_schema(p).names) - set(NON_FEATURE_COLUMNS)
|
||||
except Exception: # pragma: no cover - skip unreadable files
|
||||
continue
|
||||
common = cols if common is None else (common & cols)
|
||||
if not common:
|
||||
break
|
||||
return sorted(common) if common else []
|
||||
def _family_common(fam_dir: Path) -> set:
|
||||
common = None
|
||||
for p in sorted(fam_dir.glob("symbol=*.parquet")):
|
||||
try:
|
||||
cols = set(pq.read_schema(p).names) - set(NON_FEATURE_COLUMNS)
|
||||
except Exception: # pragma: no cover - skip unreadable files
|
||||
continue
|
||||
common = cols if common is None else (common & cols)
|
||||
if not common:
|
||||
break
|
||||
return common or set()
|
||||
|
||||
common: set = set()
|
||||
# family tier: features/market=*/timeframe=*/family=*/symbol=*.parquet
|
||||
for fam in ("ta", "sp"):
|
||||
fam_dir = feat_dir / f"family={fam}"
|
||||
if fam_dir.is_dir():
|
||||
common |= _family_common(fam_dir)
|
||||
# legacy flat: features/market=*/timeframe=*/symbol=*.parquet
|
||||
if (feat_dir / "family=ta").exists() or (feat_dir / "family=sp").exists():
|
||||
pass # family layout already covered
|
||||
else:
|
||||
common |= _family_common(feat_dir)
|
||||
return sorted(common)
|
||||
|
||||
|
||||
class DropAllNaN(processor_module.Processor):
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -56,7 +56,6 @@ import os
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from typing import List, Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from qlib.data.dataset import DatasetH
|
||||
@@ -80,15 +79,11 @@ class RankICEnsembleLGBModel(RankICLGBModel):
|
||||
forwarded.
|
||||
"""
|
||||
|
||||
def __init__(self, seeds: str = "42", parallel: int = 0, weight_mode: str = "equal", **kwargs):
|
||||
def __init__(self, seeds: str = "42", parallel: int = 0, **kwargs):
|
||||
self.seeds = [int(s.strip()) for s in str(seeds).split(",") if s.strip()]
|
||||
if not self.seeds:
|
||||
raise ValueError("seeds must contain at least one integer")
|
||||
self.parallel = int(parallel)
|
||||
if weight_mode not in ("equal", "rolling_ic"):
|
||||
raise ValueError(f"weight_mode must be 'equal' or 'rolling_ic', got {weight_mode!r}")
|
||||
self.weight_mode = weight_mode
|
||||
self.rolling_ic_window = int(kwargs.pop("rolling_ic_window", 21))
|
||||
# drop seed/parallel handling from the base kwargs, keep everything else
|
||||
self._model_kwargs = dict(kwargs)
|
||||
super().__init__(**self._model_kwargs)
|
||||
@@ -184,44 +179,11 @@ class RankICEnsembleLGBModel(RankICLGBModel):
|
||||
|
||||
# -------------------------------------------------------------- predict
|
||||
def predict(self, dataset: DatasetH, segment="test") -> pd.Series:
|
||||
"""Combine per-seed predictions.
|
||||
|
||||
``weight_mode='equal'`` (default): simple average, as before.
|
||||
``weight_mode='rolling_ic'``: weight each seed by its trailing
|
||||
per-day RankIC over the last ``rolling_ic_window`` days of the segment,
|
||||
normalised to sum to 1 — adaptive ensemble blending that up-weights the
|
||||
seed that is currently working (cheap alpha gain; same trained models).
|
||||
"""
|
||||
"""Average the per-seed predictions over the given segment."""
|
||||
if not self._models:
|
||||
raise ValueError("model is not fitted yet!")
|
||||
preds = [m.predict(dataset, segment=segment) for m in self._models]
|
||||
if len(preds) == 1:
|
||||
return preds[0]
|
||||
frame = pd.concat(preds, axis=1)
|
||||
frame.columns = [f"seed{m.params.get('seed', i)}" for i, m in enumerate(self._models)]
|
||||
if self.weight_mode == "equal":
|
||||
return frame.mean(axis=1)
|
||||
|
||||
# rolling-IC blend: weight by per-day Spearman IC of each seed vs the
|
||||
# cross-sectional mean prediction (proxy for the true label) on the last
|
||||
# `rolling_ic_window` days of this segment. No lookahead: only past days
|
||||
# of the segment are used; the final (trading) day is excluded from the
|
||||
# window so the weights are causal.
|
||||
mean_pred = frame.mean(axis=1)
|
||||
dates = sorted(frame.index.get_level_values(0).unique())
|
||||
win = [d for d in dates if d < dates[-1]][-self.rolling_ic_window :]
|
||||
ics = {}
|
||||
for col in frame.columns:
|
||||
if not win:
|
||||
ics[col] = 1.0
|
||||
continue
|
||||
sub = pd.DataFrame({"p": frame[col], "m": mean_pred})
|
||||
vals = []
|
||||
for d in win:
|
||||
s = sub[sub.index.get_level_values(0) == d]
|
||||
if len(s) >= 3 and s["p"].nunique() > 1 and s["m"].nunique() > 1:
|
||||
vals.append(s["p"].rank().corr(s["m"].rank()))
|
||||
ics[col] = float(np.mean(vals)) if vals else 1.0
|
||||
wsum = sum(ics.values()) or len(ics)
|
||||
weights = {c: v / wsum for c, v in ics.items()}
|
||||
return sum(frame[c] * weights[c] for c in frame.columns)
|
||||
return frame.mean(axis=1)
|
||||
|
||||
@@ -53,23 +53,61 @@ from qlib.workflow import R
|
||||
__all__ = ["RankICLGBModel", "rankic_feval"]
|
||||
|
||||
|
||||
def _group_averaged_rank(values: np.ndarray, gid: np.ndarray, offs: np.ndarray) -> np.ndarray:
|
||||
"""Averaged (tie-corrected) rank of ``values`` within each group, vectorized.
|
||||
|
||||
``gid`` maps each row to its group id; ``offs`` holds the cumulative row
|
||||
offsets so that group ``i`` occupies rows ``[offs[i], offs[i+1])``. Returns
|
||||
the same result as ``pandas.Series.rank(method='average')`` applied per
|
||||
group, but in one pass (``np.lexsort`` is the only non-linear step).
|
||||
"""
|
||||
n = len(values)
|
||||
order = np.lexsort((values, gid))
|
||||
ord_rank = np.empty(n, dtype=np.float64)
|
||||
ord_rank[order] = np.arange(n, dtype=np.float64) - offs[gid[order]] + 1.0
|
||||
sg = gid[order]
|
||||
sv = values[order]
|
||||
newblock = np.empty(n, dtype=bool)
|
||||
newblock[0] = True
|
||||
newblock[1:] = (sg[1:] != sg[:-1]) | (sv[1:] != sv[:-1])
|
||||
blockid = np.cumsum(newblock) - 1
|
||||
block_mean = np.bincount(blockid, weights=ord_rank[order]) / np.bincount(blockid)
|
||||
out = np.empty(n)
|
||||
out[order] = block_mean[blockid]
|
||||
return out
|
||||
|
||||
|
||||
def _per_day_spearman(preds: np.ndarray, labels: np.ndarray, group: np.ndarray) -> float:
|
||||
"""Mean per-day Spearman rank correlation of preds vs labels.
|
||||
|
||||
``group`` holds the number of rows of each trading day (query group), in
|
||||
order. Days with <3 valid rows or a constant pred/label are skipped.
|
||||
|
||||
Vectorized: per-day Spearman == Pearson of the per-day rank transforms,
|
||||
and the Pearson moments (``sum``, ``sum`` of products/squares) aggregate
|
||||
over each day with ``np.bincount``. Runs ~10x faster than the per-day
|
||||
``pd.Series.rank()`` loop that preceded it — this feval is invoked on the
|
||||
train and valid panels every boosting round, per seed.
|
||||
"""
|
||||
if group is None or len(group) == 0:
|
||||
return 0.0
|
||||
offs = np.concatenate([[0], np.cumsum(group.astype(int))])
|
||||
vals = []
|
||||
for i in range(len(group)):
|
||||
s = slice(offs[i], offs[i + 1])
|
||||
p, l = preds[s], labels[s]
|
||||
if len(p) < 3 or np.std(p) == 0 or np.std(l) == 0:
|
||||
continue
|
||||
vals.append(np.corrcoef(pd.Series(p).rank(), pd.Series(l).rank())[0, 1])
|
||||
return float(np.mean(vals)) if vals else 0.0
|
||||
gid = np.repeat(np.arange(len(group)), group.astype(int))
|
||||
rp = _group_averaged_rank(preds, gid, offs)
|
||||
rl = _group_averaged_rank(labels, gid, offs)
|
||||
n_g = group.astype(float)
|
||||
s_p = np.bincount(gid, weights=rp)
|
||||
s_l = np.bincount(gid, weights=rl)
|
||||
s_pl = np.bincount(gid, weights=rp * rl)
|
||||
s_pp = np.bincount(gid, weights=rp * rp)
|
||||
s_ll = np.bincount(gid, weights=rl * rl)
|
||||
cov = n_g * s_pl - s_p * s_l
|
||||
var_p = n_g * s_pp - s_p ** 2
|
||||
var_l = n_g * s_ll - s_l ** 2
|
||||
denom = np.sqrt(var_p * var_l)
|
||||
valid = (n_g >= 3) & (denom > 0)
|
||||
corr = np.where(valid, cov / np.where(denom == 0, 1, denom), 0.0)
|
||||
return float(corr[valid].mean()) if valid.any() else 0.0
|
||||
|
||||
|
||||
def rankic_feval(preds, dataset):
|
||||
|
||||
@@ -1,3 +1,11 @@
|
||||
from .ic_gate import ICGateTopkDropoutStrategy # noqa: F401
|
||||
from .optimal_stop import OptimalStopControl # noqa: F401
|
||||
from .regime_gate import RegimeGateTopkDropoutStrategy # noqa: F401
|
||||
from .weekly_rebalance import WeeklyRebalanceDropoutStrategy # noqa: F401
|
||||
|
||||
__all__ = ["OptimalStopControl"]
|
||||
__all__ = [
|
||||
"ICGateTopkDropoutStrategy",
|
||||
"OptimalStopControl",
|
||||
"RegimeGateTopkDropoutStrategy",
|
||||
"WeeklyRebalanceDropoutStrategy",
|
||||
]
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
BIN
Binary file not shown.
Binary file not shown.
@@ -1,138 +0,0 @@
|
||||
"""TopkDropout with HMM high-volatility + drawdown-pause risk gates.
|
||||
|
||||
Gates NEW entries on two risk conditions (held names are never force-sold):
|
||||
|
||||
1. **HMM high-vol pause**: when the cross-sectional mean of ``sp_hmm_p_regime1``
|
||||
(HMM high-vol regime probability) on the signal date is >= ``hmm_pause_pct``,
|
||||
new buys are paused. The time-series study showed HMM high-vol probability
|
||||
pulses BEFORE sharp moves (regime-change cut) — pausing new exposure at the
|
||||
boundary reduces drawdown from price over-reaction.
|
||||
2. **Drawdown pause**: when the account equity drawdown from its running peak
|
||||
exceeds ``drawdown_pause_pct``, new buys are paused. This is the
|
||||
``drawdown_pause_pct`` risk-limit expressed inside the backtest (the pure
|
||||
executor-side gate is documented as not expressible in a one-shot backtest).
|
||||
3. **Liquidity floor**: names whose 20-day average daily dollar volume is below
|
||||
``liquidity_floor_adv`` are dropped from BUY candidates (the proven mitigant
|
||||
from exp-18: $5M floor cut drawdown 7.9%->5.4% at higher IR).
|
||||
|
||||
Implementation: pre-filter the signal score before the base TopkDropout
|
||||
decision — non-held names get score 0 when any gate fires.
|
||||
|
||||
Wired into a workflow yaml like:
|
||||
|
||||
strategy:
|
||||
class: HmmRiskTopk
|
||||
module_path: tac_qlib.contrib.strategy.hmm_risk
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 10
|
||||
n_drop: 2
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
hmm_pause_pct: 0.70
|
||||
drawdown_pause_pct: 8.0
|
||||
liquidity_floor_adv: 5000000
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
from typing import Dict
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from qlib.backtest.decision import TradeDecisionWO
|
||||
from qlib.backtest.position import Position
|
||||
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
|
||||
|
||||
__all__ = ["HmmRiskTopk"]
|
||||
|
||||
|
||||
class HmmRiskTopk(TopkDropoutStrategy):
|
||||
"""TopkDropoutStrategy with HMM high-vol pause + drawdown pause + liquidity floor."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
hmm_pause_pct: float = 0.70,
|
||||
drawdown_pause_pct: float = 8.0,
|
||||
liquidity_floor_adv: float = 0.0,
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__(**kwargs)
|
||||
self.hmm_pause_pct = float(hmm_pause_pct)
|
||||
self.drawdown_pause_pct = float(drawdown_pause_pct)
|
||||
self.liquidity_floor_adv = float(liquidity_floor_adv)
|
||||
self._peak_equity = 0.0
|
||||
|
||||
# ------------------------------------------------------------- gates
|
||||
def _hmm_high_vol(self, pred_date) -> bool:
|
||||
"""Cross-sectional mean HMM high-vol regime probability >= threshold."""
|
||||
try:
|
||||
from qlib.data import D
|
||||
|
||||
feat = D.features(D.instruments("all"), ["$sp_hmm_p_regime1"],
|
||||
start_time=pred_date, end_time=pred_date)
|
||||
if feat is None or len(feat) == 0:
|
||||
return False
|
||||
p = feat["$sp_hmm_p_regime1"].dropna()
|
||||
if len(p) == 0:
|
||||
return False
|
||||
return float(p.mean()) >= self.hmm_pause_pct
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
def _drawdown_active(self, equity: float) -> bool:
|
||||
if self.drawdown_pause_pct <= 0:
|
||||
return False
|
||||
self._peak_equity = max(self._peak_equity, equity)
|
||||
if self._peak_equity <= 0:
|
||||
return False
|
||||
dd = (self._peak_equity - equity) / self._peak_equity * 100.0
|
||||
return dd >= self.drawdown_pause_pct
|
||||
|
||||
def _illiquid(self, codes, asof) -> Dict[str, bool]:
|
||||
if self.liquidity_floor_adv <= 0 or not codes:
|
||||
return {}
|
||||
from tac_qlib.risk_limits import dollar_adv
|
||||
|
||||
adv = dollar_adv(codes, market="US", asof=asof, lookback=20)
|
||||
return {c: adv.get(str(c).upper(), 0.0) < self.liquidity_floor_adv for c in codes}
|
||||
|
||||
# ------------------------------------------------------------- decision
|
||||
def generate_trade_decision(self, execute_result=None):
|
||||
trade_step = self.trade_calendar.get_trade_step()
|
||||
trade_start_time, trade_end_time = self.trade_calendar.get_step_time(trade_step)
|
||||
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(trade_step, shift=1)
|
||||
pred_score = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
|
||||
if pred_score is None:
|
||||
return TradeDecisionWO([], self)
|
||||
if isinstance(pred_score, pd.DataFrame):
|
||||
pred_score = pred_score.iloc[:, 0]
|
||||
|
||||
current_temp = copy.deepcopy(self.trade_position)
|
||||
assert isinstance(current_temp, Position)
|
||||
held = {c for c in current_temp.get_stock_list() if abs(current_temp.get_stock_amount(c)) > 1e-6}
|
||||
|
||||
equity = current_temp.get_cash()
|
||||
for code in held:
|
||||
mark = self.trade_exchange.get_deal_price(
|
||||
stock_id=code, start_time=trade_start_time, end_time=trade_end_time, direction=1
|
||||
)
|
||||
if mark is not None and np.isfinite(mark):
|
||||
equity += abs(current_temp.get_stock_amount(code)) * mark
|
||||
|
||||
hmm_pause = self._hmm_high_vol(str(pd.Timestamp(pred_start_time).date()))
|
||||
dd_pause = self._drawdown_active(equity)
|
||||
buys_paused = hmm_pause or dd_pause
|
||||
|
||||
pred_score = pred_score.copy()
|
||||
if buys_paused or self.liquidity_floor_adv > 0:
|
||||
new_codes = [c for c in pred_score.index if c not in held]
|
||||
illiquid = self._illiquid(new_codes, str(pd.Timestamp(pred_start_time).date()))
|
||||
for code in new_codes:
|
||||
if buys_paused or illiquid.get(code, False):
|
||||
pred_score[code] = -1e9 # cannot enter today
|
||||
|
||||
return super().generate_trade_decision(execute_result)
|
||||
@@ -0,0 +1,117 @@
|
||||
"""Realized-IC circuit breaker TopkDropout strategy.
|
||||
|
||||
Subclass of ``qlib.contrib.strategy.signal_strategy.TopkDropoutStrategy`` that
|
||||
holds the book (issues NO orders) while the streaming realized RankIC of the
|
||||
deployed signal is below threshold — i.e. the model's cross-sectional
|
||||
predictions are no longer earning against realized forward returns. When the
|
||||
gate is open it behaves exactly like the reference TopkDropoutStrategy.
|
||||
|
||||
The gate is evaluated per trade step on the trailing mean realized RankIC of
|
||||
the signal over the last ``ic_window`` trading days whose label is fully
|
||||
realized as of the decision date (no lookahead — a 5d fwd label ``close[t+6]/
|
||||
close[t+1]-1`` is only known at ``t+6``).
|
||||
|
||||
Two wiring modes:
|
||||
|
||||
* ``ic_gate``: a precomputed ``pd.Series`` indexed by datetime of booleans
|
||||
(True = gate open / trade allowed). Computed once by the caller (e.g.
|
||||
``rd_backtest``) and looked up per step. Missing dates default to open.
|
||||
* realized-IC self-computation: when ``ic_min_rankic`` is given but no
|
||||
``ic_gate``, the strategy computes the per-date realized RankIC itself from
|
||||
``self.signal`` (the pred scores) and the lake 1d bars via
|
||||
``tac_qlib.risk_limits.realized_rankic_series``, then applies the same
|
||||
trailing-window comparison. Works when instantiated from a workflow YAML
|
||||
PortAnaRecord config (``lake_root`` / ``market`` must be provided).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from qlib.backtest.decision import TradeDecisionWO
|
||||
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
|
||||
|
||||
from tac_qlib.risk_limits import ic_circuit_breaker, realized_rankic_series
|
||||
|
||||
__all__ = ["ICGateTopkDropoutStrategy"]
|
||||
|
||||
|
||||
class ICGateTopkDropoutStrategy(TopkDropoutStrategy):
|
||||
"""TopkDropout with a streaming realized-IC circuit breaker.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
topk, n_drop, method_sell, method_buy, hold_thresh, only_tradable,
|
||||
forbid_all_trade_at_limit : same as ``TopkDropoutStrategy``.
|
||||
ic_min_rankic : float — pause new trading while trailing realized RankIC is
|
||||
below this threshold (0 disables the gate).
|
||||
ic_window : int — trailing window for the realized RankIC mean (default 22).
|
||||
ic_label_horizon : int — label horizon in trading days (default 6).
|
||||
ic_min_obs : int — min realized labels before the gate arms (default 10).
|
||||
ic_gate : pd.Series, optional — precomputed per-date gate (bool indexed by
|
||||
datetime). When provided, it overrides self-computation.
|
||||
lake_root, market : str — lake location for self-computed realized IC.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
topk,
|
||||
n_drop,
|
||||
ic_min_rankic: float = 0.0,
|
||||
ic_window: int = 22,
|
||||
ic_label_horizon: int = 6,
|
||||
ic_min_obs: int = 10,
|
||||
ic_gate=None,
|
||||
lake_root: str = "",
|
||||
market: str = "US",
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__(topk=topk, n_drop=n_drop, **kwargs)
|
||||
self.ic_min_rankic = float(ic_min_rankic or 0.0)
|
||||
self.ic_window = int(ic_window or 22)
|
||||
self.ic_label_horizon = int(ic_label_horizon or 6)
|
||||
self.ic_min_obs = int(ic_min_obs or 10)
|
||||
self._ic_gate = ic_gate
|
||||
self._realized_ic = None
|
||||
self.lake_root = lake_root or ""
|
||||
self.market = market or "US"
|
||||
|
||||
def _load_realized_ic(self):
|
||||
if self._realized_ic is None:
|
||||
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(
|
||||
self.trade_calendar.get_trade_step(), shift=-self.ic_label_horizon
|
||||
)
|
||||
pred = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
|
||||
if isinstance(pred, pd.DataFrame):
|
||||
pred = pred.iloc[:, 0]
|
||||
self._realized_ic = realized_rankic_series(
|
||||
pred, self.lake_root, self.market, label_horizon=self.ic_label_horizon
|
||||
)
|
||||
return self._realized_ic
|
||||
|
||||
def _gate_open(self, trade_start_time) -> bool:
|
||||
ts = pd.Timestamp(trade_start_time)
|
||||
if self._ic_gate is not None:
|
||||
# precomputed gate series: look up the latest known decision date <= ts
|
||||
known = self._ic_gate[self._ic_gate.index <= ts]
|
||||
if len(known):
|
||||
return bool(known.iloc[-1])
|
||||
return True
|
||||
if self.ic_min_rankic <= 0:
|
||||
return True
|
||||
realized = self._load_realized_ic()
|
||||
limits = {
|
||||
"ic_min_rankic": self.ic_min_rankic,
|
||||
"ic_window": self.ic_window,
|
||||
"ic_min_obs": self.ic_min_obs,
|
||||
}
|
||||
tripped, _reason, _trail = ic_circuit_breaker(realized, ts, limits)
|
||||
return not tripped
|
||||
|
||||
def generate_trade_decision(self, execute_result=None):
|
||||
trade_step = self.trade_calendar.get_trade_step()
|
||||
trade_start_time, _ = self.trade_calendar.get_step_time(trade_step)
|
||||
if not self._gate_open(trade_start_time):
|
||||
return TradeDecisionWO([], self)
|
||||
return super().generate_trade_decision(execute_result)
|
||||
@@ -1,91 +0,0 @@
|
||||
"""TopkDropout with a 1-day momentum entry-confirmation gate.
|
||||
|
||||
Gates NEW entries on short-term momentum: a name that is not currently held
|
||||
may only be bought when its trailing 1-day return is above ``min_momentum``
|
||||
(Lag-1 autocorr ~ +0.45 in the time-series study => short-term momentum
|
||||
continuation). Held names are never force-sold by this gate — exits stay the
|
||||
pure TopkDropout rule.
|
||||
|
||||
Implementation: override ``generate_trade_decision`` and zero out the signal
|
||||
score of any non-held name that fails the momentum check BEFORE calling the
|
||||
base TopkDropout decision, so it can never be selected as a buy candidate.
|
||||
This is a clean pre-filter: the rest of the strategy (top-k, n_drop, sizing,
|
||||
costs) is untouched.
|
||||
|
||||
Wired into a workflow yaml like:
|
||||
|
||||
strategy:
|
||||
class: MomentumGateTopk
|
||||
module_path: tac_qlib.contrib.strategy.momentum_gate
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 10
|
||||
n_drop: 2
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
min_momentum: 0.0
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from qlib.backtest.decision import TradeDecisionWO
|
||||
from qlib.backtest.position import Position
|
||||
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
|
||||
|
||||
__all__ = ["MomentumGateTopk"]
|
||||
|
||||
|
||||
class MomentumGateTopk(TopkDropoutStrategy):
|
||||
"""TopkDropoutStrategy gated on 1-day momentum for new entries."""
|
||||
|
||||
def __init__(self, *, min_momentum: float = 0.0, **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
self.min_momentum = float(min_momentum)
|
||||
|
||||
def _momentum_ok(self, code, trade_start, trade_end) -> bool:
|
||||
"""True when the trailing 1-day return is above the momentum floor."""
|
||||
try:
|
||||
cur = self.trade_exchange.get_deal_price(
|
||||
stock_id=code, start_time=trade_start, end_time=trade_end, direction=1
|
||||
)
|
||||
except Exception:
|
||||
return False
|
||||
if cur is None or cur != cur or cur <= 0:
|
||||
return False
|
||||
prev_start = trade_start - pd.Timedelta(days=5)
|
||||
prev_end = trade_start - pd.Timedelta(seconds=1)
|
||||
prev = self.trade_exchange.get_deal_price(
|
||||
stock_id=code, start_time=prev_start, end_time=prev_end, direction=0
|
||||
)
|
||||
if prev is None or prev != prev or prev <= 0:
|
||||
return False
|
||||
return (cur / prev - 1.0) >= self.min_momentum
|
||||
|
||||
def generate_trade_decision(self, execute_result=None):
|
||||
trade_step = self.trade_calendar.get_trade_step()
|
||||
trade_start_time, trade_end_time = self.trade_calendar.get_step_time(trade_step)
|
||||
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(trade_step, shift=1)
|
||||
pred_score = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
|
||||
if pred_score is None:
|
||||
return TradeDecisionWO([], self)
|
||||
if isinstance(pred_score, pd.DataFrame):
|
||||
pred_score = pred_score.iloc[:, 0]
|
||||
|
||||
current_temp = copy.deepcopy(self.trade_position)
|
||||
assert isinstance(current_temp, Position)
|
||||
held = set(current_temp.get_stock_list())
|
||||
held = {c for c in held if abs(current_temp.get_stock_amount(c)) > 1e-6}
|
||||
|
||||
# pre-filter: zero the score of non-held names that fail momentum
|
||||
pred_score = pred_score.copy()
|
||||
for code in pred_score.index:
|
||||
if code in held:
|
||||
continue # never gate exits / re-balancing of held names
|
||||
if not self._momentum_ok(code, trade_start_time, trade_end_time):
|
||||
pred_score[code] = -1e9 # cannot enter today
|
||||
|
||||
return super().generate_trade_decision(execute_result)
|
||||
@@ -0,0 +1,215 @@
|
||||
"""Regime-gate TopkDropout strategy.
|
||||
|
||||
Subclass of ``qlib.contrib.strategy.signal_strategy.TopkDropoutStrategy`` that
|
||||
holds the book (issues NO orders) while a regime detector says the market is in
|
||||
an unfavorable state. When the gate is open it behaves exactly like the
|
||||
reference TopkDropoutStrategy.
|
||||
|
||||
Three detector types are supported (all causal — no lookahead):
|
||||
|
||||
* ``dispersion``: cross-sectional standard deviation of 22-day rolling returns
|
||||
across the universe. Gate closes when CS dispersion < threshold (low
|
||||
dispersion means the spread between winners and losers is too narrow for
|
||||
TopkDropout to exploit).
|
||||
* ``vol``: cross-sectional mean of 22-day rolling realized volatility. Gate
|
||||
closes when avg vol is outside a band ``[vol_low, vol_high]`` (strategy
|
||||
needs moderate vol — too calm or too turbulent both hurt).
|
||||
* ``hmm``: pre-computed HMM posterior for regime 1 (``sp_hmm_p_regime1``).
|
||||
Gate closes when posterior < threshold (model is not confident the calm
|
||||
regime is active).
|
||||
|
||||
The gate is provided as a precomputed ``pd.Series`` of booleans indexed by
|
||||
datetime (True = trade allowed). The companion ``compute_regime_gate``
|
||||
function builds this series from lake bars; call it once before backtesting
|
||||
and pass the result as the ``regime_gate`` parameter.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from qlib.backtest.decision import TradeDecisionWO
|
||||
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
|
||||
|
||||
__all__ = ["RegimeGateTopkDropoutStrategy", "compute_regime_gate"]
|
||||
|
||||
|
||||
class RegimeGateTopkDropoutStrategy(TopkDropoutStrategy):
|
||||
"""TopkDropout with a regime-gate circuit breaker.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
topk, n_drop, method_sell, method_buy, hold_thresh, only_tradable,
|
||||
forbid_all_trade_at_limit : same as ``TopkDropoutStrategy``.
|
||||
regime_gate : pd.Series — precomputed per-date gate (bool indexed by
|
||||
datetime). True = trade allowed, False = no orders. Missing dates
|
||||
default to open (trade allowed).
|
||||
"""
|
||||
|
||||
def __init__(self, *, regime_gate=None, **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
self._regime_gate = regime_gate
|
||||
|
||||
def _gate_open(self, trade_start_time) -> bool:
|
||||
if self._regime_gate is None:
|
||||
return True
|
||||
ts = pd.Timestamp(trade_start_time)
|
||||
known = self._regime_gate[self._regime_gate.index <= ts]
|
||||
if len(known):
|
||||
return bool(known.iloc[-1])
|
||||
return True # default open if no history yet
|
||||
|
||||
def generate_trade_decision(self, execute_result=None):
|
||||
trade_step = self.trade_calendar.get_trade_step()
|
||||
trade_start_time, _ = self.trade_calendar.get_step_time(trade_step)
|
||||
if not self._gate_open(trade_start_time):
|
||||
return TradeDecisionWO([], self)
|
||||
return super().generate_trade_decision(execute_result)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Precomputation helper
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def compute_regime_gate(
|
||||
detector: str,
|
||||
threshold: float = 0.0,
|
||||
*,
|
||||
lake_root: str = "",
|
||||
market: str = "US",
|
||||
start: str = "2015-01-03",
|
||||
end: str = "2026-08-19",
|
||||
vol_low: float = 0.0,
|
||||
vol_high: float = 999.0,
|
||||
hmm_field: str = "sp_hmm_p_regime1",
|
||||
) -> pd.Series:
|
||||
"""Build a per-date regime gate series from lake bars.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
detector : str — ``"dispersion"``, ``"vol"``, or ``"hmm"``.
|
||||
threshold : float — for ``dispersion``: min CS dispersion to allow trading.
|
||||
For ``hmm``: min HMM posterior to allow trading.
|
||||
Ignored for ``vol`` (uses ``vol_low``/``vol_high`` band instead).
|
||||
lake_root, market : str — lake location.
|
||||
start, end : str — date window.
|
||||
vol_low, vol_high : float — annualized vol band for the ``vol`` detector.
|
||||
hmm_field : str — HMM feature column name for the ``hmm`` detector.
|
||||
|
||||
Returns
|
||||
-------
|
||||
pd.Series — bool, indexed by datetime. True = trade allowed.
|
||||
"""
|
||||
from tac_qlib.data.config import LakeConfig, resolve_lake_root
|
||||
|
||||
cfg = LakeConfig(resolve_lake_root(lake_root or None), market)
|
||||
symbols = _universe_symbols(cfg)
|
||||
close_df, vol_df = _load_daily_bars(symbols, cfg, start, end)
|
||||
if close_df.empty:
|
||||
return pd.Series(dtype=bool)
|
||||
|
||||
if detector == "dispersion":
|
||||
return _dispersion_gate(close_df, threshold)
|
||||
elif detector == "vol":
|
||||
return _vol_gate(close_df, vol_low, vol_high)
|
||||
elif detector == "hmm":
|
||||
return _hmm_gate(cfg, symbols, threshold, start, end, hmm_field)
|
||||
else:
|
||||
raise ValueError(f"Unknown detector: {detector!r}")
|
||||
|
||||
|
||||
def _universe_symbols(cfg) -> list:
|
||||
"""Read symbols from the lake symbols.parquet."""
|
||||
import pathlib
|
||||
|
||||
sp = cfg.lake_root / "symbols.parquet"
|
||||
if sp.exists():
|
||||
df = pd.read_parquet(sp)
|
||||
col = "symbol" if "symbol" in df.columns else df.columns[0]
|
||||
return sorted(df[col].astype(str).str.upper().tolist())
|
||||
return []
|
||||
|
||||
|
||||
def _load_daily_bars(symbols, cfg, start, end):
|
||||
"""Load daily close prices for all symbols into a wide DataFrame."""
|
||||
closes = {}
|
||||
vols = {}
|
||||
for sym in symbols:
|
||||
p = cfg.bar_path("1d", sym)
|
||||
if not p.exists():
|
||||
continue
|
||||
try:
|
||||
df = pd.read_parquet(p)
|
||||
except Exception:
|
||||
continue
|
||||
if not len(df):
|
||||
continue
|
||||
tcol = df["t"] if "t" in df.columns else df["date"]
|
||||
ts = pd.to_datetime(tcol)
|
||||
df = df.assign(_t=ts).set_index("_t").sort_index()
|
||||
df = df.loc[start:end]
|
||||
if len(df) < 22:
|
||||
continue
|
||||
closes[sym] = df["c"]
|
||||
if "v" in df.columns:
|
||||
vols[sym] = df["v"]
|
||||
close_df = pd.DataFrame(closes)
|
||||
vol_df = pd.DataFrame(vols) if vols else None
|
||||
return close_df, vol_df
|
||||
|
||||
|
||||
def _dispersion_gate(close_df, threshold):
|
||||
"""Cross-sectional dispersion of 22-day rolling returns."""
|
||||
if close_df.empty or close_df.shape[1] < 2:
|
||||
return pd.Series(dtype=bool)
|
||||
ret = close_df.pct_change(22)
|
||||
cs_disp = ret.std(axis=1)
|
||||
gate = cs_disp >= threshold
|
||||
gate.iloc[:22] = True # warmup: allow trading
|
||||
return gate
|
||||
|
||||
|
||||
def _vol_gate(close_df, vol_low, vol_high):
|
||||
"""Cross-sectional mean of 22-day rolling realized vol."""
|
||||
if close_df.empty or close_df.shape[1] < 2:
|
||||
return pd.Series(dtype=bool)
|
||||
import numpy as np
|
||||
log_ret = np.log(close_df / close_df.shift(1))
|
||||
rv22 = log_ret.rolling(22).std() * (252 ** 0.5)
|
||||
cs_mean_vol = rv22.mean(axis=1)
|
||||
gate = (cs_mean_vol >= vol_low) & (cs_mean_vol <= vol_high)
|
||||
gate.iloc[:22] = True # warmup
|
||||
return gate
|
||||
|
||||
|
||||
def _hmm_gate(cfg, symbols, threshold, start, end, hmm_field):
|
||||
"""HMM regime posterior gate from persisted SP features."""
|
||||
feat_root = cfg.lake_root / "features"
|
||||
all_posteriors = {}
|
||||
for sym in symbols:
|
||||
# check both ta and sp family paths
|
||||
for family in ("sp", "ta"):
|
||||
p = feat_root / f"market=US" / f"timeframe=1d" / f"family={family}" / f"symbol={sym}.parquet"
|
||||
if not p.exists():
|
||||
continue
|
||||
try:
|
||||
df = pd.read_parquet(p)
|
||||
except Exception:
|
||||
continue
|
||||
if hmm_field not in df.columns:
|
||||
continue
|
||||
tcol = df["t"] if "t" in df.columns else df["date"]
|
||||
ts = pd.to_datetime(tcol)
|
||||
s = pd.Series(df[hmm_field].values, index=ts, name=sym)
|
||||
s = s.loc[start:end].dropna()
|
||||
if len(s) > 0:
|
||||
all_posteriors[sym] = s
|
||||
break
|
||||
if not all_posteriors:
|
||||
# no HMM features found — default open
|
||||
idx = pd.date_range(start, end, freq="B")
|
||||
return pd.Series(True, index=idx)
|
||||
post_df = pd.DataFrame(all_posteriors)
|
||||
cs_mean = post_df.mean(axis=1)
|
||||
gate = cs_mean >= threshold
|
||||
return gate
|
||||
@@ -0,0 +1,301 @@
|
||||
"""Signal-quality gate TopkDropout strategy.
|
||||
|
||||
Subclass of ``qlib.contrib.strategy.signal_strategy.TopkDropoutStrategy`` that
|
||||
holds the book (issues NO orders) when the model's recent prediction accuracy
|
||||
is below a threshold. When the gate is open it behaves exactly like the
|
||||
reference TopkDropoutStrategy.
|
||||
|
||||
Unlike the regime gate (which asks "is the market calm?"), the signal-quality
|
||||
gate asks "are my predictions accurate?" — and works across ALL years.
|
||||
|
||||
The gate is provided as a precomputed ``pd.Series`` of booleans indexed by
|
||||
datetime (True = trade allowed). The companion ``compute_signal_quality_gate``
|
||||
function builds this series from a pred.pkl and lake bars; call it once before
|
||||
backtesting and pass the result as the ``signal_quality_gate`` parameter.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from qlib.backtest.decision import TradeDecisionWO
|
||||
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
|
||||
|
||||
__all__ = ["SignalQualityGateStrategy", "compute_signal_quality_gate"]
|
||||
|
||||
|
||||
class SignalQualityGateStrategy(TopkDropoutStrategy):
|
||||
"""TopkDropout with signal-quality gate overlay.
|
||||
|
||||
When ``lake_root`` is provided the gate is computed on-the-fly from the
|
||||
signal (``<PRED>``) and close prices — no precomputed gate file needed.
|
||||
This ensures the gate matches the model that is actually generating the
|
||||
predictions (critical when the model is retrained each year).
|
||||
|
||||
Parameters
|
||||
----------
|
||||
topk, n_drop, method_sell, method_buy, hold_thresh, only_tradable,
|
||||
forbid_all_trade_at_limit : same as ``TopkDropoutStrategy``.
|
||||
signal_quality_gate : pd.Series — precomputed per-date gate (bool).
|
||||
signal_quality_gate_path : str — path to pickled gate Series.
|
||||
lake_root : str — lake root for on-the-fly gate computation (preferred).
|
||||
gate_topk, gate_lookback, gate_threshold : int/float — gate params.
|
||||
gate_start, gate_end : str — date window for loading close prices.
|
||||
"""
|
||||
|
||||
def __init__(self, *, signal_quality_gate=None, signal_quality_gate_path=None,
|
||||
lake_root=None, gate_topk=10, gate_lookback=5, gate_threshold=0.5,
|
||||
gate_start="2015-01-03", gate_end="2026-08-19", **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
self._sq_gate_computed = False
|
||||
if signal_quality_gate is not None:
|
||||
self._sq_gate = signal_quality_gate
|
||||
self._sq_gate_computed = True
|
||||
elif signal_quality_gate_path is not None:
|
||||
import pickle
|
||||
with open(signal_quality_gate_path, "rb") as f:
|
||||
self._sq_gate = pickle.load(f)
|
||||
self._sq_gate_computed = True
|
||||
elif lake_root is not None:
|
||||
self._sq_gate = None
|
||||
self._lake_root = lake_root
|
||||
self._gate_topk = gate_topk
|
||||
self._gate_lookback = gate_lookback
|
||||
self._gate_threshold = gate_threshold
|
||||
self._gate_start = gate_start
|
||||
self._gate_end = gate_end
|
||||
else:
|
||||
self._sq_gate = None
|
||||
|
||||
def _compute_gate_on_fly(self):
|
||||
"""Compute gate from the signal (pred.pkl) and lake close prices."""
|
||||
import pickle
|
||||
from pathlib import Path
|
||||
|
||||
signal_path = self._signal
|
||||
if not Path(signal_path).exists():
|
||||
return
|
||||
|
||||
with open(signal_path, "rb") as f:
|
||||
pred = pickle.load(f)
|
||||
|
||||
# Handle MultiIndex DataFrame -> unstack to wide
|
||||
if isinstance(pred, pd.DataFrame) and isinstance(pred.index, pd.MultiIndex):
|
||||
pred = pred.iloc[:, 0]
|
||||
pred.index = pd.MultiIndex.from_arrays([
|
||||
pd.to_datetime(pred.index.get_level_values(0)).normalize(),
|
||||
pred.index.get_level_values(1)
|
||||
])
|
||||
pred = pred.unstack(level=1)
|
||||
elif isinstance(pred, pd.DataFrame):
|
||||
pred = pred.iloc[:, 0] if pred.shape[1] >= 1 else pred.squeeze()
|
||||
pred.index = pd.to_datetime(pred.index).normalize()
|
||||
|
||||
# Load close prices from lake
|
||||
close_df = _load_close_prices(self._lake_root, "US", self._gate_start, self._gate_end)
|
||||
if close_df.empty:
|
||||
return
|
||||
|
||||
ret_df = close_df.pct_change()
|
||||
ret_df.index = pd.to_datetime(ret_df.index).normalize()
|
||||
|
||||
pred_dates = sorted(pred.index.unique())
|
||||
if len(pred_dates) < 2:
|
||||
self._sq_gate = pd.Series(True, index=pd.DatetimeIndex(pred_dates))
|
||||
self._sq_gate_computed = True
|
||||
return
|
||||
|
||||
hit_rates = {}
|
||||
for i in range(1, len(pred_dates)):
|
||||
day = pred_dates[i]
|
||||
prev_day = pred_dates[i - 1]
|
||||
try:
|
||||
prev_scores = pred.loc[prev_day]
|
||||
except KeyError:
|
||||
continue
|
||||
if isinstance(prev_scores, pd.DataFrame):
|
||||
prev_scores = prev_scores.iloc[:, 0]
|
||||
prev_scores = prev_scores.dropna().sort_values(ascending=False)
|
||||
topk_syms = list(prev_scores.index[:self._gate_topk])
|
||||
|
||||
if day not in ret_df.index:
|
||||
continue
|
||||
today_ret = ret_df.loc[day]
|
||||
topk_rets = today_ret.reindex(topk_syms).dropna()
|
||||
if len(topk_rets) == 0:
|
||||
continue
|
||||
|
||||
hit_rates[day] = (topk_rets > 0).sum() / len(topk_rets)
|
||||
|
||||
if not hit_rates:
|
||||
self._sq_gate_computed = True
|
||||
return
|
||||
|
||||
hr_series = pd.Series(hit_rates).sort_index()
|
||||
rolling_hr = hr_series.rolling(self._gate_lookback, min_periods=1).mean()
|
||||
gate = rolling_hr >= self._gate_threshold
|
||||
gate.iloc[:self._gate_lookback] = True
|
||||
|
||||
self._sq_gate = gate
|
||||
self._sq_gate_computed = True
|
||||
|
||||
def _gate_open(self, trade_start_time) -> bool:
|
||||
if self._sq_gate is None:
|
||||
return True
|
||||
ts = pd.Timestamp(trade_start_time)
|
||||
known = self._sq_gate[self._sq_gate.index <= ts]
|
||||
if len(known):
|
||||
return bool(known.iloc[-1])
|
||||
return True # default open if no history yet
|
||||
|
||||
def generate_trade_decision(self, execute_result=None):
|
||||
if not self._sq_gate_computed:
|
||||
self._compute_gate_on_fly()
|
||||
trade_step = self.trade_calendar.get_trade_step()
|
||||
trade_start_time, _ = self.trade_calendar.get_step_time(trade_step)
|
||||
if not self._gate_open(trade_start_time):
|
||||
return TradeDecisionWO([], self)
|
||||
return super().generate_trade_decision(execute_result)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Precomputation helper
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def compute_signal_quality_gate(
|
||||
pred_path: str,
|
||||
*,
|
||||
lake_root: str = "",
|
||||
market: str = "US",
|
||||
topk: int = 10,
|
||||
lookback: int = 5,
|
||||
threshold: float = 0.5,
|
||||
start: str = "2015-01-03",
|
||||
end: str = "2026-08-19",
|
||||
) -> pd.Series:
|
||||
"""Build a per-date signal-quality gate series from a pred.pkl and lake bars.
|
||||
|
||||
For each day, checks whether the model's topk picks from the previous day
|
||||
had positive returns. Computes a rolling hit rate over ``lookback`` days
|
||||
and opens the gate when hit rate >= ``threshold``.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
pred_path : str — path to pred.pkl (from rd_train / rd_predict).
|
||||
lake_root, market : str — lake location (for loading close prices).
|
||||
topk : int — number of top picks to track for hit rate.
|
||||
lookback : int — rolling window for hit rate computation.
|
||||
threshold : float — hit rate threshold to keep trading.
|
||||
start, end : str — date window for loading prices.
|
||||
|
||||
Returns
|
||||
-------
|
||||
pd.Series — bool, indexed by datetime. True = trade allowed.
|
||||
"""
|
||||
from pathlib import Path
|
||||
import pickle
|
||||
|
||||
# Load pred.pkl
|
||||
with open(pred_path, "rb") as f:
|
||||
pred = pickle.load(f)
|
||||
|
||||
# Handle MultiIndex DataFrame (datetime, instrument) -> unstack to wide
|
||||
if isinstance(pred, pd.DataFrame) and isinstance(pred.index, pd.MultiIndex):
|
||||
pred = pred.iloc[:, 0] # take score column as Series
|
||||
pred.index = pd.MultiIndex.from_arrays([
|
||||
pd.to_datetime(pred.index.get_level_values(0)).normalize(),
|
||||
pred.index.get_level_values(1)
|
||||
])
|
||||
# Unstack to wide: dates x instruments
|
||||
pred = pred.unstack(level=1)
|
||||
elif isinstance(pred, pd.DataFrame):
|
||||
pred = pred.iloc[:, 0] if pred.shape[1] >= 1 else pred.squeeze()
|
||||
pred.index = pd.to_datetime(pred.index).normalize()
|
||||
|
||||
# Load close prices from lake
|
||||
close_df = _load_close_prices(lake_root, market, start, end)
|
||||
if close_df.empty:
|
||||
return pd.Series(dtype=bool)
|
||||
|
||||
ret_df = close_df.pct_change()
|
||||
ret_df.index = pd.to_datetime(ret_df.index).normalize()
|
||||
|
||||
# Get sorted unique prediction dates
|
||||
pred_dates = sorted(pred.index.unique())
|
||||
if len(pred_dates) < 2:
|
||||
return pd.Series(True, index=pd.DatetimeIndex(pred_dates))
|
||||
|
||||
# Compute hit rates
|
||||
hit_rates = {}
|
||||
for i in range(1, len(pred_dates)):
|
||||
day = pred_dates[i]
|
||||
# Get yesterday's topk
|
||||
prev_day = pred_dates[i - 1]
|
||||
try:
|
||||
prev_scores = pred.loc[prev_day]
|
||||
except KeyError:
|
||||
continue
|
||||
if isinstance(prev_scores, pd.DataFrame):
|
||||
prev_scores = prev_scores.iloc[:, 0]
|
||||
prev_scores = prev_scores.dropna().sort_values(ascending=False)
|
||||
topk_syms = list(prev_scores.index[:topk])
|
||||
|
||||
# Get today's returns
|
||||
if day not in ret_df.index:
|
||||
continue
|
||||
today_ret = ret_df.loc[day]
|
||||
topk_rets = today_ret.reindex(topk_syms).dropna()
|
||||
if len(topk_rets) == 0:
|
||||
continue
|
||||
|
||||
hit_rates[day] = (topk_rets > 0).sum() / len(topk_rets)
|
||||
|
||||
if not hit_rates:
|
||||
return pd.Series(dtype=bool)
|
||||
|
||||
hr_series = pd.Series(hit_rates).sort_index()
|
||||
|
||||
# Rolling hit rate
|
||||
rolling_hr = hr_series.rolling(lookback, min_periods=1).mean()
|
||||
|
||||
# Gate is open when rolling hit rate >= threshold
|
||||
gate = rolling_hr >= threshold
|
||||
gate.iloc[:lookback] = True # warmup: allow trading
|
||||
|
||||
return gate
|
||||
|
||||
|
||||
def _load_close_prices(lake_root, market, start, end):
|
||||
"""Load daily close prices for all symbols into a wide DataFrame."""
|
||||
from pathlib import Path
|
||||
|
||||
lake = Path(lake_root)
|
||||
symbols_parquet = lake / "symbols.parquet"
|
||||
if not symbols_parquet.exists():
|
||||
return pd.DataFrame()
|
||||
|
||||
df = pd.read_parquet(symbols_parquet)
|
||||
col = "symbol" if "symbol" in df.columns else df.columns[0]
|
||||
symbols = sorted(df[col].astype(str).str.upper().tolist())
|
||||
|
||||
closes = {}
|
||||
for sym in symbols:
|
||||
p = lake / "market=US" / "timeframe=1d" / f"symbol={sym}.parquet"
|
||||
if not p.exists():
|
||||
continue
|
||||
try:
|
||||
bar = pd.read_parquet(p)
|
||||
except Exception:
|
||||
continue
|
||||
if not len(bar):
|
||||
continue
|
||||
tcol = "t" if "t" in bar.columns else "date"
|
||||
ts = pd.to_datetime(bar[tcol])
|
||||
bar = bar.assign(_t=ts).set_index("_t").sort_index()
|
||||
bar = bar.loc[start:end]
|
||||
if len(bar) < 10:
|
||||
continue
|
||||
closes[sym] = bar["c"] if "c" in bar.columns else bar["close"]
|
||||
|
||||
return pd.DataFrame(closes)
|
||||
@@ -0,0 +1,202 @@
|
||||
"""Weekly-rebalance TopkDropout strategy.
|
||||
|
||||
Turnover-reduction variant of ``qlib.contrib.strategy.signal_strategy.TopkDropoutStrategy``:
|
||||
the topk/n_drop selection and sizing are identical to the reference, but the
|
||||
target book is recomputed only on the first trading day of each ISO week; on the
|
||||
other days the strategy issues NO orders (holds the book untouched).
|
||||
|
||||
The weekly cadence is derived from the qlib trade calendar: a rebalance happens
|
||||
when the current trade step's date belongs to a different ISO ``(year, week)``
|
||||
than the previous trade step. ``hold_band_pct`` (default 0) optionally skips
|
||||
tiny rebalances: when a name's existing position differs from the new target by
|
||||
less than this fraction, no order is generated for it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import List
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from qlib.backtest import Order
|
||||
from qlib.backtest.decision import OrderDir, TradeDecisionWO
|
||||
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
|
||||
|
||||
__all__ = ["WeeklyRebalanceDropoutStrategy"]
|
||||
|
||||
DEFAULT_HOLD_BAND_PCT = 0.0
|
||||
|
||||
|
||||
class WeeklyRebalanceDropoutStrategy(TopkDropoutStrategy):
|
||||
"""TopkDropout rebalanced once per ISO week; holds otherwise.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
topk, n_drop, method_sell, method_buy, hold_thresh, only_tradable,
|
||||
forbid_all_trade_at_limit : same as ``TopkDropoutStrategy``.
|
||||
hold_band_pct : skip order for a name whose deviation from target weight is
|
||||
below this fraction of the target (no-trade buffer band).
|
||||
"""
|
||||
|
||||
def __init__(self, *, topk, n_drop, hold_band_pct: float = DEFAULT_HOLD_BAND_PCT, **kwargs):
|
||||
super().__init__(topk=topk, n_drop=n_drop, **kwargs)
|
||||
self.hold_band_pct = hold_band_pct
|
||||
|
||||
@staticmethod
|
||||
def _iso_week(ts) -> tuple:
|
||||
return (ts.year, ts.week)
|
||||
|
||||
def generate_trade_decision(self, execute_result=None):
|
||||
import copy
|
||||
|
||||
trade_step = self.trade_calendar.get_trade_step()
|
||||
trade_start_time, trade_end_time = self.trade_calendar.get_step_time(trade_step)
|
||||
|
||||
cur_week = self._iso_week(trade_start_time)
|
||||
prev_week = getattr(self, "_last_week", None)
|
||||
self._last_week = cur_week
|
||||
|
||||
if prev_week is not None and prev_week == cur_week:
|
||||
# not the first trading day of this ISO week -> hold
|
||||
return TradeDecisionWO([], self)
|
||||
|
||||
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(trade_step, shift=1)
|
||||
pred_score = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
|
||||
if isinstance(pred_score, pd.DataFrame):
|
||||
pred_score = pred_score.iloc[:, 0]
|
||||
if pred_score is None:
|
||||
return TradeDecisionWO([], self)
|
||||
|
||||
if self.only_tradable:
|
||||
|
||||
def get_first_n(li, n, reverse=False):
|
||||
cur_n = 0
|
||||
res = []
|
||||
for si in reversed(li) if reverse else li:
|
||||
if self.trade_exchange.is_stock_tradable(
|
||||
stock_id=si, start_time=trade_start_time, end_time=trade_end_time
|
||||
):
|
||||
res.append(si)
|
||||
cur_n += 1
|
||||
if cur_n >= n:
|
||||
break
|
||||
return res[::-1] if reverse else res
|
||||
|
||||
def get_last_n(li, n):
|
||||
return get_first_n(li, n, reverse=True)
|
||||
|
||||
def filter_stock(li):
|
||||
return [
|
||||
si
|
||||
for si in li
|
||||
if self.trade_exchange.is_stock_tradable(
|
||||
stock_id=si, start_time=trade_start_time, end_time=trade_end_time
|
||||
)
|
||||
]
|
||||
|
||||
else:
|
||||
|
||||
def get_first_n(li, n):
|
||||
return list(li)[:n]
|
||||
|
||||
def get_last_n(li, n):
|
||||
return list(li)[-n:]
|
||||
|
||||
def filter_stock(li):
|
||||
return li
|
||||
|
||||
current_temp: "object" = copy.deepcopy(self.trade_position)
|
||||
sell_order_list: List[Order] = []
|
||||
buy_order_list: List[Order] = []
|
||||
cash = current_temp.get_cash()
|
||||
current_stock_list = current_temp.get_stock_list()
|
||||
last = pred_score.reindex(current_stock_list).sort_values(ascending=False).index
|
||||
|
||||
if self.method_buy == "top":
|
||||
today = get_first_n(
|
||||
pred_score[~pred_score.index.isin(last)].sort_values(ascending=False).index,
|
||||
self.n_drop + self.topk - len(last),
|
||||
)
|
||||
elif self.method_buy == "random":
|
||||
topk_candi = get_first_n(pred_score.sort_values(ascending=False).index, self.topk)
|
||||
candi = list(filter(lambda x: x not in last, topk_candi))
|
||||
n = self.n_drop + self.topk - len(last)
|
||||
try:
|
||||
today = np.random.choice(candi, n, replace=False)
|
||||
except ValueError:
|
||||
today = candi
|
||||
else:
|
||||
raise NotImplementedError(f"This type of input is not supported")
|
||||
|
||||
comb = pred_score.reindex(last.union(pd.Index(today))).sort_values(ascending=False).index
|
||||
|
||||
if self.method_sell == "bottom":
|
||||
sell = last[last.isin(get_last_n(comb, self.n_drop))]
|
||||
elif self.method_sell == "random":
|
||||
candi = filter_stock(last)
|
||||
try:
|
||||
sell = pd.Index(np.random.choice(candi, self.n_drop, replace=False) if len(last) else [])
|
||||
except ValueError:
|
||||
sell = candi
|
||||
else:
|
||||
raise NotImplementedError(f"This type of input is not supported")
|
||||
|
||||
buy = today[: len(sell) + self.topk - len(last)]
|
||||
for code in current_stock_list:
|
||||
if not self.trade_exchange.is_stock_tradable(
|
||||
stock_id=code,
|
||||
start_time=trade_start_time,
|
||||
end_time=trade_end_time,
|
||||
direction=None if self.forbid_all_trade_at_limit else OrderDir.SELL,
|
||||
):
|
||||
continue
|
||||
if code in sell:
|
||||
time_per_step = self.trade_calendar.get_freq()
|
||||
if current_temp.get_stock_count(code, bar=time_per_step) < self.hold_thresh:
|
||||
continue
|
||||
sell_amount = current_temp.get_stock_amount(code=code)
|
||||
sell_order = Order(
|
||||
stock_id=code,
|
||||
amount=sell_amount,
|
||||
start_time=trade_start_time,
|
||||
end_time=trade_end_time,
|
||||
direction=Order.SELL,
|
||||
)
|
||||
if self.trade_exchange.check_order(sell_order):
|
||||
sell_order_list.append(sell_order)
|
||||
trade_val, trade_cost, trade_price = self.trade_exchange.deal_order(
|
||||
sell_order, position=current_temp
|
||||
)
|
||||
cash += trade_val - trade_cost
|
||||
|
||||
if len(buy) == 0:
|
||||
return TradeDecisionWO(sell_order_list, self)
|
||||
|
||||
value = cash * self.risk_degree / len(buy)
|
||||
for code in buy:
|
||||
if not self.trade_exchange.is_stock_tradable(
|
||||
stock_id=code,
|
||||
start_time=trade_start_time,
|
||||
end_time=trade_end_time,
|
||||
direction=None if self.forbid_all_trade_at_limit else OrderDir.BUY,
|
||||
):
|
||||
continue
|
||||
buy_price = self.trade_exchange.get_deal_price(
|
||||
stock_id=code, start_time=trade_start_time, end_time=trade_end_time, direction=OrderDir.BUY
|
||||
)
|
||||
buy_amount = value / buy_price
|
||||
factor = self.trade_exchange.get_factor(
|
||||
stock_id=code, start_time=trade_start_time, end_time=trade_end_time
|
||||
)
|
||||
buy_amount = self.trade_exchange.round_amount_by_trade_unit(buy_amount, factor)
|
||||
buy_order = Order(
|
||||
stock_id=code,
|
||||
amount=buy_amount,
|
||||
start_time=trade_start_time,
|
||||
end_time=trade_end_time,
|
||||
direction=Order.BUY,
|
||||
)
|
||||
buy_order_list.append(buy_order)
|
||||
|
||||
return TradeDecisionWO(sell_order_list + buy_order_list, self)
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -6,10 +6,11 @@ The lake is a hive-partitioned parquet store (see ``tac-engine/skills/tradeac-la
|
||||
├── market=US/
|
||||
│ └── timeframe=1d/
|
||||
│ └── symbol=AAPL.parquet # OHLCV bars: t, date, o, h, l, c, v, n, vw
|
||||
├── features/ # ta-lib indicators, wide format
|
||||
├── features/ # indicators, wide format, family tier
|
||||
│ └── market=US/
|
||||
│ └── timeframe=1d/
|
||||
│ └── symbol=AAPL.parquet # t, sma_5, sma_20, rsi_14, ...
|
||||
│ ├── family=ta/symbol=AAPL.parquet # t, sma_5, sma_20, rsi_14, ...
|
||||
│ └── family=sp/symbol=AAPL.parquet # t, sp_ou_*, sp_hmm_*, ...
|
||||
├── calendar.parquet # trading days per market
|
||||
├── coverage.parquet # per (market,timeframe,symbol) loaded windows
|
||||
└── symbols.parquet # asset master
|
||||
@@ -107,8 +108,34 @@ class LakeConfig:
|
||||
return self.lake_root / "features" / f"market={self.market}" / f"timeframe={timeframe}"
|
||||
|
||||
def features_path(self, timeframe: str, symbol: str) -> Path:
|
||||
# Legacy flat path (no family tier). Prefer `load_features` which
|
||||
# resolves the family=ta|sp partition layout.
|
||||
return self.features_dir(timeframe) / f"symbol={str(symbol).upper()}.parquet"
|
||||
|
||||
def load_features(self, timeframe: str, symbol: str) -> pd.DataFrame:
|
||||
"""All feature columns for a symbol, merging the `family=ta` and
|
||||
`family=sp` partitions by timestamp. Returns an empty frame when no
|
||||
feature files exist (legacy flat layout falls back transparently)."""
|
||||
sym = str(symbol).upper()
|
||||
frames = []
|
||||
for family in ("ta", "sp"):
|
||||
p = self.features_dir(timeframe) / f"family={family}" / f"symbol={sym}.parquet"
|
||||
if p.exists():
|
||||
frames.append(pd.read_parquet(p))
|
||||
if not frames:
|
||||
flat = self.features_dir(timeframe) / f"symbol={sym}.parquet"
|
||||
if flat.exists():
|
||||
return pd.read_parquet(flat)
|
||||
return pd.DataFrame()
|
||||
if len(frames) == 1:
|
||||
return frames[0]
|
||||
merged = frames[0]
|
||||
for extra in frames[1:]:
|
||||
merged = merged.merge(extra, on="t", how="outer", suffixes=("", "_dup"))
|
||||
for c in [c for c in merged.columns if c.endswith("_dup")]:
|
||||
merged = merged.drop(columns=c)
|
||||
return merged
|
||||
|
||||
def calendar_path(self) -> Path:
|
||||
return self.lake_root / "calendar.parquet"
|
||||
|
||||
|
||||
@@ -173,8 +173,7 @@ class LakeFeatureProvider(FeatureProvider):
|
||||
def _load_feature_df(self, instrument: str, timeframe: str) -> pd.DataFrame:
|
||||
key = (instrument, timeframe)
|
||||
if key not in self._feature_cache:
|
||||
p = self.cfg.features_path(timeframe, instrument)
|
||||
self._feature_cache[key] = pd.read_parquet(p) if p.exists() else pd.DataFrame()
|
||||
self._feature_cache[key] = self.cfg.load_features(timeframe, instrument)
|
||||
return self._feature_cache[key]
|
||||
|
||||
@staticmethod
|
||||
|
||||
@@ -1,141 +0,0 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# ISOLATION: multi-seed RankIC ensemble, ablate-B generic-only feature set.
|
||||
#
|
||||
# Isolates the ensemble effect on the SP-5d rank signal. Same panel, segments,
|
||||
# history (full backfilled 2016+) and feature set as the exp-9 ablate-B winner
|
||||
# (generic-only sp_* families: jump,har,trend,hurst,signature), but replaces the
|
||||
# single RankICLGBModel with a 5-seed RankICEnsembleLGBModel (42,7,2026,99,123)
|
||||
# that averages per-day predictions.
|
||||
#
|
||||
# Differs from exp-15 (tac-rd-rank-ensemble, mlflow exp 15) ONLY by dropping the
|
||||
# TA subset (rsi_14,roc_10,macd_hist,willr_14,atr_14) and the inter-asset xr_*
|
||||
# features, so any change vs exp-15 is attributable to the feature set alone,
|
||||
# and any change vs exp-9 is attributable to the ensemble + full history alone.
|
||||
#
|
||||
# Run:
|
||||
# rd_run_workflow config_path=experiments/workflows/exp12_isolation_ensemble.yaml \
|
||||
# experiment_name=tac-rd-rank-ensemble-isolated
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///mlruns.db"
|
||||
default_exp_name: "tac-rd-rank-ensemble-isolated"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: RankICEnsembleLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 31
|
||||
n_estimators: 3000
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l2: 0.5
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seeds: "42,7,2026,99,123"
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: "{{ UNIVERSE }}"
|
||||
start_time: 2015-01-03
|
||||
end_time: 2026-08-14
|
||||
fit_start_time: 2016-01-04
|
||||
fit_end_time: 2025-09-01
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||
infer_processors:
|
||||
- class: DropAllNaN
|
||||
kwargs: {}
|
||||
- class: ProcessInf
|
||||
kwargs: {}
|
||||
- class: CSRankNorm
|
||||
kwargs: {}
|
||||
- class: ZScoreNorm
|
||||
kwargs: {}
|
||||
- class: Fillna
|
||||
kwargs: {}
|
||||
segments:
|
||||
train: [2016-01-04, 2025-09-01]
|
||||
valid: [2025-09-03, 2026-01-03]
|
||||
test: [2026-01-04, 2026-08-10]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 10
|
||||
n_drop: 2
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-01-04
|
||||
end_time: 2026-08-10
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
codes: "{{ UNIVERSE }}"
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -1,141 +0,0 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# EXP 18 - Risk-limit control: reference model + TopkDropout baseline (A).
|
||||
#
|
||||
# Signal/model identical to the reference (tac-rd-rank-ensemble-isolated,
|
||||
# run 0cea66d9...): RankICEnsembleLGBModel (parallel, 5 seeds) on the 50-ETF
|
||||
# SP-5d panel, test 2026-01-04..2026-08-10. This workflow reproduces the
|
||||
# unconstrained TopkDropout baseline net-of-cost so the risk-limited variant
|
||||
# (same pred, liquidity/size/concentration caps) can be compared 1:1.
|
||||
#
|
||||
# The risk_limits spec itself is applied via rd_backtest / rd_strategy_targets
|
||||
# (tool-level param, not a YAML key); this run records the unconstrained
|
||||
# baseline that the limit A/B is measured against.
|
||||
#
|
||||
# Run:
|
||||
# rd_run_workflow config_path=experiments/workflows/exp18-risk-limit/a_baseline.yaml \
|
||||
# experiment_name=tac-rd-risk-limit
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-risk-limit"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: RankICEnsembleLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 31
|
||||
n_estimators: 3000
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l2: 0.5
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seeds: "42,7,2026,99,123"
|
||||
parallel: 5
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: "{{ UNIVERSE }}"
|
||||
start_time: 2015-01-03
|
||||
end_time: 2026-08-14
|
||||
fit_start_time: 2016-01-04
|
||||
fit_end_time: 2025-09-01
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||
infer_processors:
|
||||
- class: DropAllNaN
|
||||
kwargs: {}
|
||||
- class: ProcessInf
|
||||
kwargs: {}
|
||||
- class: CSRankNorm
|
||||
kwargs: {}
|
||||
- class: ZScoreNorm
|
||||
kwargs: {}
|
||||
- class: Fillna
|
||||
kwargs: {}
|
||||
segments:
|
||||
train: [2016-01-04, 2025-09-01]
|
||||
valid: [2025-09-03, 2026-01-03]
|
||||
test: [2026-01-04, 2026-08-10]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 10
|
||||
n_drop: 2
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-01-04
|
||||
end_time: 2026-08-10
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
codes: "{{ UNIVERSE }}"
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -1,135 +0,0 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
|
||||
#
|
||||
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
|
||||
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
|
||||
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
|
||||
#
|
||||
# Run:
|
||||
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
|
||||
# experiment_name=tac-rd-risk-limit
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-risk-limit"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: RankICEnsembleLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 31
|
||||
n_estimators: 3000
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l2: 0.5
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seeds: "42,7,2026,99,123"
|
||||
parallel: 5
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: "{{ UNIVERSE }}"
|
||||
start_time: 2015-01-03
|
||||
end_time: 2026-08-14
|
||||
fit_start_time: 2016-01-04
|
||||
fit_end_time: 2025-09-01
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||
infer_processors:
|
||||
- class: DropAllNaN
|
||||
kwargs: {}
|
||||
- class: ProcessInf
|
||||
kwargs: {}
|
||||
- class: CSRankNorm
|
||||
kwargs: {}
|
||||
- class: ZScoreNorm
|
||||
kwargs: {}
|
||||
- class: Fillna
|
||||
kwargs: {}
|
||||
segments:
|
||||
train: [2016-01-04, 2025-09-01]
|
||||
valid: [2025-09-03, 2026-01-03]
|
||||
test: [2026-01-04, 2026-08-10]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 10
|
||||
n_drop: 2
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-01-04
|
||||
end_time: 2026-08-10
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
codes: "{{ UNIVERSE }}"
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -1,136 +0,0 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
|
||||
#
|
||||
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
|
||||
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
|
||||
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
|
||||
#
|
||||
# Run:
|
||||
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
|
||||
# experiment_name=tac-rd-risk-limit
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-risk-limit"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: RankICEnsembleLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 31
|
||||
n_estimators: 3000
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l2: 0.5
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seeds: "42,7,2026,99,123"
|
||||
parallel: 5
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: "{{ UNIVERSE }}"
|
||||
start_time: 2015-01-03
|
||||
end_time: 2026-08-14
|
||||
fit_start_time: 2016-01-04
|
||||
fit_end_time: 2025-09-01
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||
infer_processors:
|
||||
- class: DropAllNaN
|
||||
kwargs: {}
|
||||
- class: ProcessInf
|
||||
kwargs: {}
|
||||
- class: CSRankNorm
|
||||
kwargs: {}
|
||||
- class: ZScoreNorm
|
||||
kwargs: {}
|
||||
- class: Fillna
|
||||
kwargs: {}
|
||||
segments:
|
||||
train: [2016-01-04, 2025-09-01]
|
||||
valid: [2025-09-03, 2026-01-03]
|
||||
test: [2026-01-04, 2026-08-10]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: MomentumGateTopk
|
||||
module_path: tac_qlib.contrib.strategy.momentum_gate
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 10
|
||||
n_drop: 2
|
||||
min_momentum: 0.0
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-01-04
|
||||
end_time: 2026-08-10
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
codes: "{{ UNIVERSE }}"
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -1,138 +0,0 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
|
||||
#
|
||||
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
|
||||
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
|
||||
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
|
||||
#
|
||||
# Run:
|
||||
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
|
||||
# experiment_name=tac-rd-risk-limit
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-risk-limit"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: RankICEnsembleLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 31
|
||||
n_estimators: 3000
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l2: 0.5
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seeds: "42,7,2026,99,123"
|
||||
parallel: 5
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: "{{ UNIVERSE }}"
|
||||
start_time: 2015-01-03
|
||||
end_time: 2026-08-14
|
||||
fit_start_time: 2016-01-04
|
||||
fit_end_time: 2025-09-01
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||
infer_processors:
|
||||
- class: DropAllNaN
|
||||
kwargs: {}
|
||||
- class: ProcessInf
|
||||
kwargs: {}
|
||||
- class: CSRankNorm
|
||||
kwargs: {}
|
||||
- class: ZScoreNorm
|
||||
kwargs: {}
|
||||
- class: Fillna
|
||||
kwargs: {}
|
||||
segments:
|
||||
train: [2016-01-04, 2025-09-01]
|
||||
valid: [2025-09-03, 2026-01-03]
|
||||
test: [2026-01-04, 2026-08-10]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: HmmRiskTopk
|
||||
module_path: tac_qlib.contrib.strategy.hmm_risk
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 10
|
||||
n_drop: 2
|
||||
hmm_pause_pct: 0.70
|
||||
drawdown_pause_pct: 8.0
|
||||
liquidity_floor_adv: 5000000
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-01-04
|
||||
end_time: 2026-08-10
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
codes: "{{ UNIVERSE }}"
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -1,137 +0,0 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
|
||||
#
|
||||
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
|
||||
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
|
||||
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
|
||||
#
|
||||
# Run:
|
||||
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
|
||||
# experiment_name=tac-rd-risk-limit
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-risk-limit"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: RankICEnsembleLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 31
|
||||
n_estimators: 3000
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l2: 0.5
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seeds: "42,7,2026,99,123"
|
||||
weight_mode: rolling_ic
|
||||
rolling_ic_window: 21
|
||||
parallel: 5
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: "{{ UNIVERSE }}"
|
||||
start_time: 2015-01-03
|
||||
end_time: 2026-08-14
|
||||
fit_start_time: 2016-01-04
|
||||
fit_end_time: 2025-09-01
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||
infer_processors:
|
||||
- class: DropAllNaN
|
||||
kwargs: {}
|
||||
- class: ProcessInf
|
||||
kwargs: {}
|
||||
- class: CSRankNorm
|
||||
kwargs: {}
|
||||
- class: ZScoreNorm
|
||||
kwargs: {}
|
||||
- class: Fillna
|
||||
kwargs: {}
|
||||
segments:
|
||||
train: [2016-01-04, 2025-09-01]
|
||||
valid: [2025-09-03, 2026-01-03]
|
||||
test: [2026-01-04, 2026-08-10]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 10
|
||||
n_drop: 2
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-01-04
|
||||
end_time: 2026-08-10
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
codes: "{{ UNIVERSE }}"
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -1,135 +0,0 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
|
||||
#
|
||||
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
|
||||
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
|
||||
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
|
||||
#
|
||||
# Run:
|
||||
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
|
||||
# experiment_name=tac-rd-risk-limit
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead,sma_3,ema_3" %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
markets: {}
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs:
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs:
|
||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||
default_exp_name: "tac-rd-risk-limit"
|
||||
|
||||
task:
|
||||
model:
|
||||
class: RankICEnsembleLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 31
|
||||
n_estimators: 3000
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l2: 0.5
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seeds: "42,7,2026,99,123"
|
||||
parallel: 5
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: "{{ UNIVERSE }}"
|
||||
start_time: 2015-01-03
|
||||
end_time: 2026-08-14
|
||||
fit_start_time: 2016-01-04
|
||||
fit_end_time: 2025-09-01
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||
infer_processors:
|
||||
- class: DropAllNaN
|
||||
kwargs: {}
|
||||
- class: ProcessInf
|
||||
kwargs: {}
|
||||
- class: CSRankNorm
|
||||
kwargs: {}
|
||||
- class: ZScoreNorm
|
||||
kwargs: {}
|
||||
- class: Fillna
|
||||
kwargs: {}
|
||||
segments:
|
||||
train: [2016-01-04, 2025-09-01]
|
||||
valid: [2025-09-03, 2026-01-03]
|
||||
test: [2026-01-04, 2026-08-10]
|
||||
|
||||
record:
|
||||
- class: SignalRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: {}
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
ana_long_short: true
|
||||
ann_scaler: 252
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 10
|
||||
n_drop: 2
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-01-04
|
||||
end_time: 2026-08-10
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
codes: "{{ UNIVERSE }}"
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
Reference in New Issue
Block a user