Compare commits

..
Author SHA1 Message Date
zhaoli 46ecbbf0dc start experiment 80 (exp/80-scheduled-algo-retrain-on-2026-09-22-tac) 2026-09-23 12:12:46 +00:00
zhaoli a37fc0869f finish experiment 79 (exp/79-scheduled-algo-retrain-on-2026-09-21-tac) 2026-09-22 12:50:58 +00:00
zhaoli 05ccf47226 exp 79: snapshot custom qlib code 2026-09-22 12:18:06 +00:00
zhaoli 080cd73da3 start experiment 79 (exp/79-scheduled-algo-retrain-on-2026-09-21-tac) 2026-09-22 12:18:00 +00:00
zhaoli 278d9830af start experiment 78 (exp/78-scheduled-algo-retrain-on-2026-09-18-tac) 2026-09-21 12:02:16 +00:00
zhaoli f9c8fbe945 start experiment 69 (exp/69-scheduled-algo-retrain-on-2026-09-03-tac) 2026-09-04 12:01:21 +00:00
zhaoli ce2e0c1a7c start experiment 67 (exp/67-scheduled-algo-retrain-on-2026-09-01-tac) 2026-09-02 12:03:33 +00:00
zhaoli f1bd6d99c3 finish experiment 63 (exp/63-scheduled-algo-retrain-on-2026-08-26-tac) 2026-08-27 12:21:58 +00:00
zhaoli b872f64ce6 start experiment 63 (exp/63-scheduled-algo-retrain-on-2026-08-26-tac) 2026-08-27 12:02:21 +00:00
zhaoli 66cf0a149f finish experiment 33 (exp/33-q01-m2-reproduction-add-spsharpe22-to-th) 2026-08-19 22:40:28 +00:00
zhaoli 6a1b05db2e add q01 m2 repro workflow 2026-08-19 22:30:10 +00:00
zhaoli c7fc73180c start experiment 33 (exp/33-q01-m2-reproduction-add-spsharpe22-to-th) 2026-08-19 22:05:08 +00:00
zhaoli 483a86e47f finish experiment 31 (exp/31-isolation-run-m3-does-adding-garch11-vol) 2026-08-18 21:41:40 +00:00
zhaoli e660b4f2dd finish experiment 30 (exp/30-isolation-run-m2-does-adding-risk-adjust) 2026-08-18 20:13:22 +00:00
zhaoli 4e1debccba M2 isolation: base + sp_sharpe_22 (trace 30) 2026-08-18 15:54:24 +00:00
zhaoli cb18467a6d start experiment 30 (exp/30-isolation-run-m2-does-adding-risk-adjust) 2026-08-18 15:52:33 +00:00
zhaoli 894ac260a6 finish experiment 26 (exp/26-test-whether-reducing-topkdropout-daily) 2026-08-18 14:05:47 +00:00
zhaoli c455000a1e exp26: compact stochastic, n_drop 1 (workflow only) 2026-08-18 12:10:29 +00:00
zhaoli 5c7b2265f4 start experiment 26 (exp/26-test-whether-reducing-topkdropout-daily) 2026-08-18 10:03:21 +00:00
zhaoli c724682f3a finish experiment 24 (exp/24-run-the-rankic-ensemble-in-mlflow-experi) 2026-08-18 09:00:11 +00:00
zhaoli 59e733d88d exp 24: add exact compact stochastic feature workflow 2026-08-18 08:03:37 +00:00
zhaoli 1fcadbfe86 start experiment 24 (exp/24-run-the-rankic-ensemble-in-mlflow-experi) 2026-08-18 08:02:27 +00:00
zhaoli a718424340 finish experiment 23 (exp/23-test-whether-the-5-day-rankic-ensemble-i) 2026-08-18 07:58:20 +00:00
zhaoli adf0bfa812 exp 23: add general stochastic feature ablation workflow 2026-08-18 07:21:33 +00:00
zhaoli b5054ccc25 start experiment 23 (exp/23-test-whether-the-5-day-rankic-ensemble-i) 2026-08-18 07:20:27 +00:00
zhaoli 3d845306fe finish experiment 22 (exp/22-re-run-experiment-16s-5-day-rankic-ensem) 2026-08-18 07:17:36 +00:00
zhaoli 1075525d6e exp 22: add validated TA SP ensemble workflow 2026-08-18 06:38:55 +00:00
zhaoli 63c1ea763e start experiment 22 (exp/22-re-run-experiment-16s-5-day-rankic-ensem) 2026-08-18 06:38:28 +00:00
zhaoli 2c2684b103 start experiment 16 (16-scheduled-algo-retrain-on-20260814-tacrd) 2026-08-15 05:14:48 +00:00
zhaoli 32477c7bb8 exp 12: isolation ensemble workflow yaml (ablate-B features + full history) + finish record 2026-08-14 05:37:13 +00:00
zhaoli e657c58758 start exp 9 (sp5d-feature-family-ablation): baseline all-24 + generic-only 19 workflow YAMLs 2026-08-13 04:32:30 +00:00
46 changed files with 2197 additions and 1029 deletions
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -1,48 +0,0 @@
#!/usr/bin/env python3
"""Precompute the signal-quality gate series and save to pickle.
Usage:
python precompute_signal_quality_gate.py <pred_path> <output_path> [topk] [lookback] [threshold]
Example:
python precompute_signal_quality_gate.py \
/home/data/lake/mlruns/49/34165f27e4a34378ad54843a079a78c0/artifacts/pred.pkl \
/app/experiments/book/data/signal_quality_gate/sq_gate_5d_0.50.pkl \
10 5 0.5
"""
import sys
import pickle
from pathlib import Path
# Add tac-qlib to path
sys.path.insert(0, "/app/tac-qlib")
from tac_qlib.contrib.strategy.signal_quality_gate import compute_signal_quality_gate
if __name__ == "__main__":
if len(sys.argv) < 3:
print(__doc__)
sys.exit(1)
pred_path = sys.argv[1]
output_path = sys.argv[2]
topk = int(sys.argv[3]) if len(sys.argv) > 3 else 10
lookback = int(sys.argv[4]) if len(sys.argv) > 4 else 5
threshold = float(sys.argv[5]) if len(sys.argv) > 5 else 0.5
lake_root = "/home/data/lake"
print(f"Computing signal-quality gate: topk={topk}, lookback={lookback}, threshold={threshold}")
gate = compute_signal_quality_gate(
pred_path,
lake_root=lake_root,
topk=topk,
lookback=lookback,
threshold=threshold,
)
print(f"Gate: {gate.sum()}/{len(gate)} days open ({gate.mean():.1%})")
with open(output_path, "wb") as f:
pickle.dump(gate, f)
print(f"Saved to {output_path}")
-115
View File
@@ -1,115 +0,0 @@
# Signal-quality gate backtest for 2021
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
{# Gate computed on-the-fly from signal + lake close prices #}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-sq-gate-2021"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
kwargs:
loss: mse
learning_rate: 0.05
num_leaves: 15
n_estimators: 200
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.01
reg_lambda: 0.01
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2021-12-31
fit_start_time: 2015-01-03
fit_end_time: 2021-01-03
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
infer_processors:
- class: DropAllNaN
kwargs: {}
- class: ProcessInf
kwargs: {}
- class: CSRankNorm
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments:
train: [2015-01-03, 2020-09-01]
valid: [2020-09-03, 2021-01-03]
test: [2021-01-04, 2021-12-31]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs: { ana_long_short: true, ann_scaler: 252 }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: SignalQualityGateStrategy
module_path: tac_qlib.contrib.strategy.signal_quality_gate
kwargs:
signal: "<PRED>"
lake_root: "{{ LAKE }}"
gate_topk: 10
gate_lookback: 5
gate_threshold: 0.5
gate_start: "2015-01-03"
gate_end: "2021-12-31"
topk: 10
n_drop: 1
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2021-01-04
end_time: 2021-12-31
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
-115
View File
@@ -1,115 +0,0 @@
# Signal-quality gate backtest for 2024
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
{# Gate computed on-the-fly from signal + lake close prices #}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-sq-gate-2024"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
kwargs:
loss: mse
learning_rate: 0.05
num_leaves: 15
n_estimators: 200
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.01
reg_lambda: 0.01
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2024-12-31
fit_start_time: 2015-01-03
fit_end_time: 2024-01-03
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
infer_processors:
- class: DropAllNaN
kwargs: {}
- class: ProcessInf
kwargs: {}
- class: CSRankNorm
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments:
train: [2015-01-03, 2023-09-01]
valid: [2023-09-03, 2024-01-03]
test: [2024-01-02, 2024-12-31]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs: { ana_long_short: true, ann_scaler: 252 }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: SignalQualityGateStrategy
module_path: tac_qlib.contrib.strategy.signal_quality_gate
kwargs:
signal: "<PRED>"
lake_root: "{{ LAKE }}"
gate_topk: 10
gate_lookback: 5
gate_threshold: 0.5
gate_start: "2015-01-03"
gate_end: "2024-12-31"
topk: 10
n_drop: 1
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2024-01-02
end_time: 2024-12-31
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
+22 -23
View File
@@ -1,35 +1,34 @@
# TradeAC custom-qlib-code snapshot (auto-generated)
# parent repo HEAD : e952feed0a66a20439f4f24ad5524233429cd0c3
# parent repo HEAD : a37fc0869f99a0eebc34c8eb98af0514c6d7fba8
# tac-qlib/tac_qlib/contrib
# tac-qlib/tac_qlib/data
# per-file hashes (git hash-object):
1b6298c4a5652f2e863cbdc385a1014a570fcd59 tac-qlib/tac_qlib/contrib/__init__.py
861592c63edd6a0853a9cb174b5970435b135fc8 tac-qlib/tac_qlib/contrib/__pycache__/__init__.cpython-312.pyc
d1a8ec0c9e6c38839e966f687b08ec412f87ec20 tac-qlib/tac_qlib/contrib/__pycache__/__init__.cpython-312.pyc
2224424d0ff193be4f55d1b791f8fce89439c5d2 tac-qlib/tac_qlib/contrib/backtest/__init__.py
0bf40dee440ddbded357d7bbb4efc67c62c4b084 tac-qlib/tac_qlib/contrib/backtest/tradeac_exchange.py
c76a9f17f680e74eea766eff27f7624359749ed6 tac-qlib/tac_qlib/contrib/data/__init__.py
5c547a2ef92e075e550fe6d01508a2f1d3f536bc tac-qlib/tac_qlib/contrib/data/__pycache__/__init__.cpython-312.pyc
4f656130d167e79dcaaeb7783a121f0b36852374 tac-qlib/tac_qlib/contrib/data/__pycache__/handler.cpython-312.pyc
0dd25ef161c6e0f15eafc84886e7e1381deb38c3 tac-qlib/tac_qlib/contrib/data/handler.py
c1543fabfc22e07c7e9942015c4462f84acc1e91 tac-qlib/tac_qlib/contrib/data/__pycache__/__init__.cpython-312.pyc
89f5cfc15b946b6ea1fb7724f021024967cdb6c5 tac-qlib/tac_qlib/contrib/data/__pycache__/handler.cpython-312.pyc
3bba0f1696e4ab4b3deebec3f31f269b2e713899 tac-qlib/tac_qlib/contrib/data/handler.py
b151d139a0dcde87d74b21e7c4b729176ba5c39b tac-qlib/tac_qlib/contrib/model/__init__.py
b1489f2fc0dee85f0a4f90b2e6ad545ed9c8967b tac-qlib/tac_qlib/contrib/model/__pycache__/__init__.cpython-312.pyc
121ef237da1df1b8e21a561c3ad0db200b901339 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_ensemble.cpython-312.pyc
74d0da348cbcc3700c96b6f4fe4391488e61efc5 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_gbdt.cpython-312.pyc
f4e9bf0a78d22e256cade0d794a6958aaa7c6721 tac-qlib/tac_qlib/contrib/model/__pycache__/__init__.cpython-312.pyc
c35c166a8449c14fa3cd4972c363d3d9d58620ef tac-qlib/tac_qlib/contrib/model/__pycache__/rank_ensemble.cpython-312.pyc
98be60fe19fedb3d9cb21c3f91ef0e176bd93287 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_gbdt.cpython-312.pyc
d3f051f3a8650c42fedc7b367b966f7c74fb5789 tac-qlib/tac_qlib/contrib/model/rank_ensemble.py
d03e6611338918d4aac5eea4adf26f85a3763652 tac-qlib/tac_qlib/contrib/model/rank_gbdt.py
2c2f167b693f4366a769998e3c9d4804f29e31e0 tac-qlib/tac_qlib/contrib/strategy/__init__.py
f9cd9ab729e3248542ccc490af7adc2e51c71914 tac-qlib/tac_qlib/contrib/strategy/__pycache__/__init__.cpython-312.pyc
cd3133cfbd2556b25c106e39ae97fb128df0e326 tac-qlib/tac_qlib/contrib/strategy/__pycache__/ic_gate.cpython-312.pyc
6dd1c568a2961842793674390d5abffd1a0e71b8 tac-qlib/tac_qlib/contrib/strategy/__pycache__/optimal_stop.cpython-312.pyc
03b5e4d80da00800f1b108bee0735d3d18d856d1 tac-qlib/tac_qlib/contrib/strategy/__pycache__/regime_gate.cpython-312.pyc
a6a1c21ab71b62080830df45c4784b73c1531036 tac-qlib/tac_qlib/contrib/strategy/__pycache__/signal_quality_gate.cpython-312.pyc
755e3b139496a5e22b0328db45c8199c33029fbc tac-qlib/tac_qlib/contrib/strategy/__pycache__/weekly_rebalance.cpython-312.pyc
519a1f4c05dbe0ac018ab8b779eb33d53b4dd545 tac-qlib/tac_qlib/contrib/strategy/ic_gate.py
184f80da8edf944bad3c8fb4d4d3d189bf4f082b tac-qlib/tac_qlib/contrib/strategy/__init__.py
69684c9c9624cb16c63f3a18a864c49a7db37fa9 tac-qlib/tac_qlib/contrib/strategy/__pycache__/__init__.cpython-312.pyc
4d3a5433e57e65e5dc96c6d2999f9d473b696dc4 tac-qlib/tac_qlib/contrib/strategy/__pycache__/long_short.cpython-312.pyc
53e72f512b47207a5292a918efef7fee162deef3 tac-qlib/tac_qlib/contrib/strategy/__pycache__/optimal_stop.cpython-312.pyc
896ef74ae47bcd1ed388e1e5d9c8d70c28097fe9 tac-qlib/tac_qlib/contrib/strategy/kelly_dropout.py
9090fc6dfbd339f2f4df4b0c9b87f400ecb5c9d5 tac-qlib/tac_qlib/contrib/strategy/long_short.py
79aaad9e39fcc740a773f4f63c512ce1086cfde0 tac-qlib/tac_qlib/contrib/strategy/optimal_stop.py
7bcee5f0b09cfa721440f1354f16f2dd9a112b12 tac-qlib/tac_qlib/contrib/strategy/regime_gate.py
16ab80b731aab6d8bb818525615d77c2d1fcb0c8 tac-qlib/tac_qlib/contrib/strategy/signal_quality_gate.py
fe60bacdfedd48617863be31f24b7c7daebfac5a tac-qlib/tac_qlib/contrib/strategy/weekly_rebalance.py
5b9acfb4340111b204249add7760bd53c6ae03f1 tac-qlib/tac_qlib/contrib/strategy/regime_gate.py
839abb89ad40cd516eabcfd91fe1be626b9f091f tac-qlib/tac_qlib/contrib/strategy/weekly_rebalance.py
92e6e90eb0cd0a25142034560f27adb6b705b1a8 tac-qlib/tac_qlib/data/__init__.py
316bf4aa160cc8d15929ea648be03f4b4999667d tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc
554a3f29d181b64effbf49a8161b32e7f93d8d3e tac-qlib/tac_qlib/data/__pycache__/config.cpython-312.pyc
8b47f6d78ac046b6b7b2fb07bd7f3382773ffb73 tac-qlib/tac_qlib/data/__pycache__/providers.cpython-312.pyc
53c9007a928841fd3c3b08450f9a6520ce1ac091 tac-qlib/tac_qlib/data/config.py
0f1eb6f41d44bd9064b1f16f529dac6bb9d7c3fd tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc
cf99dd8a481f6e2d1f323f8d8d754730eadb357a tac-qlib/tac_qlib/data/__pycache__/config.cpython-312.pyc
ddac5bab6578399c0301390d71bae13511a56cb1 tac-qlib/tac_qlib/data/__pycache__/providers.cpython-312.pyc
1953fb2a6371525db7f7b0e1c9dfbf3492d82110 tac-qlib/tac_qlib/data/config.py
8d0644f6f0d1efb94798ed444cc73e63b643459b tac-qlib/tac_qlib/data/providers.py
@@ -0,0 +1 @@
from .tradeac_exchange import TradeACExchange
@@ -0,0 +1,432 @@
# Copyright (c) Microsoft Corporation.
# Licensed under the MIT License.
"""
TradeACExchange
A short/borrow enabled Exchange implementation built on top of qlib.backtest.exchange.Exchange.
This exchange adds simple, configurable margin logic (initial/maintenance), borrowing support for
shorts, a borrow fee, and a lightweight SMA (Special Memorandum Account) concept to emulate
behaviors similar to brokers such as IBKR and Alpaca for backtesting purposes.
Notes / limitations
- This implementation is intentionally lightweight and conservative: it implements the
key behaviors needed for strategy/backtest experiments (allowing short selling, computing
margin requirements, performing margin-call checks, and tracking SMA-like excess equity).
- It makes some simplifying assumptions compared to real brokers (no per-product house margins,
simplified SMA bookkeeping, borrow availability modeled only by a per-symbol boolean/limit).
- The Position class in qlib.backtest.position was not changed. To support shorts we update the
position.position dict directly when necessary. This keeps integration simple but bypasses some
internal Position helpers. Use with care.
API additions
- allow_short: enable short selling (bool)
- initial_margin_long/short: fraction required to open a position
- maintenance_margin_long/short: fraction required to keep a position
- borrow_fee_rate: periodic borrow fee applied on short value (applied at trade time as additional cost)
- borrowable: dict mapping stock_id -> bool or float (max borrowable shares). Symbols missing from
the dict follow `borrow_default` (default True = unlimited; set False for a strict whitelist)
- get_sma(position): returns SMA-like excess equity available as "buying power credit"
- check_margin_call(position): returns True if position is below maintenance requirement
"""
from __future__ import annotations
from typing import Any, Dict, Optional, Tuple
import numpy as np
from qlib.backtest.decision import Order
from qlib.backtest.exchange import Exchange
from qlib.backtest.position import BasePosition
class TradeACExchange(Exchange):
"""An exchange that supports short selling / borrowing and basic margin rules.
The implementation aims to be compatible with the Exchange API used by Account and
Position classes in qlib.backtest. It overrides only the minimum methods required to
enable short/borrow behavior and margin calculations.
"""
def __init__(
self,
*args: Any,
allow_short: bool = True,
initial_margin_long: float = 0.5,
initial_margin_short: float = 0.5,
maintenance_margin_long: float = 0.25,
maintenance_margin_short: float = 0.3,
borrow_fee_rate: float = 0.0,
borrowable: Optional[Dict[str, float]] = None,
borrow_default: bool = True,
sma_enabled: bool = True,
**kwargs: Any,
) -> None:
"""Create TradeACExchange.
Parameters mirror Exchange with additional tradeac-specific options.
"""
super().__init__(*args, **kwargs)
self.allow_short = allow_short
self.initial_margin_long = initial_margin_long
self.initial_margin_short = initial_margin_short
self.maintenance_margin_long = maintenance_margin_long
self.maintenance_margin_short = maintenance_margin_short
self.borrow_fee_rate = borrow_fee_rate
# borrowable can be a dict with per-symbol max borrowable amount, or None (unlimited)
self.borrowable = borrowable or {}
# borrow_default: policy for symbols absent from `borrowable`.
# True -> unlisted symbols are unlimited-borrowable (legacy behavior)
# False -> unlisted symbols are NOT borrowable; only listed ones can be shorted
self.borrow_default = bool(borrow_default)
# sma_enabled: whether to expose lightweight SMA calculation
self.sma_enabled = sma_enabled
# --------------------------- Helper calculations ---------------------------
def _initial_margin_requirement(self, position: BasePosition) -> float:
"""Compute the initial margin requirement (money) for the given position.
We treat longs and shorts separately and sum their required initial margins.
"""
im_req = 0.0
for sid in position.get_stock_list():
amt = position.get_stock_amount(sid)
price = position.get_stock_price(sid)
val = amt * price
if val > 0:
im_req += abs(val) * self.initial_margin_long
elif val < 0:
im_req += abs(val) * self.initial_margin_short
return im_req
def _maintenance_margin_requirement(self, position: BasePosition) -> float:
"""Compute the maintenance margin requirement (money) for the given position."""
mm_req = 0.0
for sid in position.get_stock_list():
amt = position.get_stock_amount(sid)
price = position.get_stock_price(sid)
val = amt * price
if val > 0:
mm_req += abs(val) * self.maintenance_margin_long
elif val < 0:
mm_req += abs(val) * self.maintenance_margin_short
return mm_req
def get_equity(self, position: BasePosition) -> float:
"""Return account equity (position value + cash)."""
return position.calculate_value()
def get_sma(self, position: BasePosition) -> float:
"""Return a simplified SMA: excess equity above initial margin requirement.
Note: This is a synthetic/Simplified SMA used for strategy/backtest logic. Real-broker
SMA accounting (e.g. credits/debits across days) can be more complex.
"""
if not self.sma_enabled:
return 0.0
equity = self.get_equity(position)
im_req = self._initial_margin_requirement(position)
return max(0.0, equity - im_req)
def check_margin_call(self, position: BasePosition) -> bool:
"""Return True when the account is under maintenance margin (margin call).
Margin call condition here is simple: equity < maintenance requirement.
"""
equity = self.get_equity(position)
mm_req = self._maintenance_margin_requirement(position)
return equity < mm_req
def get_buying_power(self, position: BasePosition) -> float:
"""Estimate buying power for new long positions assuming opening margin requirement.
Simplified: the maximum notional long value = equity / initial_margin_long.
"""
equity = self.get_equity(position)
if self.initial_margin_long <= 0:
return 0.0
return equity / self.initial_margin_long
# --------------------------- Order / execution overrides ---------------------------
def _borrow_headroom(self, stock_id: str, current_short: float) -> float:
"""Remaining borrowable shares for `stock_id` given an already-open short of `current_short` shares.
borrowable values: bool (True=unlimited, False=not borrowable) or numeric max shares.
Missing symbols follow `borrow_default` (True = unlimited when allow_short is enabled).
"""
if not self.allow_short:
return 0.0
v = self.borrowable.get(stock_id, self.borrow_default)
if isinstance(v, bool):
return float("inf") if v else 0.0
try:
limit = float(v)
except (TypeError, ValueError):
return float("inf")
return max(0.0, limit - max(current_short, 0.0))
def _calc_trade_info_by_order(
self,
order: Order,
position: Optional[BasePosition],
dealt_order_amount: Dict[str, float],
) -> Tuple[float, float, float]:
"""Override to allow (optionally) short selling and to apply borrow fees.
The original Exchange implementation forbids selling more than you own. Here we allow
sell orders to create/expand short positions when allow_short is True. We still rely on
most base logic (price discovery, impact, cost calculation) by calling super(), but we
adjust the sell-side clipping behavior before delegating to the base implementation.
"""
# When selling and shorts are allowed, temporarily relax the clipping logic in the base
# implementation by monkey-patching current position check. Simpler: replicate minimal
# parts of logic from Exchange._calc_trade_info_by_order with the key change.
# Get basic trade price & volume info using Exchange helpers
trade_price = float(self.get_deal_price(order.stock_id, order.start_time, order.end_time, direction=order.direction))
total_trade_val = float(self.get_volume(order.stock_id, order.start_time, order.end_time) or 0.0) * trade_price
order.factor = self.get_factor(order.stock_id, order.start_time, order.end_time)
order.deal_amount = order.amount # attempt full
# volume clipping (same as base)
self._clip_amount_by_volume(order, dealt_order_amount)
# approximate adjusted cost ratio based on liquidity
if not total_trade_val or np.isnan(total_trade_val) or total_trade_val <= 0:
adj_cost_ratio = self.impact_cost
else:
trade_val_tmp = order.deal_amount * trade_price
adj_cost_ratio = self.impact_cost * (trade_val_tmp / total_trade_val) ** 2
# Differentiate buy / sell
if order.direction == Order.SELL:
cost_ratio = self.close_cost + adj_cost_ratio
current_amount = (
position.get_stock_amount(order.stock_id) if (position is not None and position.check_stock(order.stock_id)) else 0.0
)
long_held = max(current_amount, 0.0)
short_open = max(-current_amount, 0.0)
if position is not None:
if not self.allow_short:
# clip by current holdings only
if not np.isclose(order.deal_amount, current_amount):
order.deal_amount = self.round_amount_by_trade_unit(
min(long_held, order.deal_amount), order.factor
)
else:
# allow selling beyond holdings up to the remaining borrow limit;
# later when updating the position we create/expand a short if necessary.
max_sell = long_held + self._borrow_headroom(order.stock_id, short_open)
if order.deal_amount > max_sell and not np.isclose(order.deal_amount, max_sell):
order.deal_amount = self.round_amount_by_trade_unit(max_sell, order.factor)
elif order.direction == Order.BUY:
cost_ratio = self.open_cost + adj_cost_ratio
if position is not None:
cash = position.get_cash()
trade_val = order.deal_amount * trade_price
if cash < max(trade_val * cost_ratio, self.min_cost):
order.deal_amount = 0
self.logger.debug(f"Order clipped due to cost higher than cash: {order}")
elif cash < trade_val + max(trade_val * cost_ratio, self.min_cost):
max_buy_amount = self._get_buy_amount_by_cash_limit(trade_price, cash, cost_ratio)
order.deal_amount = self.round_amount_by_trade_unit(min(max_buy_amount, order.deal_amount), order.factor)
self.logger.debug(f"Order clipped due to cash limitation: {order}")
else:
order.deal_amount = self.round_amount_by_trade_unit(order.deal_amount, order.factor)
else:
order.deal_amount = self.round_amount_by_trade_unit(order.deal_amount, order.factor)
else:
raise NotImplementedError("order direction {} error".format(order.direction))
# compute final trade_val & trade_cost
trade_val = order.deal_amount * trade_price
# base trade_cost
trade_cost = max(trade_val * cost_ratio, self.min_cost)
# apply borrow fee only on the net-new short portion of the sell
if order.direction == Order.SELL and self.allow_short:
new_short = max(0.0, order.deal_amount - long_held)
trade_cost += new_short * trade_price * self.borrow_fee_rate
if trade_val <= 1e-5:
trade_cost = 0
return trade_price, trade_val, trade_cost
def deal_order(
self,
order: Order,
trade_account: Optional[Any] = None,
position: Optional[BasePosition] = None,
dealt_order_amount: Dict[str, float] = None,
) -> Tuple[float, float, float]:
"""Deal order and handle short position bookkeeping.
This method mirrors Exchange.deal_order but when a position is provided and shorts are
allowed it will update the Position.position dict directly to support negative amounts.
"""
if dealt_order_amount is None:
dealt_order_amount = {}
if not self.check_order(order):
order.deal_amount = 0.0
self.logger.debug(f"Order failed due to trading limitation: {order}")
return 0.0, 0.0, np.nan
if trade_account is not None and position is not None:
raise ValueError("trade_account and position can only choose one")
pos = position or (trade_account.current_position if trade_account is not None else None)
trade_price, trade_val, trade_cost = self._calc_trade_info_by_order(order, pos, dealt_order_amount)
if trade_val > 1e-5:
if trade_account is not None:
cp = trade_account.current_position
if not cp.skip_update():
held = cp.check_stock(order.stock_id)
# Account-level bookkeeping (turnover/cost/returns). Mirrors
# Account._update_state_from_order except for fresh short sales,
# where no prior price exists to compute order profit from.
if order.direction == Order.SELL and not held:
trade_account.accum_info.add_turnover(trade_val)
trade_account.accum_info.add_cost(trade_cost)
trade_account.accum_info.add_return_value(0.0)
if order.direction == Order.SELL:
# sell: update account state first (stock entry may be deleted)
if held:
trade_account._update_state_from_order(order, trade_val, trade_cost, trade_price)
self._position_sell(cp, order, trade_val, trade_cost, trade_price)
else:
# buy: update position first (entry may be created), then account state
# A buy that covers a short to exactly flat deletes the entry inside
# _position_buy; re-seed a transient zero-amount stub so the
# account's order-profit lookup still finds the trade price,
# then drop it (_update_state_from_order never mutates entries).
sid = order.stock_id
had_entry = isinstance(cp.position.get(sid), dict)
self._position_buy(cp, order, trade_val, trade_cost, trade_price)
covered_to_flat = had_entry and not isinstance(cp.position.get(sid), dict)
if covered_to_flat:
cp.position[sid] = {"amount": 0.0, "price": trade_price, "weight": 0}
trade_account._update_state_from_order(order, trade_val, trade_cost, trade_price)
if covered_to_flat:
cp.position.pop(sid, None)
elif position is not None:
if order.direction == Order.BUY:
self._position_buy(position, order, trade_val, trade_cost, trade_price)
else:
self._position_sell(position, order, trade_val, trade_cost, trade_price)
return trade_val, trade_cost, trade_price
# --------------------------- Position mutation helpers ---------------------------
def _position_buy(self, position: BasePosition, order: Order, trade_val: float, cost: float, trade_price: float) -> None:
"""Handle buy order bookkeeping against a BasePosition while supporting shorts.
Rules implemented (simplified):
- If there is an existing short (amount < 0), the buy will first cover the short.
- If covering closes the short completely, the remaining buy becomes a long
- Cash updates mimic Position._buy_stock/_sell_stock (cash decreases by trade_val+cost for buys)
"""
trade_amount = trade_val / trade_price
sid = order.stock_id
current_amount = position.get_stock_amount(sid) if position.check_stock(sid) else 0.0
# covering existing short
if current_amount < -1e-12:
# amount is negative -> we are short. Buying reduces the short.
new_amount = current_amount + trade_amount
if abs(new_amount) <= 1e-8:
# short fully covered exactly -> remove entry
if sid in position.position:
del position.position[sid]
elif new_amount > 0:
# short fully covered with leftover buy amount -> leftover becomes a long position
position.position[sid] = {"amount": new_amount, "price": trade_price, "weight": 0}
else:
# partially cover
position.position[sid]["amount"] = new_amount
position.position[sid]["price"] = trade_price
else:
# normal or increasing long
if sid not in position.position or not isinstance(position.position[sid], dict):
# initialize stock
position.position[sid] = {"amount": trade_amount, "price": trade_price, "weight": 0}
else:
position.position[sid]["amount"] = position.position[sid].get("amount", 0.0) + trade_amount
position.position[sid]["price"] = trade_price
# cash effect same as Position._buy_stock
position.position["cash"] -= trade_val + cost
def _position_sell(self, position: BasePosition, order: Order, trade_val: float, cost: float, trade_price: float) -> None:
"""Handle sell order bookkeeping against a BasePosition while supporting shorts.
Rules implemented (simplified):
- If holding enough long shares, sell will reduce/close the long position normally.
- If not holding enough long shares and shorts are allowed, the remaining sold amount will create/expand a short position.
- Cash update for sells follows Position._sell_stock logic (cash increases by trade_val - cost)
"""
trade_amount = trade_val / trade_price
sid = order.stock_id
current_amount = position.get_stock_amount(sid) if position.check_stock(sid) else 0.0
if current_amount > 1e-12:
# we have long shares; sell from them first
if trade_amount >= current_amount - 1e-8:
# selling all or more than holdings
# remove long position
if sid in position.position:
del position.position[sid]
# remaining sold amount becomes short if allowed
remain = trade_amount - current_amount
if remain > 1e-8:
if not self.allow_short:
# should not happen due to clipping earlier, but guard anyway
raise ValueError(f"Attempt to short {sid} while shorting disabled")
# create short entry
position.position[sid] = {"amount": -remain, "price": trade_price, "weight": 0}
else:
# partial sell
position.position[sid]["amount"] = current_amount - trade_amount
position.position[sid]["price"] = trade_price
else:
# currently flat or already short
if not self.allow_short:
raise ValueError(f"Attempt to short {sid} while shorting disabled")
# expand short
new_amount = current_amount - trade_amount
if sid not in position.position or not isinstance(position.position[sid], dict):
position.position[sid] = {"amount": new_amount, "price": trade_price, "weight": 0}
else:
position.position[sid]["amount"] = new_amount
position.position[sid]["price"] = trade_price
# cash effect same as Position._sell_stock
new_cash = trade_val - cost
if getattr(position, "_settle_type", None) == position.ST_CASH:
position.position["cash_delay"] = position.position.get("cash_delay", 0.0) + new_cash
else:
position.position["cash"] = position.position.get("cash", 0.0) + new_cash
# --------------------------- Borrow availability helpers ---------------------------
def is_borrowable(self, stock_id: str, amount: float) -> bool:
"""Check whether the requested amount is borrowable for the given stock.
If a borrowable dict is provided, it may contain either booleans or numeric limits (maximum borrowable shares).
Symbols absent from the dict follow `borrow_default`.
"""
if not self.allow_short:
return False
if stock_id not in self.borrowable:
return self.borrow_default
v = self.borrowable[stock_id]
if isinstance(v, bool):
return v
try:
limit = float(v)
return amount <= limit
except Exception:
return True
+179 -5
View File
@@ -15,6 +15,9 @@ import os
from inspect import getfullargspec
from typing import List, Optional, Tuple, Union
import numpy as np
import pandas as pd
from qlib.data.dataset import processor as processor_module
from qlib.data.dataset.handler import DataHandlerLP
from qlib.utils import get_callable_kwargs
@@ -22,6 +25,7 @@ from qlib.utils import get_callable_kwargs
from ...data.config import (
LakeConfig,
timeframe_for_freq,
FEATURE_FAMILIES,
NON_FEATURE_COLUMNS,
)
@@ -92,7 +96,7 @@ def get_common_feature_fields(lake_root=None, market="US", timeframe="1d") -> Li
common: set = set()
# family tier: features/market=*/timeframe=*/family=*/symbol=*.parquet
for fam in ("ta", "sp"):
for fam in FEATURE_FAMILIES:
fam_dir = feat_dir / f"family={fam}"
if fam_dir.is_dir():
common |= _family_common(fam_dir)
@@ -143,6 +147,174 @@ class DropAllNaN(processor_module.Processor):
return df
class BenchResidual(processor_module.Processor):
"""Subtract a benchmark instrument's forward return from the label, per datetime.
Turns the training target from an absolute-return rank into a *residual* rank:
``r_i - r_bench`` is ranked cross-sectionally by the downstream ``CSRankNorm`` /
``CSZScoreNorm`` processors instead of ``r_i`` alone. Must be inserted BEFORE any
per-date normalization so the ranking itself is computed on residual returns
(ordering flips exactly where the benchmark trends).
Stateless: ``fit`` is a no-op and the benchmark forward return is recomputed from
the lake parquet on first ``__call__``. Rows whose benchmark value is missing are
left untouched. Accepts ``fit_start_time``/``fit_end_time`` (ignored) so
``check_transform_proc`` can inject the fit window uniformly.
NOTE: under any cross-sectional normalization downstream (``CSRankNorm`` /
``CSZScoreNorm``) this processor is a mathematical no-op: subtracting the same
per-date constant preserves ranks, and z-scoring absorbs constant shifts. Use
``BenchBetaResidual`` for a target that actually reorders.
"""
def __init__(
self,
benchmark="SPY",
fields_group="label",
lake_root=None,
market="US",
timeframe=None,
freq="day",
fit_start_time=None,
fit_end_time=None,
):
self.benchmark = benchmark
self.fields_group = fields_group
self.lake_root = lake_root
self.market = market
self.timeframe = timeframe or timeframe_for_freq(freq)
self.fit_start_time = fit_start_time
self.fit_end_time = fit_end_time
self._bench_label = None
def _load_bench_label(self):
if self._bench_label is not None:
return self._bench_label
cfg = LakeConfig(self.lake_root, self.market)
p = cfg.bar_path(self.timeframe, self.benchmark)
if not p.exists():
raise FileNotFoundError(f"BenchResidual: benchmark bar file not found: {p}")
df = pd.read_parquet(p)
s = pd.Series(df["c"].astype(float).values, index=pd.to_datetime(df["t"])).sort_index()
s.index = s.index.normalize()
# mirror Ref($close,-6)/Ref($close,-1)-1 on the benchmark's own calendar
bench_label = s.shift(-6) / s.shift(-1) - 1
self._bench_label = bench_label[~bench_label.index.duplicated(keep="last")]
return self._bench_label
def fit(self, df=None):
return self
def __call__(self, df):
bl = self._load_bench_label()
cols = processor_module.get_group_columns(df, self.fields_group)
dt = df.index.get_level_values("datetime")
aligned = bl.reindex(pd.DatetimeIndex(dt.unique())).reindex(dt)
mask = aligned.notna().values
out = df.copy()
for c in cols:
vals = df[c].values
res = vals.copy()
res[mask] = np.asarray(vals[mask], dtype=float) - aligned[mask].values
out[c] = res
return out
class BenchBetaResidual(processor_module.Processor):
"""Residualize the label against a beta-scaled benchmark move: ``r_i - b_i * r_bench``.
Unlike a plain constant subtraction (see ``BenchResidual``), the name-specific rolling
beta ``b_i`` makes this survive cross-sectional normalization: in up-weeks high-beta
names lose rank, in down-weeks they gain — exactly the relative structure an absolute-
return ranking hides.
Beta is estimated from *past* data only (rolling ``window`` trading days of daily close
returns of each instrument vs the benchmark, both read up to and including ``t``), so
no lookahead enters the target. The benchmark leg uses the same horizon as the label
expression (``Ref($close,-6)/Ref($close,-1)-1`` by default via ``horizon``/``base``,
matching the yaml's 6-day label). Rows with missing beta or benchmark values keep
their raw label.
Requires ``$close`` to be present in the feature group (it always is for TACHandler).
Stateless; accepts ``fit_start_time``/``fit_end_time`` (ignored) for uniform kwargs
injection. Must be inserted BEFORE any per-date normalization processor.
"""
def __init__(
self,
benchmark="SPY",
fields_group="label",
lake_root=None,
market="US",
timeframe=None,
freq="day",
window=63,
horizon=6,
base=1,
feature_field="$close",
fit_start_time=None,
fit_end_time=None,
):
self.benchmark = benchmark
self.fields_group = fields_group
self.lake_root = lake_root
self.market = market
self.timeframe = timeframe or timeframe_for_freq(freq)
self.window = int(window)
self.horizon = int(horizon)
self.base = int(base)
self.feature_field = feature_field
self.fit_start_time = fit_start_time
self.fit_end_time = fit_end_time
self._bench = None
def _load_bench_close(self):
if self._bench is not None:
return self._bench
cfg = LakeConfig(self.lake_root, self.market)
p = cfg.bar_path(self.timeframe, self.benchmark)
if not p.exists():
raise FileNotFoundError(f"BenchBetaResidual: benchmark bar file not found: {p}")
df = pd.read_parquet(p)
s = pd.Series(df["c"].astype(float).values, index=pd.to_datetime(df["t"])).sort_index()
s.index = s.index.normalize()
self._bench = s[~s.index.duplicated(keep="last")]
return self._bench
def fit(self, df=None):
return self
def __call__(self, df):
bench = self._load_bench_close()
# benchmark forward return over the same horizon as the label expression
fwd = bench.shift(-(self.base + self.horizon - 1)) / bench.shift(-self.base) - 1
px_col = ("feature", self.feature_field)
if px_col not in df.columns:
raise KeyError(f"BenchBetaResidual: {self.feature_field} not found in features")
px = df[px_col].unstack("instrument").sort_index()
rets = px / px.shift(1) - 1
bret = bench.reindex(px.index).pct_change()
# rolling beta per instrument using data <= t (no lookahead)
cov = rets.rolling(self.window, min_periods=max(10, self.window // 2)).cov(bret)
var = bret.rolling(self.window, min_periods=max(10, self.window // 2)).var()
beta = cov.div(var, axis=0)
contrib = beta.mul(fwd.reindex(px.index), axis=0)
cols = list(processor_module.get_group_columns(df, self.fields_group))
out = df.copy()
for c in cols:
lab = df[c].unstack("instrument").reindex(px.index)
resid = lab - contrib.where(contrib.notna() & lab.notna(), 0.0)
new_vals = resid.stack()
new_vals.index.names = df.index.names
# residual where available, raw label otherwise (e.g. beta warm-up rows)
out[c] = new_vals.reindex(out.index).fillna(df[c])
return out
class TACHandler(DataHandlerLP):
"""DataHandlerLP backed by the TradeAC parquet lake.
@@ -245,10 +417,12 @@ class TACHandler(DataHandlerLP):
return get_common_feature_fields(lake_root, market, timeframe_for_freq(freq))
__all__ = ["TACHandler", "DropAllNaN", "get_common_feature_fields"]
__all__ = ["TACHandler", "DropAllNaN", "BenchResidual", "BenchBetaResidual", "get_common_feature_fields"]
# Make `DropAllNaN` resolvable by bare name from processor configs (e.g. the default
# ``infer_processors`` and workflow yamls that reference it without a ``module_path``),
# mirroring how qlib registers its own processors in ``qlib.data.dataset.processor``.
# Make `DropAllNaN`/`BenchResidual`/`BenchBetaResidual` resolvable by bare name from processor
# configs (e.g. the default ``infer_processors`` and workflow yamls that reference them without a
# ``module_path``), mirroring how qlib registers its own processors in ``qlib.data.dataset.processor``.
processor_module.DropAllNaN = DropAllNaN
processor_module.BenchResidual = BenchResidual
processor_module.BenchBetaResidual = BenchBetaResidual
@@ -1,11 +1,4 @@
from .ic_gate import ICGateTopkDropoutStrategy # noqa: F401
from .optimal_stop import OptimalStopControl # noqa: F401
from .regime_gate import RegimeGateTopkDropoutStrategy # noqa: F401
from .weekly_rebalance import WeeklyRebalanceDropoutStrategy # noqa: F401
from .long_short import LongShortTopkStrategy # noqa: F401
__all__ = [
"ICGateTopkDropoutStrategy",
"OptimalStopControl",
"RegimeGateTopkDropoutStrategy",
"WeeklyRebalanceDropoutStrategy",
]
__all__ = ["OptimalStopControl", "LongShortTopkStrategy"]
@@ -1,117 +0,0 @@
"""Realized-IC circuit breaker TopkDropout strategy.
Subclass of ``qlib.contrib.strategy.signal_strategy.TopkDropoutStrategy`` that
holds the book (issues NO orders) while the streaming realized RankIC of the
deployed signal is below threshold — i.e. the model's cross-sectional
predictions are no longer earning against realized forward returns. When the
gate is open it behaves exactly like the reference TopkDropoutStrategy.
The gate is evaluated per trade step on the trailing mean realized RankIC of
the signal over the last ``ic_window`` trading days whose label is fully
realized as of the decision date (no lookahead — a 5d fwd label ``close[t+6]/
close[t+1]-1`` is only known at ``t+6``).
Two wiring modes:
* ``ic_gate``: a precomputed ``pd.Series`` indexed by datetime of booleans
(True = gate open / trade allowed). Computed once by the caller (e.g.
``rd_backtest``) and looked up per step. Missing dates default to open.
* realized-IC self-computation: when ``ic_min_rankic`` is given but no
``ic_gate``, the strategy computes the per-date realized RankIC itself from
``self.signal`` (the pred scores) and the lake 1d bars via
``tac_qlib.risk_limits.realized_rankic_series``, then applies the same
trailing-window comparison. Works when instantiated from a workflow YAML
PortAnaRecord config (``lake_root`` / ``market`` must be provided).
"""
from __future__ import annotations
import pandas as pd
from qlib.backtest.decision import TradeDecisionWO
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
from tac_qlib.risk_limits import ic_circuit_breaker, realized_rankic_series
__all__ = ["ICGateTopkDropoutStrategy"]
class ICGateTopkDropoutStrategy(TopkDropoutStrategy):
"""TopkDropout with a streaming realized-IC circuit breaker.
Parameters
----------
topk, n_drop, method_sell, method_buy, hold_thresh, only_tradable,
forbid_all_trade_at_limit : same as ``TopkDropoutStrategy``.
ic_min_rankic : float — pause new trading while trailing realized RankIC is
below this threshold (0 disables the gate).
ic_window : int — trailing window for the realized RankIC mean (default 22).
ic_label_horizon : int — label horizon in trading days (default 6).
ic_min_obs : int — min realized labels before the gate arms (default 10).
ic_gate : pd.Series, optional — precomputed per-date gate (bool indexed by
datetime). When provided, it overrides self-computation.
lake_root, market : str — lake location for self-computed realized IC.
"""
def __init__(
self,
*,
topk,
n_drop,
ic_min_rankic: float = 0.0,
ic_window: int = 22,
ic_label_horizon: int = 6,
ic_min_obs: int = 10,
ic_gate=None,
lake_root: str = "",
market: str = "US",
**kwargs,
):
super().__init__(topk=topk, n_drop=n_drop, **kwargs)
self.ic_min_rankic = float(ic_min_rankic or 0.0)
self.ic_window = int(ic_window or 22)
self.ic_label_horizon = int(ic_label_horizon or 6)
self.ic_min_obs = int(ic_min_obs or 10)
self._ic_gate = ic_gate
self._realized_ic = None
self.lake_root = lake_root or ""
self.market = market or "US"
def _load_realized_ic(self):
if self._realized_ic is None:
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(
self.trade_calendar.get_trade_step(), shift=-self.ic_label_horizon
)
pred = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
if isinstance(pred, pd.DataFrame):
pred = pred.iloc[:, 0]
self._realized_ic = realized_rankic_series(
pred, self.lake_root, self.market, label_horizon=self.ic_label_horizon
)
return self._realized_ic
def _gate_open(self, trade_start_time) -> bool:
ts = pd.Timestamp(trade_start_time)
if self._ic_gate is not None:
# precomputed gate series: look up the latest known decision date <= ts
known = self._ic_gate[self._ic_gate.index <= ts]
if len(known):
return bool(known.iloc[-1])
return True
if self.ic_min_rankic <= 0:
return True
realized = self._load_realized_ic()
limits = {
"ic_min_rankic": self.ic_min_rankic,
"ic_window": self.ic_window,
"ic_min_obs": self.ic_min_obs,
}
tripped, _reason, _trail = ic_circuit_breaker(realized, ts, limits)
return not tripped
def generate_trade_decision(self, execute_result=None):
trade_step = self.trade_calendar.get_trade_step()
trade_start_time, _ = self.trade_calendar.get_step_time(trade_step)
if not self._gate_open(trade_start_time):
return TradeDecisionWO([], self)
return super().generate_trade_decision(execute_result)
@@ -0,0 +1,201 @@
"""Fractional-Kelly dropout strategy for cross-sectional signals.
Sizing rule variant of ``qlib.contrib.strategy.signal_strategy.TopkDropoutStrategy``:
the topk/n_drop SELECTION is identical to the reference, but the buy size is
proportional to the score MAGNITUDE (edge) instead of equal-weight, capped at a
fraction ``cap_frac`` of the equal-weight notional so a single name cannot
over-concentrate the book.
``cap_frac`` is the fraction of the equal-weight per-name notional that a top
signal can deploy at most (e.g. 0.5 = at most half the equal-weight size).
Names whose score is below the median of the buy set get a proportionally
smaller slice; the residual stays in cash (that is the point of the rule:
throw away less edge per name, deploy less capital when conviction is low).
"""
from __future__ import annotations
from typing import List
import numpy as np
import pandas as pd
from qlib.backtest import Order
from qlib.backtest.decision import OrderDir, TradeDecisionWO
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
__all__ = ["FractionalKellyDropoutStrategy"]
DEFAULT_CAP_FRAC = 0.5
class FractionalKellyDropoutStrategy(TopkDropoutStrategy):
"""TopkDropout selection with score-magnitude (fractional-Kelly) sizing.
Parameters
----------
topk, n_drop, method_sell, method_buy, hold_thresh, only_tradable,
forbid_all_trade_at_limit : same as ``TopkDropoutStrategy``.
cap_frac : max buy notional as a fraction of the equal-weight notional.
"""
def __init__(self, *, topk, n_drop, cap_frac: float = DEFAULT_CAP_FRAC, **kwargs):
super().__init__(topk=topk, n_drop=n_drop, **kwargs)
self.cap_frac = cap_frac
def generate_trade_decision(self, execute_result=None):
import copy
trade_step = self.trade_calendar.get_trade_step()
trade_start_time, trade_end_time = self.trade_calendar.get_step_time(trade_step)
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(trade_step, shift=1)
pred_score = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
if isinstance(pred_score, pd.DataFrame):
pred_score = pred_score.iloc[:, 0]
if pred_score is None:
return TradeDecisionWO([], self)
if self.only_tradable:
def get_first_n(li, n, reverse=False):
cur_n = 0
res = []
for si in reversed(li) if reverse else li:
if self.trade_exchange.is_stock_tradable(
stock_id=si, start_time=trade_start_time, end_time=trade_end_time
):
res.append(si)
cur_n += 1
if cur_n >= n:
break
return res[::-1] if reverse else res
def get_last_n(li, n):
return get_first_n(li, n, reverse=True)
def filter_stock(li):
return [
si
for si in li
if self.trade_exchange.is_stock_tradable(
stock_id=si, start_time=trade_start_time, end_time=trade_end_time
)
]
else:
def get_first_n(li, n):
return list(li)[:n]
def get_last_n(li, n):
return list(li)[-n:]
def filter_stock(li):
return li
current_temp: "object" = copy.deepcopy(self.trade_position)
sell_order_list: List[Order] = []
buy_order_list: List[Order] = []
cash = current_temp.get_cash()
current_stock_list = current_temp.get_stock_list()
last = pred_score.reindex(current_stock_list).sort_values(ascending=False).index
if self.method_buy == "top":
today = get_first_n(
pred_score[~pred_score.index.isin(last)].sort_values(ascending=False).index,
self.n_drop + self.topk - len(last),
)
elif self.method_buy == "random":
topk_candi = get_first_n(pred_score.sort_values(ascending=False).index, self.topk)
candi = list(filter(lambda x: x not in last, topk_candi))
n = self.n_drop + self.topk - len(last)
try:
today = np.random.choice(candi, n, replace=False)
except ValueError:
today = candi
else:
raise NotImplementedError(f"This type of input is not supported")
comb = pred_score.reindex(last.union(pd.Index(today))).sort_values(ascending=False).index
if self.method_sell == "bottom":
sell = last[last.isin(get_last_n(comb, self.n_drop))]
elif self.method_sell == "random":
candi = filter_stock(last)
try:
sell = pd.Index(np.random.choice(candi, self.n_drop, replace=False) if len(last) else [])
except ValueError:
sell = candi
else:
raise NotImplementedError(f"This type of input is not supported")
buy = today[: len(sell) + self.topk - len(last)]
for code in current_stock_list:
if not self.trade_exchange.is_stock_tradable(
stock_id=code,
start_time=trade_start_time,
end_time=trade_end_time,
direction=None if self.forbid_all_trade_at_limit else OrderDir.SELL,
):
continue
if code in sell:
time_per_step = self.trade_calendar.get_freq()
if current_temp.get_stock_count(code, bar=time_per_step) < self.hold_thresh:
continue
sell_amount = current_temp.get_stock_amount(code=code)
sell_order = Order(
stock_id=code,
amount=sell_amount,
start_time=trade_start_time,
end_time=trade_end_time,
direction=Order.SELL,
)
if self.trade_exchange.check_order(sell_order):
sell_order_list.append(sell_order)
trade_val, trade_cost, trade_price = self.trade_exchange.deal_order(
sell_order, position=current_temp
)
cash += trade_val - trade_cost
if len(buy) == 0:
return TradeDecisionWO(sell_order_list, self)
# ---- fractional-Kelly sizing --------------------------------------
# equal-weight notional (reference baseline)
eq_notional = cash * self.risk_degree / len(buy)
buy_scores = pred_score.reindex(buy).astype(float)
lo, hi = buy_scores.min(), buy_scores.max()
if hi == lo:
w = pd.Series(1.0, index=buy_scores.index)
else:
w = (buy_scores - lo) / (hi - lo) # [0,1] edge magnitude
w = w.clip(lower=0.0)
w_max = w.max()
w = w / w_max if w_max > 0 else w # max == 1.0
for code in buy:
if not self.trade_exchange.is_stock_tradable(
stock_id=code,
start_time=trade_start_time,
end_time=trade_end_time,
direction=None if self.forbid_all_trade_at_limit else OrderDir.BUY,
):
continue
buy_price = self.trade_exchange.get_deal_price(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time, direction=OrderDir.BUY
)
notional = eq_notional * min(self.cap_frac, float(w.get(code, 0.0)))
buy_amount = notional / buy_price
factor = self.trade_exchange.get_factor(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time
)
buy_amount = self.trade_exchange.round_amount_by_trade_unit(buy_amount, factor)
buy_order = Order(
stock_id=code,
amount=buy_amount,
start_time=trade_start_time,
end_time=trade_end_time,
direction=Order.BUY,
)
buy_order_list.append(buy_order)
return TradeDecisionWO(sell_order_list + buy_order_list, self)
@@ -0,0 +1,361 @@
"""Long-short Top-K strategy for cross-sectional signals.
Each day the strategy ranks the cross-section by prediction score and rebalances
to an equal-weight two-sided book: the ``topk`` highest-ranked names go long and
the ``k_short`` lowest-ranked names go short. Net-new shorts are opened by
selling beyond current holdings, which requires a short-aware exchange such as
``tac_qlib.contrib.backtest.tradeac_exchange.TradeACExchange`` with
``allow_short=True`` (borrow limits, margin requirements and borrow fees are
enforced there, not here).
Sizing deploys ``equity * risk_degree`` as gross notional split evenly across
all long and short legs, so the book is approximately market neutral.
``allow_short=False`` disables the short side entirely (long-only ``topk``).
Short eligibility can be restricted further, with static or dynamic gates:
``short_whitelist`` limits shorts to an explicit symbol set; ``short_vol_top_pct``
requires a candidate's trailing realized volatility to rank in the top fraction
of that day's cross-section; ``short_max_mom`` (falling-knife filter) only
allows shorting names whose own trailing momentum is at/below a threshold;
``short_regime_sma`` disables shorts entirely while the benchmark trades above
its moving average (risk-on). Borrow availability itself is enforced by the
exchange (``borrowable`` whitelist / per-symbol caps via ``TradeACExchange``).
Wired into qrun workflows like any ``BaseStrategy`` (see ``PortAnaRecord``
config). Mirrors the API usage of qlib's ``TopkDropoutStrategy``: ``Order`` /
``OrderDir`` from ``qlib.backtest.decision``, ``trade_calendar`` /
``trade_exchange`` / ``trade_position`` injected by the backtest executor.
"""
from __future__ import annotations
import copy
from typing import Dict, List, Optional
import pandas as pd
from qlib.backtest import Order
from qlib.backtest.decision import OrderDir, TradeDecisionWO
from qlib.contrib.strategy.signal_strategy import BaseSignalStrategy
from qlib.log import get_module_logger
__all__ = ["LongShortTopkStrategy"]
class LongShortTopkStrategy(BaseSignalStrategy):
"""Equal-weight long-short Top-K strategy over a cross-sectional signal.
Parameters
----------
topk : number of long legs (highest-ranked names).
k_short : number of short legs (lowest-ranked names).
hold_thresh : minimum holding days before a leg may be closed/reduced.
only_tradable : only select candidates tradable on the trade date.
rebalance_tol : skip rebalances smaller than this fraction of a leg's
target notional (turnover control).
allow_short : enable/disable the short side. With ``False`` the bottom-ranked
legs are dropped and the book is long-only ``topk``; pair with
``allow_short=False`` on the exchange for a fully borrow-free run.
Legacy alias ``enable_short`` is accepted.
short_whitelist : optional list of symbols eligible for shorting; candidates
outside the list are skipped (``None`` = all names eligible).
short_vol_window : trailing window (trading days) for realized-vol estimation.
short_vol_top_pct : if set, a short candidate's trailing realized volatility
must rank at or above this percentile of that day's cross-section
(e.g. ``0.5`` = only the more volatile half may be shorted). Candidates
without measurable vol are never shorted.
short_mom_window : trailing window (trading days) for the candidate momentum
used by the falling-knife gate.
short_max_mom : if set, a candidate's trailing ``short_mom_window``-day return
must be <= this value to be shortable (e.g. ``0.0`` = only short names
that are actually falling). Candidates without measurable momentum are
never shorted.
short_regime_symbol : benchmark symbol for the regime gate (default SPY).
short_regime_sma : if set, shorts are only allowed on days where the regime
symbol's last close (strictly before the execution bar) is BELOW its
``short_regime_sma``-day moving average — i.e. shorts are disabled in
risk-on regimes and enabled in drawdowns.
"""
def __init__(
self,
*,
signal=None,
topk: int = 4,
k_short: int = 2,
hold_thresh: int = 1,
only_tradable: bool = True,
rebalance_tol: float = 0.05,
allow_short: Optional[bool] = None,
enable_short: Optional[bool] = None,
short_whitelist: Optional[List[str]] = None,
short_vol_window: int = 20,
short_vol_top_pct: Optional[float] = None,
short_mom_window: int = 20,
short_max_mom: Optional[float] = None,
short_regime_symbol: str = "SPY",
short_regime_sma: Optional[int] = None,
risk_degree: float = 0.95,
trade_exchange=None,
level_infra=None,
common_infra=None,
**kwargs,
):
super().__init__(
signal=signal,
risk_degree=risk_degree,
trade_exchange=trade_exchange,
level_infra=level_infra,
common_infra=common_infra,
**kwargs,
)
if allow_short is None:
allow_short = True if enable_short is None else bool(enable_short)
self.allow_short = bool(allow_short)
self.topk = topk
self.k_short = k_short
self.hold_thresh = hold_thresh
self.only_tradable = only_tradable
self.rebalance_tol = rebalance_tol
self.short_whitelist = set(short_whitelist) if short_whitelist is not None else None
if not 0 < float(short_vol_window) <= 1000:
raise ValueError(f"short_vol_window must be in (0, 1000], got {short_vol_window}")
self.short_vol_window = int(short_vol_window)
if short_vol_top_pct is not None and not 0.0 < float(short_vol_top_pct) <= 1.0:
raise ValueError(f"short_vol_top_pct must be in (0, 1], got {short_vol_top_pct}")
self.short_vol_top_pct = None if short_vol_top_pct is None else float(short_vol_top_pct)
if not 0 < float(short_mom_window) <= 1000:
raise ValueError(f"short_mom_window must be in (0, 1000], got {short_mom_window}")
self.short_mom_window = int(short_mom_window)
self.short_max_mom = None if short_max_mom is None else float(short_max_mom)
self.short_regime_symbol = str(short_regime_symbol)
if short_regime_sma is not None and not 1 < int(short_regime_sma) <= 1000:
raise ValueError(f"short_regime_sma must be in (1, 1000], got {short_regime_sma}")
self.short_regime_sma = None if short_regime_sma is None else int(short_regime_sma)
# per-day caches (keyed by trade date)
self._vol_cache_key: Optional[str] = None
self._vol_cache_val: Dict[str, Dict[str, float]] = {}
self._regime_cache: Dict[str, bool] = {}
# ------------------------------------------------------------------ utils
def _is_tradable(self, code, start, end) -> bool:
if not self.only_tradable:
return True
try:
return self.trade_exchange.is_stock_tradable(stock_id=code, start_time=start, end_time=end)
except TypeError:
return True
def _mark_price(self, code, start, end) -> Optional[float]:
try:
px = self.trade_exchange.get_deal_price(
stock_id=code, start_time=start, end_time=end, direction=OrderDir.BUY
)
except (KeyError, ValueError):
return None
if px is None or px != px or px <= 0:
return None
return float(px)
def _day_stats(self, codes: List[str], trade_start) -> Dict[str, Dict[str, float]]:
"""Per-day cross-sectional stats used by the dynamic short gates.
For each code, returns ``{"vol_rank": r}`` (percentile of trailing
realized vol over ``short_vol_window`` bars across that day's
cross-section) when the vol gate is on, and ``{"mom": m}`` (trailing
``short_mom_window``-bar return) when the falling-knife gate is on.
All series end on the last bar strictly BEFORE the execution bar (no
lookahead). Codes without measurable data are simply absent — such
candidates are never shorted (fail-closed).
"""
if self.short_vol_top_pct is None and self.short_max_mom is None:
return {}
key = str(pd.Timestamp(trade_start))
if self._vol_cache_key == key:
return self._vol_cache_val
out: Dict[str, Dict[str, float]] = {}
try:
from qlib.data import D
end = pd.Timestamp(trade_start)
buf = max(self.short_vol_window, self.short_mom_window) * 3 + 30
df = D.features(
sorted(codes),
["$close"],
start_time=end - pd.Timedelta(days=buf),
end_time=end - pd.Timedelta(days=1),
)
close = df["$close"].unstack(level="instrument") if isinstance(df.index, pd.MultiIndex) else df["$close"]
if self.short_vol_top_pct is not None:
vol = close.pct_change().rolling(self.short_vol_window).std().iloc[-1]
for code, rank in vol.rank(pct=True).dropna().items():
out.setdefault(str(code), {})["vol_rank"] = float(rank)
if self.short_max_mom is not None:
w = min(self.short_mom_window, len(close) - 1)
mom = close.iloc[-1] / close.iloc[-(w + 1)] - 1
for code, m in mom.items():
if m == m:
out.setdefault(str(code), {})["mom"] = float(m)
except Exception as e: # noqa: BLE001 - degrade to fail-closed (no shorts)
get_module_logger(self.__class__.__name__).warning(
f"short gates unavailable ({type(e).__name__}: {e}); no shorts this step"
)
self._vol_cache_key, self._vol_cache_val = key, out
return out
def _regime_ok(self, trade_start) -> bool:
"""True when shorting is allowed by the benchmark-regime gate.
With ``short_regime_sma`` set, shorts are permitted only while the
regime symbol's last close strictly before the execution bar sits below
its moving average (risk-off). Data failure fails closed (no shorts).
"""
if self.short_regime_sma is None:
return True
key = str(pd.Timestamp(trade_start))
cached = self._regime_cache.get(key)
if cached is not None:
return cached
ok = False
try:
from qlib.data import D
end = pd.Timestamp(trade_start)
df = D.features(
[self.short_regime_symbol],
["$close"],
start_time=end - pd.Timedelta(days=int(self.short_regime_sma * 3 + 30)),
end_time=end - pd.Timedelta(days=1),
)
s = df["$close"]
if isinstance(s.index, pd.MultiIndex):
s = s.droplevel("instrument")
sma = s.rolling(self.short_regime_sma).mean().iloc[-1]
px = s.iloc[-1]
ok = bool(px < sma)
except Exception as e: # noqa: BLE001 - degrade to fail-closed (no shorts)
get_module_logger(self.__class__.__name__).warning(
f"regime gate unavailable ({type(e).__name__}: {e}); no shorts this step"
)
self._regime_cache[key] = ok
return ok
# ------------------------------------------------------------- decision
def generate_trade_decision(self, execute_result=None):
trade_step = self.trade_calendar.get_trade_step()
trade_start, trade_end = self.trade_calendar.get_step_time(trade_step)
pred_start, pred_end = self.trade_calendar.get_step_time(trade_step, shift=1)
pred_score = self.signal.get_signal(start_time=pred_start, end_time=pred_end)
if isinstance(pred_score, pd.DataFrame):
pred_score = pred_score.iloc[:, 0]
if pred_score is None or len(pred_score) == 0:
return TradeDecisionWO([], self)
pred_score = pred_score.dropna()
if pred_score.empty:
return TradeDecisionWO([], self)
time_per_step = self.trade_calendar.get_freq()
current_temp = copy.deepcopy(self.trade_position)
# ---- signed current holdings ---------------------------------------
cur_amount: Dict[str, float] = {}
for code in current_temp.get_stock_list():
amt = float(current_temp.get_stock_amount(code))
if abs(amt) > 1e-6:
cur_amount[code] = amt
# ---- targets: top-k long, bottom-k_short short ----------------------
ranked = list(pred_score.sort_values(ascending=False).index)
longs: List[str] = []
for code in ranked:
if len(longs) >= self.topk:
break
if self._is_tradable(code, trade_start, trade_end):
longs.append(code)
shorts: List[str] = []
if self.allow_short and self._regime_ok(trade_start):
stats = self._day_stats(list(ranked), trade_start)
for code in reversed(ranked):
if len(shorts) >= self.k_short:
break
if code in longs:
continue
if not self._is_tradable(code, trade_start, trade_end):
continue
if self.short_whitelist is not None and code not in self.short_whitelist:
continue
st = stats.get(code)
if self.short_vol_top_pct is not None:
rank = None if st is None else st.get("vol_rank")
if rank is None or rank < self.short_vol_top_pct:
continue
if self.short_max_mom is not None:
mom = None if st is None else st.get("mom")
if mom is None or mom > self.short_max_mom:
continue
shorts.append(code)
# ---- marks & equity --------------------------------------------------
marks: Dict[str, float] = {}
for code in set(cur_amount) | set(longs) | set(shorts):
px = self._mark_price(code, trade_start, trade_end)
if px is not None:
marks[code] = px
equity = current_temp.get_cash()
for code, amt in cur_amount.items():
if code in marks:
equity += amt * marks[code]
if equity <= 0:
return TradeDecisionWO([], self)
n_legs = len([c for c in longs if c in marks]) + len([c for c in shorts if c in marks])
if n_legs == 0:
return TradeDecisionWO([], self)
per_leg = equity * self.risk_degree / n_legs
target_signed: Dict[str, float] = {}
for code in longs:
if code in marks:
target_signed[code] = per_leg / marks[code]
for code in shorts:
if code in marks:
target_signed[code] = -(per_leg / marks[code])
# ---- order generation -------------------------------------------------
sell_orders: List[Order] = []
buy_orders: List[Order] = []
def submit(code: str, amount: float, direction: int) -> None:
factor = self.trade_exchange.get_factor(stock_id=code, start_time=trade_start, end_time=trade_end)
amount = self.trade_exchange.round_amount_by_trade_unit(amount, factor)
if amount <= 1e-6:
return
o = Order(stock_id=code, amount=amount, start_time=trade_start, end_time=trade_end, direction=direction)
if self.trade_exchange.check_order(o):
(buy_orders if direction == Order.BUY else sell_orders).append(o)
# close holdings that are no longer targeted (frees cash / unwinds shorts)
for code, amt in cur_amount.items():
if code in target_signed:
continue
if marks.get(code) is None:
continue
if current_temp.get_stock_count(code, bar=time_per_step) < self.hold_thresh:
continue
submit(code, abs(amt), Order.SELL if amt > 0 else Order.BUY)
# rebalance targeted legs toward their signed target quantity
for code, tgt in target_signed.items():
cur = cur_amount.get(code, 0.0)
delta = tgt - cur
if abs(delta * marks[code]) < max(self.rebalance_tol * per_leg, 1.0):
continue
if delta > 0:
submit(code, delta, Order.BUY)
else:
if cur > 0 and current_temp.get_stock_count(code, bar=time_per_step) < self.hold_thresh:
continue
submit(code, -delta, Order.SELL)
return TradeDecisionWO(sell_orders + buy_orders, self)
@@ -1,215 +1,231 @@
"""Regime-gate TopkDropout strategy.
"""HMM-regime overlay TopkDropout strategy.
Subclass of ``qlib.contrib.strategy.signal_strategy.TopkDropoutStrategy`` that
holds the book (issues NO orders) while a regime detector says the market is in
an unfavorable state. When the gate is open it behaves exactly like the
reference TopkDropoutStrategy.
Regime-gate overlay on ``qlib.contrib.strategy.signal_strategy.TopkDropoutStrategy``:
selection and sizing are identical to the reference, but a name is only BOUGHT
(entry gate) when its per-symbol HMM regime posterior ``sp_hmm_p_regime1`` on
the signal date is >= ``regime_threshold``; otherwise it is held in cash instead
of being opened.
Three detector types are supported (all causal — no lookahead):
The regime posterior is read from the lake feature provider on the fly via
``qlib.data.D.features`` (field ``$sp_hmm_p_regime1``) for the signal window, so
no regime column needs to enter the model's ``feature_fields`` — the gate is a
pure overlay (book ch.01: regime flags regressed as model features, survived
only as an overlay). The HMM itself was fit with ``fit_end=<train end>`` when
the lake features were backfilled, so there is no lookahead.
* ``dispersion``: cross-sectional standard deviation of 22-day rolling returns
across the universe. Gate closes when CS dispersion < threshold (low
dispersion means the spread between winners and losers is too narrow for
TopkDropout to exploit).
* ``vol``: cross-sectional mean of 22-day rolling realized volatility. Gate
closes when avg vol is outside a band ``[vol_low, vol_high]`` (strategy
needs moderate vol — too calm or too turbulent both hurt).
* ``hmm``: pre-computed HMM posterior for regime 1 (``sp_hmm_p_regime1``).
Gate closes when posterior < threshold (model is not confident the calm
regime is active).
The gate is provided as a precomputed ``pd.Series`` of booleans indexed by
datetime (True = trade allowed). The companion ``compute_regime_gate``
function builds this series from lake bars; call it once before backtesting
and pass the result as the ``regime_gate`` parameter.
Names already held are NOT force-sold when the regime turns unfavourable
(entry gate only, matching the queue-10 design).
"""
from __future__ import annotations
from typing import List
import numpy as np
import pandas as pd
from qlib.backtest.decision import TradeDecisionWO
from qlib.backtest import Order
from qlib.backtest.decision import OrderDir, TradeDecisionWO
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
__all__ = ["RegimeGateTopkDropoutStrategy", "compute_regime_gate"]
try:
from qlib.data import D
except ImportError: # pragma: no cover - qlib always present in this stack
D = None
__all__ = ["RegimeGateDropoutStrategy"]
DEFAULT_REGIME_THRESHOLD = 0.5
REGIME_FIELD = "$sp_hmm_p_regime1"
class RegimeGateTopkDropoutStrategy(TopkDropoutStrategy):
"""TopkDropout with a regime-gate circuit breaker.
class RegimeGateDropoutStrategy(TopkDropoutStrategy):
"""TopkDropout with an HMM-regime entry gate on buy candidates.
Parameters
----------
topk, n_drop, method_sell, method_buy, hold_thresh, only_tradable,
forbid_all_trade_at_limit : same as ``TopkDropoutStrategy``.
regime_gate : pd.Series — precomputed per-date gate (bool indexed by
datetime). True = trade allowed, False = no orders. Missing dates
default to open (trade allowed).
regime_threshold : minimum ``sp_hmm_p_regime1`` posterior required to open a
new position (default 0.5).
"""
def __init__(self, *, regime_gate=None, **kwargs):
super().__init__(**kwargs)
self._regime_gate = regime_gate
def __init__(self, *, topk, n_drop, regime_threshold: float = DEFAULT_REGIME_THRESHOLD, **kwargs):
super().__init__(topk=topk, n_drop=n_drop, **kwargs)
self.regime_threshold = regime_threshold
def _gate_open(self, trade_start_time) -> bool:
if self._regime_gate is None:
return True
ts = pd.Timestamp(trade_start_time)
known = self._regime_gate[self._regime_gate.index <= ts]
if len(known):
return bool(known.iloc[-1])
return True # default open if no history yet
def _regime_for(self, codes, pred_start, pred_end) -> pd.Series:
"""Return {code: sp_hmm_p_regime1} for the signal window (last day)."""
if D is None:
return pd.Series(dtype=float)
try:
df = D.features(list(codes), [REGIME_FIELD], start_time=pred_start, end_time=pred_end, freq="day")
except Exception: # noqa: BLE001 - a regime read failure should gate open, not crash
return pd.Series(dtype=float)
if df is None or len(df) == 0:
return pd.Series(dtype=float)
# df index is MultiIndex (datetime, instrument); take the last day's values
df = df.reset_index()
ts_col = "datetime" if "datetime" in df.columns else df.columns[0]
sym_col = "instrument" if "instrument" in df.columns else df.columns[1]
last_ts = df[ts_col].max()
last = df[df[ts_col] == last_ts]
out = {}
for _, row in last.iterrows():
sym = str(row[sym_col]).split("/")[-1].upper()
val = row.iloc[-1]
out[sym] = float(val) if val == val else np.nan
return pd.Series(out)
def generate_trade_decision(self, execute_result=None):
import copy
trade_step = self.trade_calendar.get_trade_step()
trade_start_time, _ = self.trade_calendar.get_step_time(trade_step)
if not self._gate_open(trade_start_time):
trade_start_time, trade_end_time = self.trade_calendar.get_step_time(trade_step)
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(trade_step, shift=1)
pred_score = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
if isinstance(pred_score, pd.DataFrame):
pred_score = pred_score.iloc[:, 0]
if pred_score is None:
return TradeDecisionWO([], self)
return super().generate_trade_decision(execute_result)
if self.only_tradable:
# ---------------------------------------------------------------------------
# Precomputation helper
# ---------------------------------------------------------------------------
def get_first_n(li, n, reverse=False):
cur_n = 0
res = []
for si in reversed(li) if reverse else li:
if self.trade_exchange.is_stock_tradable(
stock_id=si, start_time=trade_start_time, end_time=trade_end_time
):
res.append(si)
cur_n += 1
if cur_n >= n:
break
return res[::-1] if reverse else res
def compute_regime_gate(
detector: str,
threshold: float = 0.0,
*,
lake_root: str = "",
market: str = "US",
start: str = "2015-01-03",
end: str = "2026-08-19",
vol_low: float = 0.0,
vol_high: float = 999.0,
hmm_field: str = "sp_hmm_p_regime1",
) -> pd.Series:
"""Build a per-date regime gate series from lake bars.
def get_last_n(li, n):
return get_first_n(li, n, reverse=True)
Parameters
----------
detector : str — ``"dispersion"``, ``"vol"``, or ``"hmm"``.
threshold : float — for ``dispersion``: min CS dispersion to allow trading.
For ``hmm``: min HMM posterior to allow trading.
Ignored for ``vol`` (uses ``vol_low``/``vol_high`` band instead).
lake_root, market : str — lake location.
start, end : str — date window.
vol_low, vol_high : float — annualized vol band for the ``vol`` detector.
hmm_field : str — HMM feature column name for the ``hmm`` detector.
def filter_stock(li):
return [
si
for si in li
if self.trade_exchange.is_stock_tradable(
stock_id=si, start_time=trade_start_time, end_time=trade_end_time
)
]
Returns
-------
pd.Series — bool, indexed by datetime. True = trade allowed.
"""
from tac_qlib.data.config import LakeConfig, resolve_lake_root
else:
cfg = LakeConfig(resolve_lake_root(lake_root or None), market)
symbols = _universe_symbols(cfg)
close_df, vol_df = _load_daily_bars(symbols, cfg, start, end)
if close_df.empty:
return pd.Series(dtype=bool)
def get_first_n(li, n):
return list(li)[:n]
if detector == "dispersion":
return _dispersion_gate(close_df, threshold)
elif detector == "vol":
return _vol_gate(close_df, vol_low, vol_high)
elif detector == "hmm":
return _hmm_gate(cfg, symbols, threshold, start, end, hmm_field)
else:
raise ValueError(f"Unknown detector: {detector!r}")
def get_last_n(li, n):
return list(li)[-n:]
def filter_stock(li):
return li
def _universe_symbols(cfg) -> list:
"""Read symbols from the lake symbols.parquet."""
import pathlib
current_temp: "object" = copy.deepcopy(self.trade_position)
sell_order_list: List[Order] = []
buy_order_list: List[Order] = []
cash = current_temp.get_cash()
current_stock_list = current_temp.get_stock_list()
last = pred_score.reindex(current_stock_list).sort_values(ascending=False).index
sp = cfg.lake_root / "symbols.parquet"
if sp.exists():
df = pd.read_parquet(sp)
col = "symbol" if "symbol" in df.columns else df.columns[0]
return sorted(df[col].astype(str).str.upper().tolist())
return []
def _load_daily_bars(symbols, cfg, start, end):
"""Load daily close prices for all symbols into a wide DataFrame."""
closes = {}
vols = {}
for sym in symbols:
p = cfg.bar_path("1d", sym)
if not p.exists():
continue
try:
df = pd.read_parquet(p)
except Exception:
continue
if not len(df):
continue
tcol = df["t"] if "t" in df.columns else df["date"]
ts = pd.to_datetime(tcol)
df = df.assign(_t=ts).set_index("_t").sort_index()
df = df.loc[start:end]
if len(df) < 22:
continue
closes[sym] = df["c"]
if "v" in df.columns:
vols[sym] = df["v"]
close_df = pd.DataFrame(closes)
vol_df = pd.DataFrame(vols) if vols else None
return close_df, vol_df
def _dispersion_gate(close_df, threshold):
"""Cross-sectional dispersion of 22-day rolling returns."""
if close_df.empty or close_df.shape[1] < 2:
return pd.Series(dtype=bool)
ret = close_df.pct_change(22)
cs_disp = ret.std(axis=1)
gate = cs_disp >= threshold
gate.iloc[:22] = True # warmup: allow trading
return gate
def _vol_gate(close_df, vol_low, vol_high):
"""Cross-sectional mean of 22-day rolling realized vol."""
if close_df.empty or close_df.shape[1] < 2:
return pd.Series(dtype=bool)
import numpy as np
log_ret = np.log(close_df / close_df.shift(1))
rv22 = log_ret.rolling(22).std() * (252 ** 0.5)
cs_mean_vol = rv22.mean(axis=1)
gate = (cs_mean_vol >= vol_low) & (cs_mean_vol <= vol_high)
gate.iloc[:22] = True # warmup
return gate
def _hmm_gate(cfg, symbols, threshold, start, end, hmm_field):
"""HMM regime posterior gate from persisted SP features."""
feat_root = cfg.lake_root / "features"
all_posteriors = {}
for sym in symbols:
# check both ta and sp family paths
for family in ("sp", "ta"):
p = feat_root / f"market=US" / f"timeframe=1d" / f"family={family}" / f"symbol={sym}.parquet"
if not p.exists():
continue
if self.method_buy == "top":
today = get_first_n(
pred_score[~pred_score.index.isin(last)].sort_values(ascending=False).index,
self.n_drop + self.topk - len(last),
)
elif self.method_buy == "random":
topk_candi = get_first_n(pred_score.sort_values(ascending=False).index, self.topk)
candi = list(filter(lambda x: x not in last, topk_candi))
n = self.n_drop + self.topk - len(last)
try:
df = pd.read_parquet(p)
except Exception:
today = np.random.choice(candi, n, replace=False)
except ValueError:
today = candi
else:
raise NotImplementedError(f"This type of input is not supported")
comb = pred_score.reindex(last.union(pd.Index(today))).sort_values(ascending=False).index
if self.method_sell == "bottom":
sell = last[last.isin(get_last_n(comb, self.n_drop))]
elif self.method_sell == "random":
candi = filter_stock(last)
try:
sell = pd.Index(np.random.choice(candi, self.n_drop, replace=False) if len(last) else [])
except ValueError:
sell = candi
else:
raise NotImplementedError(f"This type of input is not supported")
buy = today[: len(sell) + self.topk - len(last)]
# ---- regime gate -----------------------------------------------------
if buy:
regime = self._regime_for(buy, pred_start_time, pred_end_time)
gated = [c for c in buy if regime.get(c, np.nan) >= self.regime_threshold]
else:
gated = []
for code in current_stock_list:
if not self.trade_exchange.is_stock_tradable(
stock_id=code,
start_time=trade_start_time,
end_time=trade_end_time,
direction=None if self.forbid_all_trade_at_limit else OrderDir.SELL,
):
continue
if hmm_field not in df.columns:
if code in sell:
time_per_step = self.trade_calendar.get_freq()
if current_temp.get_stock_count(code, bar=time_per_step) < self.hold_thresh:
continue
sell_amount = current_temp.get_stock_amount(code=code)
sell_order = Order(
stock_id=code,
amount=sell_amount,
start_time=trade_start_time,
end_time=trade_end_time,
direction=Order.SELL,
)
if self.trade_exchange.check_order(sell_order):
sell_order_list.append(sell_order)
trade_val, trade_cost, trade_price = self.trade_exchange.deal_order(
sell_order, position=current_temp
)
cash += trade_val - trade_cost
if len(gated) == 0:
return TradeDecisionWO(sell_order_list, self)
value = cash * self.risk_degree / len(gated)
for code in gated:
if not self.trade_exchange.is_stock_tradable(
stock_id=code,
start_time=trade_start_time,
end_time=trade_end_time,
direction=None if self.forbid_all_trade_at_limit else OrderDir.BUY,
):
continue
tcol = df["t"] if "t" in df.columns else df["date"]
ts = pd.to_datetime(tcol)
s = pd.Series(df[hmm_field].values, index=ts, name=sym)
s = s.loc[start:end].dropna()
if len(s) > 0:
all_posteriors[sym] = s
break
if not all_posteriors:
# no HMM features found — default open
idx = pd.date_range(start, end, freq="B")
return pd.Series(True, index=idx)
post_df = pd.DataFrame(all_posteriors)
cs_mean = post_df.mean(axis=1)
gate = cs_mean >= threshold
return gate
buy_price = self.trade_exchange.get_deal_price(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time, direction=OrderDir.BUY
)
buy_amount = value / buy_price
factor = self.trade_exchange.get_factor(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time
)
buy_amount = self.trade_exchange.round_amount_by_trade_unit(buy_amount, factor)
buy_order = Order(
stock_id=code,
amount=buy_amount,
start_time=trade_start_time,
end_time=trade_end_time,
direction=Order.BUY,
)
buy_order_list.append(buy_order)
return TradeDecisionWO(sell_order_list + buy_order_list, self)
@@ -1,301 +0,0 @@
"""Signal-quality gate TopkDropout strategy.
Subclass of ``qlib.contrib.strategy.signal_strategy.TopkDropoutStrategy`` that
holds the book (issues NO orders) when the model's recent prediction accuracy
is below a threshold. When the gate is open it behaves exactly like the
reference TopkDropoutStrategy.
Unlike the regime gate (which asks "is the market calm?"), the signal-quality
gate asks "are my predictions accurate?" — and works across ALL years.
The gate is provided as a precomputed ``pd.Series`` of booleans indexed by
datetime (True = trade allowed). The companion ``compute_signal_quality_gate``
function builds this series from a pred.pkl and lake bars; call it once before
backtesting and pass the result as the ``signal_quality_gate`` parameter.
"""
from __future__ import annotations
import numpy as np
import pandas as pd
from qlib.backtest.decision import TradeDecisionWO
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
__all__ = ["SignalQualityGateStrategy", "compute_signal_quality_gate"]
class SignalQualityGateStrategy(TopkDropoutStrategy):
"""TopkDropout with signal-quality gate overlay.
When ``lake_root`` is provided the gate is computed on-the-fly from the
signal (``<PRED>``) and close prices — no precomputed gate file needed.
This ensures the gate matches the model that is actually generating the
predictions (critical when the model is retrained each year).
Parameters
----------
topk, n_drop, method_sell, method_buy, hold_thresh, only_tradable,
forbid_all_trade_at_limit : same as ``TopkDropoutStrategy``.
signal_quality_gate : pd.Series — precomputed per-date gate (bool).
signal_quality_gate_path : str — path to pickled gate Series.
lake_root : str — lake root for on-the-fly gate computation (preferred).
gate_topk, gate_lookback, gate_threshold : int/float — gate params.
gate_start, gate_end : str — date window for loading close prices.
"""
def __init__(self, *, signal_quality_gate=None, signal_quality_gate_path=None,
lake_root=None, gate_topk=10, gate_lookback=5, gate_threshold=0.5,
gate_start="2015-01-03", gate_end="2026-08-19", **kwargs):
super().__init__(**kwargs)
self._sq_gate_computed = False
if signal_quality_gate is not None:
self._sq_gate = signal_quality_gate
self._sq_gate_computed = True
elif signal_quality_gate_path is not None:
import pickle
with open(signal_quality_gate_path, "rb") as f:
self._sq_gate = pickle.load(f)
self._sq_gate_computed = True
elif lake_root is not None:
self._sq_gate = None
self._lake_root = lake_root
self._gate_topk = gate_topk
self._gate_lookback = gate_lookback
self._gate_threshold = gate_threshold
self._gate_start = gate_start
self._gate_end = gate_end
else:
self._sq_gate = None
def _compute_gate_on_fly(self):
"""Compute gate from the signal (pred.pkl) and lake close prices."""
import pickle
from pathlib import Path
signal_path = self._signal
if not Path(signal_path).exists():
return
with open(signal_path, "rb") as f:
pred = pickle.load(f)
# Handle MultiIndex DataFrame -> unstack to wide
if isinstance(pred, pd.DataFrame) and isinstance(pred.index, pd.MultiIndex):
pred = pred.iloc[:, 0]
pred.index = pd.MultiIndex.from_arrays([
pd.to_datetime(pred.index.get_level_values(0)).normalize(),
pred.index.get_level_values(1)
])
pred = pred.unstack(level=1)
elif isinstance(pred, pd.DataFrame):
pred = pred.iloc[:, 0] if pred.shape[1] >= 1 else pred.squeeze()
pred.index = pd.to_datetime(pred.index).normalize()
# Load close prices from lake
close_df = _load_close_prices(self._lake_root, "US", self._gate_start, self._gate_end)
if close_df.empty:
return
ret_df = close_df.pct_change()
ret_df.index = pd.to_datetime(ret_df.index).normalize()
pred_dates = sorted(pred.index.unique())
if len(pred_dates) < 2:
self._sq_gate = pd.Series(True, index=pd.DatetimeIndex(pred_dates))
self._sq_gate_computed = True
return
hit_rates = {}
for i in range(1, len(pred_dates)):
day = pred_dates[i]
prev_day = pred_dates[i - 1]
try:
prev_scores = pred.loc[prev_day]
except KeyError:
continue
if isinstance(prev_scores, pd.DataFrame):
prev_scores = prev_scores.iloc[:, 0]
prev_scores = prev_scores.dropna().sort_values(ascending=False)
topk_syms = list(prev_scores.index[:self._gate_topk])
if day not in ret_df.index:
continue
today_ret = ret_df.loc[day]
topk_rets = today_ret.reindex(topk_syms).dropna()
if len(topk_rets) == 0:
continue
hit_rates[day] = (topk_rets > 0).sum() / len(topk_rets)
if not hit_rates:
self._sq_gate_computed = True
return
hr_series = pd.Series(hit_rates).sort_index()
rolling_hr = hr_series.rolling(self._gate_lookback, min_periods=1).mean()
gate = rolling_hr >= self._gate_threshold
gate.iloc[:self._gate_lookback] = True
self._sq_gate = gate
self._sq_gate_computed = True
def _gate_open(self, trade_start_time) -> bool:
if self._sq_gate is None:
return True
ts = pd.Timestamp(trade_start_time)
known = self._sq_gate[self._sq_gate.index <= ts]
if len(known):
return bool(known.iloc[-1])
return True # default open if no history yet
def generate_trade_decision(self, execute_result=None):
if not self._sq_gate_computed:
self._compute_gate_on_fly()
trade_step = self.trade_calendar.get_trade_step()
trade_start_time, _ = self.trade_calendar.get_step_time(trade_step)
if not self._gate_open(trade_start_time):
return TradeDecisionWO([], self)
return super().generate_trade_decision(execute_result)
# ---------------------------------------------------------------------------
# Precomputation helper
# ---------------------------------------------------------------------------
def compute_signal_quality_gate(
pred_path: str,
*,
lake_root: str = "",
market: str = "US",
topk: int = 10,
lookback: int = 5,
threshold: float = 0.5,
start: str = "2015-01-03",
end: str = "2026-08-19",
) -> pd.Series:
"""Build a per-date signal-quality gate series from a pred.pkl and lake bars.
For each day, checks whether the model's topk picks from the previous day
had positive returns. Computes a rolling hit rate over ``lookback`` days
and opens the gate when hit rate >= ``threshold``.
Parameters
----------
pred_path : str — path to pred.pkl (from rd_train / rd_predict).
lake_root, market : str — lake location (for loading close prices).
topk : int — number of top picks to track for hit rate.
lookback : int — rolling window for hit rate computation.
threshold : float — hit rate threshold to keep trading.
start, end : str — date window for loading prices.
Returns
-------
pd.Series — bool, indexed by datetime. True = trade allowed.
"""
from pathlib import Path
import pickle
# Load pred.pkl
with open(pred_path, "rb") as f:
pred = pickle.load(f)
# Handle MultiIndex DataFrame (datetime, instrument) -> unstack to wide
if isinstance(pred, pd.DataFrame) and isinstance(pred.index, pd.MultiIndex):
pred = pred.iloc[:, 0] # take score column as Series
pred.index = pd.MultiIndex.from_arrays([
pd.to_datetime(pred.index.get_level_values(0)).normalize(),
pred.index.get_level_values(1)
])
# Unstack to wide: dates x instruments
pred = pred.unstack(level=1)
elif isinstance(pred, pd.DataFrame):
pred = pred.iloc[:, 0] if pred.shape[1] >= 1 else pred.squeeze()
pred.index = pd.to_datetime(pred.index).normalize()
# Load close prices from lake
close_df = _load_close_prices(lake_root, market, start, end)
if close_df.empty:
return pd.Series(dtype=bool)
ret_df = close_df.pct_change()
ret_df.index = pd.to_datetime(ret_df.index).normalize()
# Get sorted unique prediction dates
pred_dates = sorted(pred.index.unique())
if len(pred_dates) < 2:
return pd.Series(True, index=pd.DatetimeIndex(pred_dates))
# Compute hit rates
hit_rates = {}
for i in range(1, len(pred_dates)):
day = pred_dates[i]
# Get yesterday's topk
prev_day = pred_dates[i - 1]
try:
prev_scores = pred.loc[prev_day]
except KeyError:
continue
if isinstance(prev_scores, pd.DataFrame):
prev_scores = prev_scores.iloc[:, 0]
prev_scores = prev_scores.dropna().sort_values(ascending=False)
topk_syms = list(prev_scores.index[:topk])
# Get today's returns
if day not in ret_df.index:
continue
today_ret = ret_df.loc[day]
topk_rets = today_ret.reindex(topk_syms).dropna()
if len(topk_rets) == 0:
continue
hit_rates[day] = (topk_rets > 0).sum() / len(topk_rets)
if not hit_rates:
return pd.Series(dtype=bool)
hr_series = pd.Series(hit_rates).sort_index()
# Rolling hit rate
rolling_hr = hr_series.rolling(lookback, min_periods=1).mean()
# Gate is open when rolling hit rate >= threshold
gate = rolling_hr >= threshold
gate.iloc[:lookback] = True # warmup: allow trading
return gate
def _load_close_prices(lake_root, market, start, end):
"""Load daily close prices for all symbols into a wide DataFrame."""
from pathlib import Path
lake = Path(lake_root)
symbols_parquet = lake / "symbols.parquet"
if not symbols_parquet.exists():
return pd.DataFrame()
df = pd.read_parquet(symbols_parquet)
col = "symbol" if "symbol" in df.columns else df.columns[0]
symbols = sorted(df[col].astype(str).str.upper().tolist())
closes = {}
for sym in symbols:
p = lake / "market=US" / "timeframe=1d" / f"symbol={sym}.parquet"
if not p.exists():
continue
try:
bar = pd.read_parquet(p)
except Exception:
continue
if not len(bar):
continue
tcol = "t" if "t" in bar.columns else "date"
ts = pd.to_datetime(bar[tcol])
bar = bar.assign(_t=ts).set_index("_t").sort_index()
bar = bar.loc[start:end]
if len(bar) < 10:
continue
closes[sym] = bar["c"] if "c" in bar.columns else bar["close"]
return pd.DataFrame(closes)
@@ -37,11 +37,20 @@ class WeeklyRebalanceDropoutStrategy(TopkDropoutStrategy):
forbid_all_trade_at_limit : same as ``TopkDropoutStrategy``.
hold_band_pct : skip order for a name whose deviation from target weight is
below this fraction of the target (no-trade buffer band).
rebalance_every_n_weeks : rebalance every N ISO weeks instead of every week
(default 1 = weekly; 2 = biweekly). Ignored when
``rebalance_every_n_days`` is set.
rebalance_every_n_days : rebalance every N trading days (daily when N=1).
When set, overrides the weekly gating logic entirely.
"""
def __init__(self, *, topk, n_drop, hold_band_pct: float = DEFAULT_HOLD_BAND_PCT, **kwargs):
def __init__(self, *, topk, n_drop, hold_band_pct: float = DEFAULT_HOLD_BAND_PCT,
rebalance_every_n_weeks: int = 1,
rebalance_every_n_days: int = 0, **kwargs):
super().__init__(topk=topk, n_drop=n_drop, **kwargs)
self.hold_band_pct = hold_band_pct
self.rebalance_every_n_weeks = rebalance_every_n_weeks
self.rebalance_every_n_days = rebalance_every_n_days
@staticmethod
def _iso_week(ts) -> tuple:
@@ -53,13 +62,25 @@ class WeeklyRebalanceDropoutStrategy(TopkDropoutStrategy):
trade_step = self.trade_calendar.get_trade_step()
trade_start_time, trade_end_time = self.trade_calendar.get_step_time(trade_step)
cur_week = self._iso_week(trade_start_time)
prev_week = getattr(self, "_last_week", None)
self._last_week = cur_week
if self.rebalance_every_n_days > 0:
# daily gating: count trading steps since last rebalance
step_num = trade_step
if hasattr(self, "_last_rebal_step"):
if (step_num - self._last_rebal_step) < self.rebalance_every_n_days:
return TradeDecisionWO([], self)
self._last_rebal_step = step_num
else:
cur_week = self._iso_week(trade_start_time)
prev_week = getattr(self, "_last_week", None)
self._last_week = cur_week
if prev_week is not None and prev_week == cur_week:
# not the first trading day of this ISO week -> hold
return TradeDecisionWO([], self)
if prev_week is not None and prev_week == cur_week:
return TradeDecisionWO([], self)
if self.rebalance_every_n_weeks > 1:
week_num = cur_week[1]
if prev_week is not None and (week_num % self.rebalance_every_n_weeks) != 1:
return TradeDecisionWO([], self)
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(trade_step, shift=1)
pred_score = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
+8 -3
View File
@@ -61,6 +61,11 @@ UNKNOWN_FIELD_NAMES = ("factor", "change", "trade_unit", "suspend_flag")
#: columns in the parquet files that are not features
NON_FEATURE_COLUMNS = ("t", "date", "market", "timeframe", "symbol")
#: Feature-family partitions merged by ``LakeConfig.load_features`` and scanned
#: by the handler's field discovery. ``macro`` holds broadcast market-state
#: columns (see skills/tac-qlib-custom/examples/persist_macro_broadcast.py).
FEATURE_FAMILIES = ("ta", "sp", "macro")
def timeframe_for_freq(freq: str) -> str:
"""Map a qlib frequency (e.g. ``day``, ``1min``) to a lake timeframe (e.g. ``1d``)."""
@@ -113,12 +118,12 @@ class LakeConfig:
return self.features_dir(timeframe) / f"symbol={str(symbol).upper()}.parquet"
def load_features(self, timeframe: str, symbol: str) -> pd.DataFrame:
"""All feature columns for a symbol, merging the `family=ta` and
`family=sp` partitions by timestamp. Returns an empty frame when no
"""All feature columns for a symbol, merging the `family=ta|sp|macro`
partitions by timestamp. Returns an empty frame when no
feature files exist (legacy flat layout falls back transparently)."""
sym = str(symbol).upper()
frames = []
for family in ("ta", "sp"):
for family in FEATURE_FAMILIES:
p = self.features_dir(timeframe) / f"family={family}" / f"symbol={sym}.parquet"
if p.exists():
frames.append(pd.read_parquet(p))
@@ -1,44 +1,73 @@
# Signal-quality gate backtest for 2023
# -----------------------------------------------------------------------------
# QUEUE-01 — M2 reproduction: risk-adjusted 22d Sharpe drift (sp_sharpe_22).
#
# Hypothesis (book ch.01/ch.07, EVIDENCE#018 -> exp 30): adding the
# risk-adjusted 22d Sharpe drift feature (sp_sharpe_22) to the compact
# stochastic reference IMPROVES net portfolio performance (exp 30: net +6.53%
# IR 0.62 vs reference +2.13% IR 0.21) while rank metrics dip (RankIC 0.0576 vs
# 0.0663). exp 30 is a SINGLE clean-lake run, unreproduced -> HYPOTHESIS.
#
# Change vs exp-26 reference (EVIDENCE#015, run 21afc6af...): ONE feature added,
# feature_fields = compact set + sp_sharpe_22. Everything else byte-identical.
#
# Acceptance: net_ann_return > +2.13% AND net_IR > 0.21 (else HYPOTHESIS -> REFUTED).
# Run: rd_run_workflow config_path=<repo>/experiments/queue/workflows/q01_m2_sharpe22_repro.yaml \
# experiment_name=tac-rd-q01-m2-sharpe22-repro
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
{# Gate computed on-the-fly from signal + lake close prices #}
{%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead,sp_sharpe_22" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-sq-gate-2023"
default_exp_name: "tac-rd-q01-m2-sharpe22-repro"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.05
num_leaves: 15
n_estimators: 200
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.01
reg_lambda: 0.01
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
parallel: 5
dataset:
class: DatasetH
@@ -50,14 +79,14 @@ task:
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2023-12-29
fit_start_time: 2015-01-03
fit_end_time: 2023-01-03
end_time: 2026-08-10
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
feature_fields: "{{ FEATURES }}"
infer_processors:
- class: DropAllNaN
kwargs: {}
@@ -70,9 +99,9 @@ task:
- class: Fillna
kwargs: {}
segments:
train: [2015-01-03, 2022-09-01]
valid: [2022-09-03, 2023-01-03]
test: [2023-01-03, 2023-12-29]
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- class: SignalRecord
@@ -80,29 +109,25 @@ task:
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs: { ana_long_short: true, ann_scaler: 252 }
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: SignalQualityGateStrategy
module_path: tac_qlib.contrib.strategy.signal_quality_gate
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
lake_root: "{{ LAKE }}"
gate_topk: 10
gate_lookback: 5
gate_threshold: 0.5
gate_start: "2015-01-03"
gate_end: "2023-12-29"
topk: 10
n_drop: 1
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2023-01-03
end_time: 2023-12-29
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
@@ -112,4 +137,4 @@ task:
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
risk_analysis_freq: 1d
@@ -1,16 +1,16 @@
# -----------------------------------------------------------------------------
# Signal-quality gate: TopkDropout gated by rolling hit-rate of topk picks.
# ABLATION A (baseline): LightGBM with RankIC early-stopping on the 50-ETF SP-5d
# panel, using ALL 24 sp_* feature columns (ou,hmm,jump,har,trend,hurst,
# signature). Copy of the canonical workflow_lgb_sp5d_rankic.yaml with a
# distinct experiment name so the ablation runs are isolated.
#
# 1. Compute the gate: python precompute_signal_quality_gate.py <pred.pkl> <gate.pkl>
# 2. Run this workflow: rd_run_workflow config_path=<this yaml> experiment_name=<exp>
#
# The strategy loads the precomputed gate from signal_quality_gate_path.
# When hit rate >= threshold, trade; otherwise, go to cash.
# Run:
# rd_run_workflow config_path=tac-qlib/workflows/ablate_baseline_all_sp_fields.yaml \
# experiment_name=tac-rd-rank-ablate
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
{%- set GATE_PATH = "/app/experiments/book/data/signal_quality_gate/sq_gate_5d_0.50.pkl" %}
qlib_init:
provider_uri: "{{ LAKE }}"
@@ -40,22 +40,27 @@ qlib_init:
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-sq-gate"
default_exp_name: "tac-rd-rank-ablate"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
class: RankICLGBModel
module_path: tac_qlib.contrib.model.rank_gbdt
kwargs:
loss: mse
learning_rate: 0.05
num_leaves: 15
n_estimators: 200
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.01
reg_lambda: 0.01
reg_alpha: 0.1
reg_lambda: 1.0
seed: 42
dataset:
class: DatasetH
@@ -105,13 +110,12 @@ task:
kwargs:
config:
strategy:
class: SignalQualityGateStrategy
module_path: tac_qlib.contrib.strategy.signal_quality_gate
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
signal_quality_gate_path: "{{ GATE_PATH }}"
topk: 10
n_drop: 1
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
@@ -1,44 +1,67 @@
# Signal-quality gate backtest for 2026
# -----------------------------------------------------------------------------
# ABLATION B (generic-only): same panel/model as the baseline, but feature
# fields restricted to the model-free / generic stochastic-process families
# (jump,har,trend,hurst,signature). Drops the model-specific ou (OU/AR-1
# half-life) and hmm (2-state regime) families to test whether the generic
# families alone dominate the rank dimension.
#
# Run:
# rd_run_workflow config_path=tac-qlib/workflows/ablate_generic_only_sp_fields.yaml \
# experiment_name=tac-rd-rank-ablate
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
{# Gate computed on-the-fly from signal + lake close prices #}
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-sq-gate-2026"
default_exp_name: "tac-rd-rank-ablate"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
class: RankICLGBModel
module_path: tac_qlib.contrib.model.rank_gbdt
kwargs:
loss: mse
learning_rate: 0.05
num_leaves: 15
n_estimators: 200
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.01
reg_lambda: 0.01
reg_alpha: 0.1
reg_lambda: 1.0
seed: 42
dataset:
class: DatasetH
@@ -52,7 +75,7 @@ task:
start_time: 2015-01-03
end_time: 2026-08-10
fit_start_time: 2015-01-03
fit_end_time: 2026-01-03
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
@@ -80,24 +103,20 @@ task:
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs: { ana_long_short: true, ann_scaler: 252 }
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: SignalQualityGateStrategy
module_path: tac_qlib.contrib.strategy.signal_quality_gate
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
lake_root: "{{ LAKE }}"
gate_topk: 10
gate_lookback: 5
gate_threshold: 0.5
gate_start: "2015-01-03"
gate_end: "2026-08-19"
topk: 10
n_drop: 1
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
@@ -1,44 +1,74 @@
# Signal-quality gate backtest for 2025
# -----------------------------------------------------------------------------
# ISOLATION: multi-seed RankIC ensemble, ablate-B generic-only feature set.
#
# Isolates the ensemble effect on the SP-5d rank signal. Same panel, segments,
# history (full backfilled 2016+) and feature set as the exp-9 ablate-B winner
# (generic-only sp_* families: jump,har,trend,hurst,signature), but replaces the
# single RankICLGBModel with a 5-seed RankICEnsembleLGBModel (42,7,2026,99,123)
# that averages per-day predictions.
#
# Differs from exp-15 (tac-rd-rank-ensemble, mlflow exp 15) ONLY by dropping the
# TA subset (rsi_14,roc_10,macd_hist,willr_14,atr_14) and the inter-asset xr_*
# features, so any change vs exp-15 is attributable to the feature set alone,
# and any change vs exp-9 is attributable to the ensemble + full history alone.
#
# Run:
# rd_run_workflow config_path=experiments/workflows/exp12_isolation_ensemble.yaml \
# experiment_name=tac-rd-rank-ensemble-isolated
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
{# Gate computed on-the-fly from signal + lake close prices #}
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-sq-gate-2025"
uri: "sqlite:///mlruns.db"
default_exp_name: "tac-rd-rank-ensemble-isolated"
task:
model:
class: LGBModel
module_path: qlib.contrib.model.gbdt
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.05
num_leaves: 15
n_estimators: 200
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.01
reg_lambda: 0.01
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
dataset:
class: DatasetH
@@ -50,9 +80,9 @@ task:
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2025-12-31
fit_start_time: 2015-01-03
fit_end_time: 2026-01-03
end_time: 2026-08-14
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
@@ -70,9 +100,9 @@ task:
- class: Fillna
kwargs: {}
segments:
train: [2015-01-03, 2025-09-01]
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2025-01-02, 2025-12-31]
test: [2026-01-04, 2026-08-10]
record:
- class: SignalRecord
@@ -80,29 +110,25 @@ task:
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs: { ana_long_short: true, ann_scaler: 252 }
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: SignalQualityGateStrategy
module_path: tac_qlib.contrib.strategy.signal_quality_gate
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
lake_root: "{{ LAKE }}"
gate_topk: 10
gate_lookback: 5
gate_threshold: 0.5
gate_start: "2015-01-03"
gate_end: "2025-12-31"
topk: 10
n_drop: 1
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2025-01-02
end_time: 2025-12-31
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
+97
View File
@@ -0,0 +1,97 @@
# Re-run of experiment 16 with validated family=ta and family=sp lake features.
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sma_5,sma_20,ema_12,ema_26,rsi_14,macd,macd_signal,macd_hist,bb_upper,bb_middle,bb_lower,atr_14,adx_14,sp_ret,sp_ou_half_life,sp_ou_revert,sp_ou_zscore,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_down,sp_max_move,sp_max_up,sp_rv1,sp_rv5,sp_rv22,sp_rv_ac1,sp_rv_cv_22,sp_vol_ratio_1_22,sp_vol_ratio_5_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_rskew_5,sp_rskew_22,sp_rkurt_5,sp_rkurt_22,sp_dsv_1,sp_dsv_5,sp_dsv_22,sp_dsv_ratio_1,sp_dsv_ratio_5,sp_dsv_ratio_22,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead,sp_sig_level2_lead_lag_5,sp_sig_level2_lag_lead_5" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-exp16-db-ta-sp" }
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-10
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "{{ FEATURES }}"
infer_processors:
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: ProcessInf, kwargs: {} }
- { class: CSRankNorm, kwargs: {} }
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: Fillna, kwargs: {} }
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs: { signal: "<PRED>", topk: 10, n_drop: 2, only_tradable: true, risk_degree: 0.95 }
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
+97
View File
@@ -0,0 +1,97 @@
# General stochastic-process feature ablation: no TA, HMM, or OU fields.
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_down,sp_max_move,sp_max_up,sp_rv1,sp_rv5,sp_rv22,sp_rv_ac1,sp_rv_cv_22,sp_vol_ratio_1_22,sp_vol_ratio_5_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_rskew_5,sp_rskew_22,sp_rkurt_5,sp_rkurt_22,sp_dsv_1,sp_dsv_5,sp_dsv_22,sp_dsv_ratio_1,sp_dsv_ratio_5,sp_dsv_ratio_22,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead,sp_sig_level2_lead_lag_5,sp_sig_level2_lag_lead_5" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-exp22-stochastic-general" }
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-10
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "{{ FEATURES }}"
infer_processors:
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: ProcessInf, kwargs: {} }
- { class: CSRankNorm, kwargs: {} }
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: Fillna, kwargs: {} }
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs: { signal: "<PRED>", topk: 10, n_drop: 2, only_tradable: true, risk_degree: 0.95 }
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
+97
View File
@@ -0,0 +1,97 @@
# Exact compact stochastic feature set requested for a new run in MLflow exp 25.
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-exp22-stochastic-general" }
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-10
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "{{ FEATURES }}"
infer_processors:
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: ProcessInf, kwargs: {} }
- { class: CSRankNorm, kwargs: {} }
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: Fillna, kwargs: {} }
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs: { signal: "<PRED>", topk: 10, n_drop: 2, only_tradable: true, risk_degree: 0.95 }
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
+98
View File
@@ -0,0 +1,98 @@
# Compact stochastic feature set with reduced turnover: n_drop=1 instead of 2.
# Same setup as exp24 (compact baseline) but replacing the TopkDropout n_drop 2 with 1.
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-exp22-stochastic-general" }
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-10
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "{{ FEATURES }}"
infer_processors:
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: ProcessInf, kwargs: {} }
- { class: CSRankNorm, kwargs: {} }
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: Fillna, kwargs: {} }
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs: { signal: "<PRED>", topk: 10, n_drop: 1, only_tradable: true, risk_degree: 0.95 }
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
+99
View File
@@ -0,0 +1,99 @@
# M2 isolation run: base compact set + risk-adjusted drift sp_sharpe_22.
# Exact copy of exp26 (reference: expId=25 run=21afc6afdb674a399b59dd76c97628ce)
# except feature_fields. 5-seed ensemble.
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead,sp_sharpe_22" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-exp30-m2-sharpe" }
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-10
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "{{ FEATURES }}"
infer_processors:
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: ProcessInf, kwargs: {} }
- { class: CSRankNorm, kwargs: {} }
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: Fillna, kwargs: {} }
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs: { signal: "<PRED>", topk: 10, n_drop: 1, only_tradable: true, risk_degree: 0.95 }
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
+99
View File
@@ -0,0 +1,99 @@
# M3 isolation run: base compact set + GARCH(1,1) vol-regime trio.
# Exact copy of exp26 (reference: expId=25 run=21afc6afdb674a399b59dd76c97628ce)
# except feature_fields. 5-seed ensemble.
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead,sp_garch_cond_var,sp_garch_persistence,sp_garch_std_resid" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-exp31-m3-garch" }
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-10
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "{{ FEATURES }}"
infer_processors:
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: ProcessInf, kwargs: {} }
- { class: CSRankNorm, kwargs: {} }
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: Fillna, kwargs: {} }
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs: { signal: "<PRED>", topk: 10, n_drop: 1, only_tradable: true, risk_degree: 0.95 }
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d