Compare commits
6
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3d845306fe | ||
|
|
1075525d6e | ||
|
|
63c1ea763e | ||
|
|
2c2684b103 | ||
|
|
32477c7bb8 | ||
|
|
e657c58758 |
+5
-13
@@ -1,27 +1,19 @@
|
|||||||
# TradeAC custom-qlib-code snapshot (auto-generated)
|
# TradeAC custom-qlib-code snapshot (auto-generated)
|
||||||
# parent repo HEAD : 49f8ba2b5447f82f1c70451f0c5f290d3dac321d
|
# parent repo HEAD : 1075525d6e954dca0bb31daf6675904f8b569f1a
|
||||||
# tac-qlib/tac_qlib/contrib
|
# tac-qlib/tac_qlib/contrib
|
||||||
# tac-qlib/tac_qlib/data
|
# tac-qlib/tac_qlib/data
|
||||||
# per-file hashes (git hash-object):
|
# per-file hashes (git hash-object):
|
||||||
1b6298c4a5652f2e863cbdc385a1014a570fcd59 tac-qlib/tac_qlib/contrib/__init__.py
|
1b6298c4a5652f2e863cbdc385a1014a570fcd59 tac-qlib/tac_qlib/contrib/__init__.py
|
||||||
b419ee55ed455a1c45423d1c9025ca5cc0a98576 tac-qlib/tac_qlib/contrib/__pycache__/__init__.cpython-312.pyc
|
|
||||||
c76a9f17f680e74eea766eff27f7624359749ed6 tac-qlib/tac_qlib/contrib/data/__init__.py
|
c76a9f17f680e74eea766eff27f7624359749ed6 tac-qlib/tac_qlib/contrib/data/__init__.py
|
||||||
2f6c67620aa2f9e6aaaef3369361d9b3eac3d6ca tac-qlib/tac_qlib/contrib/data/__pycache__/__init__.cpython-312.pyc
|
|
||||||
fdd5923a70a399e8680913593ff111641947898e tac-qlib/tac_qlib/contrib/data/__pycache__/handler.cpython-312.pyc
|
|
||||||
0dd25ef161c6e0f15eafc84886e7e1381deb38c3 tac-qlib/tac_qlib/contrib/data/handler.py
|
0dd25ef161c6e0f15eafc84886e7e1381deb38c3 tac-qlib/tac_qlib/contrib/data/handler.py
|
||||||
b151d139a0dcde87d74b21e7c4b729176ba5c39b tac-qlib/tac_qlib/contrib/model/__init__.py
|
b151d139a0dcde87d74b21e7c4b729176ba5c39b tac-qlib/tac_qlib/contrib/model/__init__.py
|
||||||
08dec87ccdf6bb5d2cf611ca3032a4280aaab8cf tac-qlib/tac_qlib/contrib/model/__pycache__/__init__.cpython-312.pyc
|
|
||||||
6fb61946ea9a83dfb560de3717f5fbf482c4c00e tac-qlib/tac_qlib/contrib/model/__pycache__/rank_ensemble.cpython-312.pyc
|
|
||||||
3e80f2e08b661ddd2f58ffe5a6196063fa41ae51 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_gbdt.cpython-312.pyc
|
|
||||||
d3f051f3a8650c42fedc7b367b966f7c74fb5789 tac-qlib/tac_qlib/contrib/model/rank_ensemble.py
|
d3f051f3a8650c42fedc7b367b966f7c74fb5789 tac-qlib/tac_qlib/contrib/model/rank_ensemble.py
|
||||||
d03e6611338918d4aac5eea4adf26f85a3763652 tac-qlib/tac_qlib/contrib/model/rank_gbdt.py
|
ccfe7d554989aa7f3e5a2128ae663e51b2207149 tac-qlib/tac_qlib/contrib/model/rank_gbdt.py
|
||||||
4afcf9058231111c412925f4c4b84e81d656db87 tac-qlib/tac_qlib/contrib/strategy/__init__.py
|
4afcf9058231111c412925f4c4b84e81d656db87 tac-qlib/tac_qlib/contrib/strategy/__init__.py
|
||||||
6ad10c2ebe37c16417e67c7aeb731ad1fcb6da2f tac-qlib/tac_qlib/contrib/strategy/__pycache__/__init__.cpython-312.pyc
|
|
||||||
8d684b3216b040071d9ee4fa920a0e0c7486d278 tac-qlib/tac_qlib/contrib/strategy/__pycache__/optimal_stop.cpython-312.pyc
|
|
||||||
79aaad9e39fcc740a773f4f63c512ce1086cfde0 tac-qlib/tac_qlib/contrib/strategy/optimal_stop.py
|
79aaad9e39fcc740a773f4f63c512ce1086cfde0 tac-qlib/tac_qlib/contrib/strategy/optimal_stop.py
|
||||||
92e6e90eb0cd0a25142034560f27adb6b705b1a8 tac-qlib/tac_qlib/data/__init__.py
|
92e6e90eb0cd0a25142034560f27adb6b705b1a8 tac-qlib/tac_qlib/data/__init__.py
|
||||||
7c4e6c345fad1978efe8860c0d977d0c02d6f8d9 tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc
|
b6dc9ced54f4044f5954b60ddd199acae9eef456 tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc
|
||||||
99e602392d51663cb06d5c425000b1ed1e5a916b tac-qlib/tac_qlib/data/__pycache__/config.cpython-312.pyc
|
5cede5184d17910b0232d343f8ecca460098ea11 tac-qlib/tac_qlib/data/__pycache__/config.cpython-312.pyc
|
||||||
020dcdcf288e4832c8cf2386351f78d5ceb4fe13 tac-qlib/tac_qlib/data/__pycache__/providers.cpython-312.pyc
|
3a5accd6d239f342354cf1fa9410a6d2fb921de0 tac-qlib/tac_qlib/data/__pycache__/providers.cpython-312.pyc
|
||||||
53c9007a928841fd3c3b08450f9a6520ce1ac091 tac-qlib/tac_qlib/data/config.py
|
53c9007a928841fd3c3b08450f9a6520ce1ac091 tac-qlib/tac_qlib/data/config.py
|
||||||
8d0644f6f0d1efb94798ed444cc73e63b643459b tac-qlib/tac_qlib/data/providers.py
|
8d0644f6f0d1efb94798ed444cc73e63b643459b tac-qlib/tac_qlib/data/providers.py
|
||||||
|
|||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -53,61 +53,23 @@ from qlib.workflow import R
|
|||||||
__all__ = ["RankICLGBModel", "rankic_feval"]
|
__all__ = ["RankICLGBModel", "rankic_feval"]
|
||||||
|
|
||||||
|
|
||||||
def _group_averaged_rank(values: np.ndarray, gid: np.ndarray, offs: np.ndarray) -> np.ndarray:
|
|
||||||
"""Averaged (tie-corrected) rank of ``values`` within each group, vectorized.
|
|
||||||
|
|
||||||
``gid`` maps each row to its group id; ``offs`` holds the cumulative row
|
|
||||||
offsets so that group ``i`` occupies rows ``[offs[i], offs[i+1])``. Returns
|
|
||||||
the same result as ``pandas.Series.rank(method='average')`` applied per
|
|
||||||
group, but in one pass (``np.lexsort`` is the only non-linear step).
|
|
||||||
"""
|
|
||||||
n = len(values)
|
|
||||||
order = np.lexsort((values, gid))
|
|
||||||
ord_rank = np.empty(n, dtype=np.float64)
|
|
||||||
ord_rank[order] = np.arange(n, dtype=np.float64) - offs[gid[order]] + 1.0
|
|
||||||
sg = gid[order]
|
|
||||||
sv = values[order]
|
|
||||||
newblock = np.empty(n, dtype=bool)
|
|
||||||
newblock[0] = True
|
|
||||||
newblock[1:] = (sg[1:] != sg[:-1]) | (sv[1:] != sv[:-1])
|
|
||||||
blockid = np.cumsum(newblock) - 1
|
|
||||||
block_mean = np.bincount(blockid, weights=ord_rank[order]) / np.bincount(blockid)
|
|
||||||
out = np.empty(n)
|
|
||||||
out[order] = block_mean[blockid]
|
|
||||||
return out
|
|
||||||
|
|
||||||
|
|
||||||
def _per_day_spearman(preds: np.ndarray, labels: np.ndarray, group: np.ndarray) -> float:
|
def _per_day_spearman(preds: np.ndarray, labels: np.ndarray, group: np.ndarray) -> float:
|
||||||
"""Mean per-day Spearman rank correlation of preds vs labels.
|
"""Mean per-day Spearman rank correlation of preds vs labels.
|
||||||
|
|
||||||
``group`` holds the number of rows of each trading day (query group), in
|
``group`` holds the number of rows of each trading day (query group), in
|
||||||
order. Days with <3 valid rows or a constant pred/label are skipped.
|
order. Days with <3 valid rows or a constant pred/label are skipped.
|
||||||
|
|
||||||
Vectorized: per-day Spearman == Pearson of the per-day rank transforms,
|
|
||||||
and the Pearson moments (``sum``, ``sum`` of products/squares) aggregate
|
|
||||||
over each day with ``np.bincount``. Runs ~10x faster than the per-day
|
|
||||||
``pd.Series.rank()`` loop that preceded it — this feval is invoked on the
|
|
||||||
train and valid panels every boosting round, per seed.
|
|
||||||
"""
|
"""
|
||||||
if group is None or len(group) == 0:
|
if group is None or len(group) == 0:
|
||||||
return 0.0
|
return 0.0
|
||||||
offs = np.concatenate([[0], np.cumsum(group.astype(int))])
|
offs = np.concatenate([[0], np.cumsum(group.astype(int))])
|
||||||
gid = np.repeat(np.arange(len(group)), group.astype(int))
|
vals = []
|
||||||
rp = _group_averaged_rank(preds, gid, offs)
|
for i in range(len(group)):
|
||||||
rl = _group_averaged_rank(labels, gid, offs)
|
s = slice(offs[i], offs[i + 1])
|
||||||
n_g = group.astype(float)
|
p, l = preds[s], labels[s]
|
||||||
s_p = np.bincount(gid, weights=rp)
|
if len(p) < 3 or np.std(p) == 0 or np.std(l) == 0:
|
||||||
s_l = np.bincount(gid, weights=rl)
|
continue
|
||||||
s_pl = np.bincount(gid, weights=rp * rl)
|
vals.append(np.corrcoef(pd.Series(p).rank(), pd.Series(l).rank())[0, 1])
|
||||||
s_pp = np.bincount(gid, weights=rp * rp)
|
return float(np.mean(vals)) if vals else 0.0
|
||||||
s_ll = np.bincount(gid, weights=rl * rl)
|
|
||||||
cov = n_g * s_pl - s_p * s_l
|
|
||||||
var_p = n_g * s_pp - s_p ** 2
|
|
||||||
var_l = n_g * s_ll - s_l ** 2
|
|
||||||
denom = np.sqrt(var_p * var_l)
|
|
||||||
valid = (n_g >= 3) & (denom > 0)
|
|
||||||
corr = np.where(valid, cov / np.where(denom == 0, 1, denom), 0.0)
|
|
||||||
return float(corr[valid].mean()) if valid.any() else 0.0
|
|
||||||
|
|
||||||
|
|
||||||
def rankic_feval(preds, dataset):
|
def rankic_feval(preds, dataset):
|
||||||
|
|||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,133 @@
|
|||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
# ABLATION A (baseline): LightGBM with RankIC early-stopping on the 50-ETF SP-5d
|
||||||
|
# panel, using ALL 24 sp_* feature columns (ou,hmm,jump,har,trend,hurst,
|
||||||
|
# signature). Copy of the canonical workflow_lgb_sp5d_rankic.yaml with a
|
||||||
|
# distinct experiment name so the ablation runs are isolated.
|
||||||
|
#
|
||||||
|
# Run:
|
||||||
|
# rd_run_workflow config_path=tac-qlib/workflows/ablate_baseline_all_sp_fields.yaml \
|
||||||
|
# experiment_name=tac-rd-rank-ablate
|
||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
{%- set LAKE = TAC_LAKE_DIR %}
|
||||||
|
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||||
|
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||||
|
|
||||||
|
qlib_init:
|
||||||
|
provider_uri: "{{ LAKE }}"
|
||||||
|
region: us
|
||||||
|
expression_cache: null
|
||||||
|
dataset_cache: null
|
||||||
|
|
||||||
|
calendar_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
instrument_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
markets: {}
|
||||||
|
feature_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
|
||||||
|
exp_manager:
|
||||||
|
class: MLflowExpManager
|
||||||
|
module_path: qlib.workflow.expm
|
||||||
|
kwargs:
|
||||||
|
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||||
|
default_exp_name: "tac-rd-rank-ablate"
|
||||||
|
|
||||||
|
task:
|
||||||
|
model:
|
||||||
|
class: RankICLGBModel
|
||||||
|
module_path: tac_qlib.contrib.model.rank_gbdt
|
||||||
|
kwargs:
|
||||||
|
loss: mse
|
||||||
|
learning_rate: 0.02
|
||||||
|
num_leaves: 31
|
||||||
|
n_estimators: 3000
|
||||||
|
num_boost_round: 3000
|
||||||
|
early_stopping_rounds: 200
|
||||||
|
min_data_in_leaf: 20
|
||||||
|
lambda_l2: 0.5
|
||||||
|
colsample_bytree: 0.8
|
||||||
|
subsample: 0.8
|
||||||
|
subsample_freq: 1
|
||||||
|
reg_alpha: 0.1
|
||||||
|
reg_lambda: 1.0
|
||||||
|
seed: 42
|
||||||
|
|
||||||
|
dataset:
|
||||||
|
class: DatasetH
|
||||||
|
module_path: qlib.data.dataset
|
||||||
|
kwargs:
|
||||||
|
handler:
|
||||||
|
class: TACHandler
|
||||||
|
module_path: tac_qlib.contrib.data.handler
|
||||||
|
kwargs:
|
||||||
|
instruments: "{{ UNIVERSE }}"
|
||||||
|
start_time: 2015-01-03
|
||||||
|
end_time: 2026-08-10
|
||||||
|
fit_start_time: 2015-01-03
|
||||||
|
fit_end_time: 2025-09-01
|
||||||
|
freq: day
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||||
|
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||||
|
infer_processors:
|
||||||
|
- class: DropAllNaN
|
||||||
|
kwargs: {}
|
||||||
|
- class: ProcessInf
|
||||||
|
kwargs: {}
|
||||||
|
- class: CSRankNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: ZScoreNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: Fillna
|
||||||
|
kwargs: {}
|
||||||
|
segments:
|
||||||
|
train: [2015-01-03, 2025-09-01]
|
||||||
|
valid: [2025-09-03, 2026-01-03]
|
||||||
|
test: [2026-01-04, 2026-08-10]
|
||||||
|
|
||||||
|
record:
|
||||||
|
- class: SignalRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs: {}
|
||||||
|
- class: SigAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
ana_long_short: true
|
||||||
|
ann_scaler: 252
|
||||||
|
- class: PortAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
config:
|
||||||
|
strategy:
|
||||||
|
class: TopkDropoutStrategy
|
||||||
|
module_path: qlib.contrib.strategy
|
||||||
|
kwargs:
|
||||||
|
signal: "<PRED>"
|
||||||
|
topk: 10
|
||||||
|
n_drop: 2
|
||||||
|
only_tradable: true
|
||||||
|
risk_degree: 0.95
|
||||||
|
backtest:
|
||||||
|
start_time: 2026-01-04
|
||||||
|
end_time: 2026-08-10
|
||||||
|
account: 1000000
|
||||||
|
benchmark: SPY
|
||||||
|
exchange_kwargs:
|
||||||
|
codes: "{{ UNIVERSE }}"
|
||||||
|
deal_price: $close
|
||||||
|
freq: day
|
||||||
|
open_cost: 0.0005
|
||||||
|
close_cost: 0.0015
|
||||||
|
min_cost: 5.0
|
||||||
|
risk_analysis_freq: 1d
|
||||||
@@ -0,0 +1,134 @@
|
|||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
# ABLATION B (generic-only): same panel/model as the baseline, but feature
|
||||||
|
# fields restricted to the model-free / generic stochastic-process families
|
||||||
|
# (jump,har,trend,hurst,signature). Drops the model-specific ou (OU/AR-1
|
||||||
|
# half-life) and hmm (2-state regime) families to test whether the generic
|
||||||
|
# families alone dominate the rank dimension.
|
||||||
|
#
|
||||||
|
# Run:
|
||||||
|
# rd_run_workflow config_path=tac-qlib/workflows/ablate_generic_only_sp_fields.yaml \
|
||||||
|
# experiment_name=tac-rd-rank-ablate
|
||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
{%- set LAKE = TAC_LAKE_DIR %}
|
||||||
|
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||||
|
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||||
|
|
||||||
|
qlib_init:
|
||||||
|
provider_uri: "{{ LAKE }}"
|
||||||
|
region: us
|
||||||
|
expression_cache: null
|
||||||
|
dataset_cache: null
|
||||||
|
|
||||||
|
calendar_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
instrument_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
markets: {}
|
||||||
|
feature_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
|
||||||
|
exp_manager:
|
||||||
|
class: MLflowExpManager
|
||||||
|
module_path: qlib.workflow.expm
|
||||||
|
kwargs:
|
||||||
|
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||||
|
default_exp_name: "tac-rd-rank-ablate"
|
||||||
|
|
||||||
|
task:
|
||||||
|
model:
|
||||||
|
class: RankICLGBModel
|
||||||
|
module_path: tac_qlib.contrib.model.rank_gbdt
|
||||||
|
kwargs:
|
||||||
|
loss: mse
|
||||||
|
learning_rate: 0.02
|
||||||
|
num_leaves: 31
|
||||||
|
n_estimators: 3000
|
||||||
|
num_boost_round: 3000
|
||||||
|
early_stopping_rounds: 200
|
||||||
|
min_data_in_leaf: 20
|
||||||
|
lambda_l2: 0.5
|
||||||
|
colsample_bytree: 0.8
|
||||||
|
subsample: 0.8
|
||||||
|
subsample_freq: 1
|
||||||
|
reg_alpha: 0.1
|
||||||
|
reg_lambda: 1.0
|
||||||
|
seed: 42
|
||||||
|
|
||||||
|
dataset:
|
||||||
|
class: DatasetH
|
||||||
|
module_path: qlib.data.dataset
|
||||||
|
kwargs:
|
||||||
|
handler:
|
||||||
|
class: TACHandler
|
||||||
|
module_path: tac_qlib.contrib.data.handler
|
||||||
|
kwargs:
|
||||||
|
instruments: "{{ UNIVERSE }}"
|
||||||
|
start_time: 2015-01-03
|
||||||
|
end_time: 2026-08-10
|
||||||
|
fit_start_time: 2015-01-03
|
||||||
|
fit_end_time: 2025-09-01
|
||||||
|
freq: day
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||||
|
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||||
|
infer_processors:
|
||||||
|
- class: DropAllNaN
|
||||||
|
kwargs: {}
|
||||||
|
- class: ProcessInf
|
||||||
|
kwargs: {}
|
||||||
|
- class: CSRankNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: ZScoreNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: Fillna
|
||||||
|
kwargs: {}
|
||||||
|
segments:
|
||||||
|
train: [2015-01-03, 2025-09-01]
|
||||||
|
valid: [2025-09-03, 2026-01-03]
|
||||||
|
test: [2026-01-04, 2026-08-10]
|
||||||
|
|
||||||
|
record:
|
||||||
|
- class: SignalRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs: {}
|
||||||
|
- class: SigAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
ana_long_short: true
|
||||||
|
ann_scaler: 252
|
||||||
|
- class: PortAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
config:
|
||||||
|
strategy:
|
||||||
|
class: TopkDropoutStrategy
|
||||||
|
module_path: qlib.contrib.strategy
|
||||||
|
kwargs:
|
||||||
|
signal: "<PRED>"
|
||||||
|
topk: 10
|
||||||
|
n_drop: 2
|
||||||
|
only_tradable: true
|
||||||
|
risk_degree: 0.95
|
||||||
|
backtest:
|
||||||
|
start_time: 2026-01-04
|
||||||
|
end_time: 2026-08-10
|
||||||
|
account: 1000000
|
||||||
|
benchmark: SPY
|
||||||
|
exchange_kwargs:
|
||||||
|
codes: "{{ UNIVERSE }}"
|
||||||
|
deal_price: $close
|
||||||
|
freq: day
|
||||||
|
open_cost: 0.0005
|
||||||
|
close_cost: 0.0015
|
||||||
|
min_cost: 5.0
|
||||||
|
risk_analysis_freq: 1d
|
||||||
@@ -1,22 +1,24 @@
|
|||||||
# -----------------------------------------------------------------------------
|
# -----------------------------------------------------------------------------
|
||||||
# QUEUE-03 — topk 20 vs 10 diversification on the compact reference.
|
# ISOLATION: multi-seed RankIC ensemble, ablate-B generic-only feature set.
|
||||||
#
|
#
|
||||||
# Hypothesis (book ch.05/chat-ideas): the effective independent names in the
|
# Isolates the ensemble effect on the SP-5d rank signal. Same panel, segments,
|
||||||
# 50-ETF book is small (~4, chat-derived eigenvalue analysis); raising topk
|
# history (full backfilled 2016+) and feature set as the exp-9 ablate-B winner
|
||||||
# diversifies the book and should cut drawdown / raise net IR without hurting
|
# (generic-only sp_* families: jump,har,trend,hurst,signature), but replaces the
|
||||||
# the (weak) rank signal — cost relief by spreading the book wider.
|
# single RankICLGBModel with a 5-seed RankICEnsembleLGBModel (42,7,2026,99,123)
|
||||||
|
# that averages per-day predictions.
|
||||||
#
|
#
|
||||||
# Change vs exp-26 reference: ONE variable — strategy topk 10 -> 20 (n_drop 1).
|
# Differs from exp-15 (tac-rd-rank-ensemble, mlflow exp 15) ONLY by dropping the
|
||||||
# Everything else identical.
|
# TA subset (rsi_14,roc_10,macd_hist,willr_14,atr_14) and the inter-asset xr_*
|
||||||
|
# features, so any change vs exp-15 is attributable to the feature set alone,
|
||||||
|
# and any change vs exp-9 is attributable to the ensemble + full history alone.
|
||||||
#
|
#
|
||||||
# Acceptance: net_IR > 0.21 AND net_max_drawdown < 7.69% AND net_ann_return >=
|
# Run:
|
||||||
# +2.13%; watch total_cost — more names held must not raise turnover/cost.
|
# rd_run_workflow config_path=experiments/workflows/exp12_isolation_ensemble.yaml \
|
||||||
# Run: rd_run_workflow config_path=<repo>/experiments/queue/workflows/q03_topk20.yaml \
|
# experiment_name=tac-rd-rank-ensemble-isolated
|
||||||
# experiment_name=tac-rd-q03-topk20
|
|
||||||
# -----------------------------------------------------------------------------
|
# -----------------------------------------------------------------------------
|
||||||
{%- set LAKE = TAC_LAKE_DIR %}
|
{%- set LAKE = TAC_LAKE_DIR %}
|
||||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||||
{%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||||
|
|
||||||
qlib_init:
|
qlib_init:
|
||||||
provider_uri: "{{ LAKE }}"
|
provider_uri: "{{ LAKE }}"
|
||||||
@@ -45,8 +47,8 @@ qlib_init:
|
|||||||
class: MLflowExpManager
|
class: MLflowExpManager
|
||||||
module_path: qlib.workflow.expm
|
module_path: qlib.workflow.expm
|
||||||
kwargs:
|
kwargs:
|
||||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
uri: "sqlite:///mlruns.db"
|
||||||
default_exp_name: "tac-rd-q03-topk20"
|
default_exp_name: "tac-rd-rank-ensemble-isolated"
|
||||||
|
|
||||||
task:
|
task:
|
||||||
model:
|
model:
|
||||||
@@ -67,7 +69,6 @@ task:
|
|||||||
reg_alpha: 0.1
|
reg_alpha: 0.1
|
||||||
reg_lambda: 1.0
|
reg_lambda: 1.0
|
||||||
seeds: "42,7,2026,99,123"
|
seeds: "42,7,2026,99,123"
|
||||||
parallel: 5
|
|
||||||
|
|
||||||
dataset:
|
dataset:
|
||||||
class: DatasetH
|
class: DatasetH
|
||||||
@@ -79,14 +80,14 @@ task:
|
|||||||
kwargs:
|
kwargs:
|
||||||
instruments: "{{ UNIVERSE }}"
|
instruments: "{{ UNIVERSE }}"
|
||||||
start_time: 2015-01-03
|
start_time: 2015-01-03
|
||||||
end_time: 2026-08-10
|
end_time: 2026-08-14
|
||||||
fit_start_time: 2016-01-04
|
fit_start_time: 2016-01-04
|
||||||
fit_end_time: 2025-09-01
|
fit_end_time: 2025-09-01
|
||||||
freq: day
|
freq: day
|
||||||
lake_root: "{{ LAKE }}"
|
lake_root: "{{ LAKE }}"
|
||||||
market: US
|
market: US
|
||||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||||
feature_fields: "{{ FEATURES }}"
|
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||||
infer_processors:
|
infer_processors:
|
||||||
- class: DropAllNaN
|
- class: DropAllNaN
|
||||||
kwargs: {}
|
kwargs: {}
|
||||||
@@ -121,8 +122,8 @@ task:
|
|||||||
module_path: qlib.contrib.strategy
|
module_path: qlib.contrib.strategy
|
||||||
kwargs:
|
kwargs:
|
||||||
signal: "<PRED>"
|
signal: "<PRED>"
|
||||||
topk: 20
|
topk: 10
|
||||||
n_drop: 1
|
n_drop: 2
|
||||||
only_tradable: true
|
only_tradable: true
|
||||||
risk_degree: 0.95
|
risk_degree: 0.95
|
||||||
backtest:
|
backtest:
|
||||||
@@ -0,0 +1,97 @@
|
|||||||
|
# Re-run of experiment 16 with validated family=ta and family=sp lake features.
|
||||||
|
{%- set LAKE = TAC_LAKE_DIR %}
|
||||||
|
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||||
|
{%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sma_5,sma_20,ema_12,ema_26,rsi_14,macd,macd_signal,macd_hist,bb_upper,bb_middle,bb_lower,atr_14,adx_14,sp_ret,sp_ou_half_life,sp_ou_revert,sp_ou_zscore,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_down,sp_max_move,sp_max_up,sp_rv1,sp_rv5,sp_rv22,sp_rv_ac1,sp_rv_cv_22,sp_vol_ratio_1_22,sp_vol_ratio_5_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_rskew_5,sp_rskew_22,sp_rkurt_5,sp_rkurt_22,sp_dsv_1,sp_dsv_5,sp_dsv_22,sp_dsv_ratio_1,sp_dsv_ratio_5,sp_dsv_ratio_22,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead,sp_sig_level2_lead_lag_5,sp_sig_level2_lag_lead_5" %}
|
||||||
|
|
||||||
|
qlib_init:
|
||||||
|
provider_uri: "{{ LAKE }}"
|
||||||
|
region: us
|
||||||
|
expression_cache: null
|
||||||
|
dataset_cache: null
|
||||||
|
calendar_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||||
|
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||||
|
instrument_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||||
|
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
|
||||||
|
feature_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||||
|
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||||
|
exp_manager:
|
||||||
|
class: MLflowExpManager
|
||||||
|
module_path: qlib.workflow.expm
|
||||||
|
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-exp16-db-ta-sp" }
|
||||||
|
|
||||||
|
task:
|
||||||
|
model:
|
||||||
|
class: RankICEnsembleLGBModel
|
||||||
|
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||||
|
kwargs:
|
||||||
|
loss: mse
|
||||||
|
learning_rate: 0.02
|
||||||
|
num_leaves: 31
|
||||||
|
n_estimators: 3000
|
||||||
|
num_boost_round: 3000
|
||||||
|
early_stopping_rounds: 200
|
||||||
|
min_data_in_leaf: 20
|
||||||
|
lambda_l2: 0.5
|
||||||
|
colsample_bytree: 0.8
|
||||||
|
subsample: 0.8
|
||||||
|
subsample_freq: 1
|
||||||
|
reg_alpha: 0.1
|
||||||
|
reg_lambda: 1.0
|
||||||
|
seeds: "42,7,2026,99,123"
|
||||||
|
|
||||||
|
dataset:
|
||||||
|
class: DatasetH
|
||||||
|
module_path: qlib.data.dataset
|
||||||
|
kwargs:
|
||||||
|
handler:
|
||||||
|
class: TACHandler
|
||||||
|
module_path: tac_qlib.contrib.data.handler
|
||||||
|
kwargs:
|
||||||
|
instruments: "{{ UNIVERSE }}"
|
||||||
|
start_time: 2015-01-03
|
||||||
|
end_time: 2026-08-10
|
||||||
|
fit_start_time: 2016-01-04
|
||||||
|
fit_end_time: 2025-09-01
|
||||||
|
freq: day
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||||
|
feature_fields: "{{ FEATURES }}"
|
||||||
|
infer_processors:
|
||||||
|
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
|
||||||
|
- { class: ProcessInf, kwargs: {} }
|
||||||
|
- { class: CSRankNorm, kwargs: {} }
|
||||||
|
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
|
||||||
|
- { class: Fillna, kwargs: {} }
|
||||||
|
segments:
|
||||||
|
train: [2016-01-04, 2025-09-01]
|
||||||
|
valid: [2025-09-03, 2026-01-03]
|
||||||
|
test: [2026-01-04, 2026-08-10]
|
||||||
|
|
||||||
|
record:
|
||||||
|
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
|
||||||
|
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
|
||||||
|
- class: PortAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
config:
|
||||||
|
strategy:
|
||||||
|
class: TopkDropoutStrategy
|
||||||
|
module_path: qlib.contrib.strategy
|
||||||
|
kwargs: { signal: "<PRED>", topk: 10, n_drop: 2, only_tradable: true, risk_degree: 0.95 }
|
||||||
|
backtest:
|
||||||
|
start_time: 2026-01-04
|
||||||
|
end_time: 2026-08-10
|
||||||
|
account: 1000000
|
||||||
|
benchmark: SPY
|
||||||
|
exchange_kwargs:
|
||||||
|
codes: "{{ UNIVERSE }}"
|
||||||
|
deal_price: $close
|
||||||
|
freq: day
|
||||||
|
open_cost: 0.0005
|
||||||
|
close_cost: 0.0015
|
||||||
|
min_cost: 5.0
|
||||||
|
risk_analysis_freq: 1d
|
||||||
Reference in New Issue
Block a user