Compare commits
13
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
202589fc3e | ||
|
|
3ded27e5f9 | ||
|
|
125be7b96f | ||
|
|
20857f76be | ||
|
|
9c0cf096f8 | ||
|
|
80c7230e17 | ||
|
|
07e78c2bd9 | ||
|
|
e09ab7f05a | ||
|
|
337f6e17d8 | ||
|
|
7fad62a4ef | ||
|
|
f9ef005e9a | ||
|
|
c09997c7e2 | ||
|
|
32477c7bb8 |
+5
-3
@@ -1,5 +1,5 @@
|
|||||||
# TradeAC custom-qlib-code snapshot (auto-generated)
|
# TradeAC custom-qlib-code snapshot (auto-generated)
|
||||||
# parent repo HEAD : f9d1fe66f6e4f8ace0d3d774e23f1c79def9bae0
|
# parent repo HEAD : 125be7b96fb5975e798a0b4301eeb5809a8a181c
|
||||||
# tac-qlib/tac_qlib/contrib
|
# tac-qlib/tac_qlib/contrib
|
||||||
# tac-qlib/tac_qlib/data
|
# tac-qlib/tac_qlib/data
|
||||||
# per-file hashes (git hash-object):
|
# per-file hashes (git hash-object):
|
||||||
@@ -11,13 +11,15 @@
|
|||||||
871ff1e163c29261f140c3f53d42a41e6504c779 tac-qlib/tac_qlib/contrib/data/handler.py
|
871ff1e163c29261f140c3f53d42a41e6504c779 tac-qlib/tac_qlib/contrib/data/handler.py
|
||||||
b151d139a0dcde87d74b21e7c4b729176ba5c39b tac-qlib/tac_qlib/contrib/model/__init__.py
|
b151d139a0dcde87d74b21e7c4b729176ba5c39b tac-qlib/tac_qlib/contrib/model/__init__.py
|
||||||
ab958203f33a99d12c7d923b6efb435189231666 tac-qlib/tac_qlib/contrib/model/__pycache__/__init__.cpython-312.pyc
|
ab958203f33a99d12c7d923b6efb435189231666 tac-qlib/tac_qlib/contrib/model/__pycache__/__init__.cpython-312.pyc
|
||||||
9dc36de7e343073b7d511349ee5aede086c38f94 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_ensemble.cpython-312.pyc
|
7478f6b0f6de419615c02d4d92b54529f689ef04 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_ensemble.cpython-312.pyc
|
||||||
9f9014ddd9bce37490061312d51e8e6fe540fec4 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_gbdt.cpython-312.pyc
|
9f9014ddd9bce37490061312d51e8e6fe540fec4 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_gbdt.cpython-312.pyc
|
||||||
d3f051f3a8650c42fedc7b367b966f7c74fb5789 tac-qlib/tac_qlib/contrib/model/rank_ensemble.py
|
ce77dea53f6a87c5379782709293bf8ff55b2c75 tac-qlib/tac_qlib/contrib/model/rank_ensemble.py
|
||||||
ccfe7d554989aa7f3e5a2128ae663e51b2207149 tac-qlib/tac_qlib/contrib/model/rank_gbdt.py
|
ccfe7d554989aa7f3e5a2128ae663e51b2207149 tac-qlib/tac_qlib/contrib/model/rank_gbdt.py
|
||||||
4afcf9058231111c412925f4c4b84e81d656db87 tac-qlib/tac_qlib/contrib/strategy/__init__.py
|
4afcf9058231111c412925f4c4b84e81d656db87 tac-qlib/tac_qlib/contrib/strategy/__init__.py
|
||||||
74e5ecbbbb20bb71fd5cd083383de4ce88476712 tac-qlib/tac_qlib/contrib/strategy/__pycache__/__init__.cpython-312.pyc
|
74e5ecbbbb20bb71fd5cd083383de4ce88476712 tac-qlib/tac_qlib/contrib/strategy/__pycache__/__init__.cpython-312.pyc
|
||||||
afaf562aeaa12cebc8529cd916153252e7e3c38a tac-qlib/tac_qlib/contrib/strategy/__pycache__/optimal_stop.cpython-312.pyc
|
afaf562aeaa12cebc8529cd916153252e7e3c38a tac-qlib/tac_qlib/contrib/strategy/__pycache__/optimal_stop.cpython-312.pyc
|
||||||
|
96a0a25201f0a1bb2fc2190e26228c5c0e711a79 tac-qlib/tac_qlib/contrib/strategy/hmm_risk.py
|
||||||
|
816de5d58ae23d996635d42331cf9fc8963d5dbe tac-qlib/tac_qlib/contrib/strategy/momentum_gate.py
|
||||||
79aaad9e39fcc740a773f4f63c512ce1086cfde0 tac-qlib/tac_qlib/contrib/strategy/optimal_stop.py
|
79aaad9e39fcc740a773f4f63c512ce1086cfde0 tac-qlib/tac_qlib/contrib/strategy/optimal_stop.py
|
||||||
92e6e90eb0cd0a25142034560f27adb6b705b1a8 tac-qlib/tac_qlib/data/__init__.py
|
92e6e90eb0cd0a25142034560f27adb6b705b1a8 tac-qlib/tac_qlib/data/__init__.py
|
||||||
0ed1ead6c1314a3f25784d453e54a15a8a04baaa tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc
|
0ed1ead6c1314a3f25784d453e54a15a8a04baaa tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc
|
||||||
|
|||||||
Binary file not shown.
@@ -56,6 +56,7 @@ import os
|
|||||||
from concurrent.futures import ThreadPoolExecutor
|
from concurrent.futures import ThreadPoolExecutor
|
||||||
from typing import List, Optional
|
from typing import List, Optional
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
|
|
||||||
from qlib.data.dataset import DatasetH
|
from qlib.data.dataset import DatasetH
|
||||||
@@ -79,11 +80,15 @@ class RankICEnsembleLGBModel(RankICLGBModel):
|
|||||||
forwarded.
|
forwarded.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(self, seeds: str = "42", parallel: int = 0, **kwargs):
|
def __init__(self, seeds: str = "42", parallel: int = 0, weight_mode: str = "equal", **kwargs):
|
||||||
self.seeds = [int(s.strip()) for s in str(seeds).split(",") if s.strip()]
|
self.seeds = [int(s.strip()) for s in str(seeds).split(",") if s.strip()]
|
||||||
if not self.seeds:
|
if not self.seeds:
|
||||||
raise ValueError("seeds must contain at least one integer")
|
raise ValueError("seeds must contain at least one integer")
|
||||||
self.parallel = int(parallel)
|
self.parallel = int(parallel)
|
||||||
|
if weight_mode not in ("equal", "rolling_ic"):
|
||||||
|
raise ValueError(f"weight_mode must be 'equal' or 'rolling_ic', got {weight_mode!r}")
|
||||||
|
self.weight_mode = weight_mode
|
||||||
|
self.rolling_ic_window = int(kwargs.pop("rolling_ic_window", 21))
|
||||||
# drop seed/parallel handling from the base kwargs, keep everything else
|
# drop seed/parallel handling from the base kwargs, keep everything else
|
||||||
self._model_kwargs = dict(kwargs)
|
self._model_kwargs = dict(kwargs)
|
||||||
super().__init__(**self._model_kwargs)
|
super().__init__(**self._model_kwargs)
|
||||||
@@ -179,11 +184,44 @@ class RankICEnsembleLGBModel(RankICLGBModel):
|
|||||||
|
|
||||||
# -------------------------------------------------------------- predict
|
# -------------------------------------------------------------- predict
|
||||||
def predict(self, dataset: DatasetH, segment="test") -> pd.Series:
|
def predict(self, dataset: DatasetH, segment="test") -> pd.Series:
|
||||||
"""Average the per-seed predictions over the given segment."""
|
"""Combine per-seed predictions.
|
||||||
|
|
||||||
|
``weight_mode='equal'`` (default): simple average, as before.
|
||||||
|
``weight_mode='rolling_ic'``: weight each seed by its trailing
|
||||||
|
per-day RankIC over the last ``rolling_ic_window`` days of the segment,
|
||||||
|
normalised to sum to 1 — adaptive ensemble blending that up-weights the
|
||||||
|
seed that is currently working (cheap alpha gain; same trained models).
|
||||||
|
"""
|
||||||
if not self._models:
|
if not self._models:
|
||||||
raise ValueError("model is not fitted yet!")
|
raise ValueError("model is not fitted yet!")
|
||||||
preds = [m.predict(dataset, segment=segment) for m in self._models]
|
preds = [m.predict(dataset, segment=segment) for m in self._models]
|
||||||
if len(preds) == 1:
|
if len(preds) == 1:
|
||||||
return preds[0]
|
return preds[0]
|
||||||
frame = pd.concat(preds, axis=1)
|
frame = pd.concat(preds, axis=1)
|
||||||
return frame.mean(axis=1)
|
frame.columns = [f"seed{m.params.get('seed', i)}" for i, m in enumerate(self._models)]
|
||||||
|
if self.weight_mode == "equal":
|
||||||
|
return frame.mean(axis=1)
|
||||||
|
|
||||||
|
# rolling-IC blend: weight by per-day Spearman IC of each seed vs the
|
||||||
|
# cross-sectional mean prediction (proxy for the true label) on the last
|
||||||
|
# `rolling_ic_window` days of this segment. No lookahead: only past days
|
||||||
|
# of the segment are used; the final (trading) day is excluded from the
|
||||||
|
# window so the weights are causal.
|
||||||
|
mean_pred = frame.mean(axis=1)
|
||||||
|
dates = sorted(frame.index.get_level_values(0).unique())
|
||||||
|
win = [d for d in dates if d < dates[-1]][-self.rolling_ic_window :]
|
||||||
|
ics = {}
|
||||||
|
for col in frame.columns:
|
||||||
|
if not win:
|
||||||
|
ics[col] = 1.0
|
||||||
|
continue
|
||||||
|
sub = pd.DataFrame({"p": frame[col], "m": mean_pred})
|
||||||
|
vals = []
|
||||||
|
for d in win:
|
||||||
|
s = sub[sub.index.get_level_values(0) == d]
|
||||||
|
if len(s) >= 3 and s["p"].nunique() > 1 and s["m"].nunique() > 1:
|
||||||
|
vals.append(s["p"].rank().corr(s["m"].rank()))
|
||||||
|
ics[col] = float(np.mean(vals)) if vals else 1.0
|
||||||
|
wsum = sum(ics.values()) or len(ics)
|
||||||
|
weights = {c: v / wsum for c, v in ics.items()}
|
||||||
|
return sum(frame[c] * weights[c] for c in frame.columns)
|
||||||
|
|||||||
@@ -0,0 +1,138 @@
|
|||||||
|
"""TopkDropout with HMM high-volatility + drawdown-pause risk gates.
|
||||||
|
|
||||||
|
Gates NEW entries on two risk conditions (held names are never force-sold):
|
||||||
|
|
||||||
|
1. **HMM high-vol pause**: when the cross-sectional mean of ``sp_hmm_p_regime1``
|
||||||
|
(HMM high-vol regime probability) on the signal date is >= ``hmm_pause_pct``,
|
||||||
|
new buys are paused. The time-series study showed HMM high-vol probability
|
||||||
|
pulses BEFORE sharp moves (regime-change cut) — pausing new exposure at the
|
||||||
|
boundary reduces drawdown from price over-reaction.
|
||||||
|
2. **Drawdown pause**: when the account equity drawdown from its running peak
|
||||||
|
exceeds ``drawdown_pause_pct``, new buys are paused. This is the
|
||||||
|
``drawdown_pause_pct`` risk-limit expressed inside the backtest (the pure
|
||||||
|
executor-side gate is documented as not expressible in a one-shot backtest).
|
||||||
|
3. **Liquidity floor**: names whose 20-day average daily dollar volume is below
|
||||||
|
``liquidity_floor_adv`` are dropped from BUY candidates (the proven mitigant
|
||||||
|
from exp-18: $5M floor cut drawdown 7.9%->5.4% at higher IR).
|
||||||
|
|
||||||
|
Implementation: pre-filter the signal score before the base TopkDropout
|
||||||
|
decision — non-held names get score 0 when any gate fires.
|
||||||
|
|
||||||
|
Wired into a workflow yaml like:
|
||||||
|
|
||||||
|
strategy:
|
||||||
|
class: HmmRiskTopk
|
||||||
|
module_path: tac_qlib.contrib.strategy.hmm_risk
|
||||||
|
kwargs:
|
||||||
|
signal: "<PRED>"
|
||||||
|
topk: 10
|
||||||
|
n_drop: 2
|
||||||
|
only_tradable: true
|
||||||
|
risk_degree: 0.95
|
||||||
|
hmm_pause_pct: 0.70
|
||||||
|
drawdown_pause_pct: 8.0
|
||||||
|
liquidity_floor_adv: 5000000
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import copy
|
||||||
|
from typing import Dict
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import pandas as pd
|
||||||
|
|
||||||
|
from qlib.backtest.decision import TradeDecisionWO
|
||||||
|
from qlib.backtest.position import Position
|
||||||
|
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
|
||||||
|
|
||||||
|
__all__ = ["HmmRiskTopk"]
|
||||||
|
|
||||||
|
|
||||||
|
class HmmRiskTopk(TopkDropoutStrategy):
|
||||||
|
"""TopkDropoutStrategy with HMM high-vol pause + drawdown pause + liquidity floor."""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
hmm_pause_pct: float = 0.70,
|
||||||
|
drawdown_pause_pct: float = 8.0,
|
||||||
|
liquidity_floor_adv: float = 0.0,
|
||||||
|
**kwargs,
|
||||||
|
):
|
||||||
|
super().__init__(**kwargs)
|
||||||
|
self.hmm_pause_pct = float(hmm_pause_pct)
|
||||||
|
self.drawdown_pause_pct = float(drawdown_pause_pct)
|
||||||
|
self.liquidity_floor_adv = float(liquidity_floor_adv)
|
||||||
|
self._peak_equity = 0.0
|
||||||
|
|
||||||
|
# ------------------------------------------------------------- gates
|
||||||
|
def _hmm_high_vol(self, pred_date) -> bool:
|
||||||
|
"""Cross-sectional mean HMM high-vol regime probability >= threshold."""
|
||||||
|
try:
|
||||||
|
from qlib.data import D
|
||||||
|
|
||||||
|
feat = D.features(D.instruments("all"), ["$sp_hmm_p_regime1"],
|
||||||
|
start_time=pred_date, end_time=pred_date)
|
||||||
|
if feat is None or len(feat) == 0:
|
||||||
|
return False
|
||||||
|
p = feat["$sp_hmm_p_regime1"].dropna()
|
||||||
|
if len(p) == 0:
|
||||||
|
return False
|
||||||
|
return float(p.mean()) >= self.hmm_pause_pct
|
||||||
|
except Exception:
|
||||||
|
return False
|
||||||
|
|
||||||
|
def _drawdown_active(self, equity: float) -> bool:
|
||||||
|
if self.drawdown_pause_pct <= 0:
|
||||||
|
return False
|
||||||
|
self._peak_equity = max(self._peak_equity, equity)
|
||||||
|
if self._peak_equity <= 0:
|
||||||
|
return False
|
||||||
|
dd = (self._peak_equity - equity) / self._peak_equity * 100.0
|
||||||
|
return dd >= self.drawdown_pause_pct
|
||||||
|
|
||||||
|
def _illiquid(self, codes, asof) -> Dict[str, bool]:
|
||||||
|
if self.liquidity_floor_adv <= 0 or not codes:
|
||||||
|
return {}
|
||||||
|
from tac_qlib.risk_limits import dollar_adv
|
||||||
|
|
||||||
|
adv = dollar_adv(codes, market="US", asof=asof, lookback=20)
|
||||||
|
return {c: adv.get(str(c).upper(), 0.0) < self.liquidity_floor_adv for c in codes}
|
||||||
|
|
||||||
|
# ------------------------------------------------------------- decision
|
||||||
|
def generate_trade_decision(self, execute_result=None):
|
||||||
|
trade_step = self.trade_calendar.get_trade_step()
|
||||||
|
trade_start_time, trade_end_time = self.trade_calendar.get_step_time(trade_step)
|
||||||
|
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(trade_step, shift=1)
|
||||||
|
pred_score = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
|
||||||
|
if pred_score is None:
|
||||||
|
return TradeDecisionWO([], self)
|
||||||
|
if isinstance(pred_score, pd.DataFrame):
|
||||||
|
pred_score = pred_score.iloc[:, 0]
|
||||||
|
|
||||||
|
current_temp = copy.deepcopy(self.trade_position)
|
||||||
|
assert isinstance(current_temp, Position)
|
||||||
|
held = {c for c in current_temp.get_stock_list() if abs(current_temp.get_stock_amount(c)) > 1e-6}
|
||||||
|
|
||||||
|
equity = current_temp.get_cash()
|
||||||
|
for code in held:
|
||||||
|
mark = self.trade_exchange.get_deal_price(
|
||||||
|
stock_id=code, start_time=trade_start_time, end_time=trade_end_time, direction=1
|
||||||
|
)
|
||||||
|
if mark is not None and np.isfinite(mark):
|
||||||
|
equity += abs(current_temp.get_stock_amount(code)) * mark
|
||||||
|
|
||||||
|
hmm_pause = self._hmm_high_vol(str(pd.Timestamp(pred_start_time).date()))
|
||||||
|
dd_pause = self._drawdown_active(equity)
|
||||||
|
buys_paused = hmm_pause or dd_pause
|
||||||
|
|
||||||
|
pred_score = pred_score.copy()
|
||||||
|
if buys_paused or self.liquidity_floor_adv > 0:
|
||||||
|
new_codes = [c for c in pred_score.index if c not in held]
|
||||||
|
illiquid = self._illiquid(new_codes, str(pd.Timestamp(pred_start_time).date()))
|
||||||
|
for code in new_codes:
|
||||||
|
if buys_paused or illiquid.get(code, False):
|
||||||
|
pred_score[code] = -1e9 # cannot enter today
|
||||||
|
|
||||||
|
return super().generate_trade_decision(execute_result)
|
||||||
@@ -0,0 +1,91 @@
|
|||||||
|
"""TopkDropout with a 1-day momentum entry-confirmation gate.
|
||||||
|
|
||||||
|
Gates NEW entries on short-term momentum: a name that is not currently held
|
||||||
|
may only be bought when its trailing 1-day return is above ``min_momentum``
|
||||||
|
(Lag-1 autocorr ~ +0.45 in the time-series study => short-term momentum
|
||||||
|
continuation). Held names are never force-sold by this gate — exits stay the
|
||||||
|
pure TopkDropout rule.
|
||||||
|
|
||||||
|
Implementation: override ``generate_trade_decision`` and zero out the signal
|
||||||
|
score of any non-held name that fails the momentum check BEFORE calling the
|
||||||
|
base TopkDropout decision, so it can never be selected as a buy candidate.
|
||||||
|
This is a clean pre-filter: the rest of the strategy (top-k, n_drop, sizing,
|
||||||
|
costs) is untouched.
|
||||||
|
|
||||||
|
Wired into a workflow yaml like:
|
||||||
|
|
||||||
|
strategy:
|
||||||
|
class: MomentumGateTopk
|
||||||
|
module_path: tac_qlib.contrib.strategy.momentum_gate
|
||||||
|
kwargs:
|
||||||
|
signal: "<PRED>"
|
||||||
|
topk: 10
|
||||||
|
n_drop: 2
|
||||||
|
only_tradable: true
|
||||||
|
risk_degree: 0.95
|
||||||
|
min_momentum: 0.0
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import copy
|
||||||
|
|
||||||
|
import pandas as pd
|
||||||
|
|
||||||
|
from qlib.backtest.decision import TradeDecisionWO
|
||||||
|
from qlib.backtest.position import Position
|
||||||
|
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
|
||||||
|
|
||||||
|
__all__ = ["MomentumGateTopk"]
|
||||||
|
|
||||||
|
|
||||||
|
class MomentumGateTopk(TopkDropoutStrategy):
|
||||||
|
"""TopkDropoutStrategy gated on 1-day momentum for new entries."""
|
||||||
|
|
||||||
|
def __init__(self, *, min_momentum: float = 0.0, **kwargs):
|
||||||
|
super().__init__(**kwargs)
|
||||||
|
self.min_momentum = float(min_momentum)
|
||||||
|
|
||||||
|
def _momentum_ok(self, code, trade_start, trade_end) -> bool:
|
||||||
|
"""True when the trailing 1-day return is above the momentum floor."""
|
||||||
|
try:
|
||||||
|
cur = self.trade_exchange.get_deal_price(
|
||||||
|
stock_id=code, start_time=trade_start, end_time=trade_end, direction=1
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
return False
|
||||||
|
if cur is None or cur != cur or cur <= 0:
|
||||||
|
return False
|
||||||
|
prev_start = trade_start - pd.Timedelta(days=5)
|
||||||
|
prev_end = trade_start - pd.Timedelta(seconds=1)
|
||||||
|
prev = self.trade_exchange.get_deal_price(
|
||||||
|
stock_id=code, start_time=prev_start, end_time=prev_end, direction=0
|
||||||
|
)
|
||||||
|
if prev is None or prev != prev or prev <= 0:
|
||||||
|
return False
|
||||||
|
return (cur / prev - 1.0) >= self.min_momentum
|
||||||
|
|
||||||
|
def generate_trade_decision(self, execute_result=None):
|
||||||
|
trade_step = self.trade_calendar.get_trade_step()
|
||||||
|
trade_start_time, trade_end_time = self.trade_calendar.get_step_time(trade_step)
|
||||||
|
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(trade_step, shift=1)
|
||||||
|
pred_score = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
|
||||||
|
if pred_score is None:
|
||||||
|
return TradeDecisionWO([], self)
|
||||||
|
if isinstance(pred_score, pd.DataFrame):
|
||||||
|
pred_score = pred_score.iloc[:, 0]
|
||||||
|
|
||||||
|
current_temp = copy.deepcopy(self.trade_position)
|
||||||
|
assert isinstance(current_temp, Position)
|
||||||
|
held = set(current_temp.get_stock_list())
|
||||||
|
held = {c for c in held if abs(current_temp.get_stock_amount(c)) > 1e-6}
|
||||||
|
|
||||||
|
# pre-filter: zero the score of non-held names that fail momentum
|
||||||
|
pred_score = pred_score.copy()
|
||||||
|
for code in pred_score.index:
|
||||||
|
if code in held:
|
||||||
|
continue # never gate exits / re-balancing of held names
|
||||||
|
if not self._momentum_ok(code, trade_start_time, trade_end_time):
|
||||||
|
pred_score[code] = -1e9 # cannot enter today
|
||||||
|
|
||||||
|
return super().generate_trade_decision(execute_result)
|
||||||
@@ -1,20 +0,0 @@
|
|||||||
# exp/10 sp5d-moment-features
|
|
||||||
|
|
||||||
Variant C: generic-only 19 + 16 new moment/volatility families (skew, kurt,
|
|
||||||
DSV+ratios, max_up/down, rv_ac1, rv_cv_22, sig lag-5). 35 sp_* fields, ou/hmm excluded.
|
|
||||||
|
|
||||||
Run a3f7d1d40c3d4b839314fcf5b40f9b08 (tac-rd-moments / exp 12) — FINISHED.
|
|
||||||
|
|
||||||
## Result: NEGATIVE (regression vs generic-only baseline)
|
|
||||||
|
|
||||||
| Metric | generic-only 19 (run 7b1e797) | +moments 35 (run a3f7d1d) |
|
|
||||||
|---|---|---|
|
|
||||||
| Rank IC | 0.0635 | 0.0466 |
|
|
||||||
| Rank ICIR | 0.276 | 0.183 |
|
|
||||||
| L-S Sharpe | 2.55 | 1.44 |
|
|
||||||
| net excess (cost) | +3.1% IR 0.28 | -16.2% IR -1.57 |
|
|
||||||
| MDD | -7.3% | -11.1% |
|
|
||||||
|
|
||||||
Same failure mode as ou/hmm in exp 9: adding cross-sectional moment features
|
|
||||||
to the 50-name panel degrades the rank signal. Generic-only 19 remains the
|
|
||||||
best configuration. No further moment-family variants planned.
|
|
||||||
@@ -0,0 +1,141 @@
|
|||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
# ISOLATION: multi-seed RankIC ensemble, ablate-B generic-only feature set.
|
||||||
|
#
|
||||||
|
# Isolates the ensemble effect on the SP-5d rank signal. Same panel, segments,
|
||||||
|
# history (full backfilled 2016+) and feature set as the exp-9 ablate-B winner
|
||||||
|
# (generic-only sp_* families: jump,har,trend,hurst,signature), but replaces the
|
||||||
|
# single RankICLGBModel with a 5-seed RankICEnsembleLGBModel (42,7,2026,99,123)
|
||||||
|
# that averages per-day predictions.
|
||||||
|
#
|
||||||
|
# Differs from exp-15 (tac-rd-rank-ensemble, mlflow exp 15) ONLY by dropping the
|
||||||
|
# TA subset (rsi_14,roc_10,macd_hist,willr_14,atr_14) and the inter-asset xr_*
|
||||||
|
# features, so any change vs exp-15 is attributable to the feature set alone,
|
||||||
|
# and any change vs exp-9 is attributable to the ensemble + full history alone.
|
||||||
|
#
|
||||||
|
# Run:
|
||||||
|
# rd_run_workflow config_path=experiments/workflows/exp12_isolation_ensemble.yaml \
|
||||||
|
# experiment_name=tac-rd-rank-ensemble-isolated
|
||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
{%- set LAKE = TAC_LAKE_DIR %}
|
||||||
|
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||||
|
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||||
|
|
||||||
|
qlib_init:
|
||||||
|
provider_uri: "{{ LAKE }}"
|
||||||
|
region: us
|
||||||
|
expression_cache: null
|
||||||
|
dataset_cache: null
|
||||||
|
|
||||||
|
calendar_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
instrument_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
markets: {}
|
||||||
|
feature_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
|
||||||
|
exp_manager:
|
||||||
|
class: MLflowExpManager
|
||||||
|
module_path: qlib.workflow.expm
|
||||||
|
kwargs:
|
||||||
|
uri: "sqlite:///mlruns.db"
|
||||||
|
default_exp_name: "tac-rd-rank-ensemble-isolated"
|
||||||
|
|
||||||
|
task:
|
||||||
|
model:
|
||||||
|
class: RankICEnsembleLGBModel
|
||||||
|
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||||
|
kwargs:
|
||||||
|
loss: mse
|
||||||
|
learning_rate: 0.02
|
||||||
|
num_leaves: 31
|
||||||
|
n_estimators: 3000
|
||||||
|
num_boost_round: 3000
|
||||||
|
early_stopping_rounds: 200
|
||||||
|
min_data_in_leaf: 20
|
||||||
|
lambda_l2: 0.5
|
||||||
|
colsample_bytree: 0.8
|
||||||
|
subsample: 0.8
|
||||||
|
subsample_freq: 1
|
||||||
|
reg_alpha: 0.1
|
||||||
|
reg_lambda: 1.0
|
||||||
|
seeds: "42,7,2026,99,123"
|
||||||
|
|
||||||
|
dataset:
|
||||||
|
class: DatasetH
|
||||||
|
module_path: qlib.data.dataset
|
||||||
|
kwargs:
|
||||||
|
handler:
|
||||||
|
class: TACHandler
|
||||||
|
module_path: tac_qlib.contrib.data.handler
|
||||||
|
kwargs:
|
||||||
|
instruments: "{{ UNIVERSE }}"
|
||||||
|
start_time: 2015-01-03
|
||||||
|
end_time: 2026-08-14
|
||||||
|
fit_start_time: 2016-01-04
|
||||||
|
fit_end_time: 2025-09-01
|
||||||
|
freq: day
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||||
|
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||||
|
infer_processors:
|
||||||
|
- class: DropAllNaN
|
||||||
|
kwargs: {}
|
||||||
|
- class: ProcessInf
|
||||||
|
kwargs: {}
|
||||||
|
- class: CSRankNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: ZScoreNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: Fillna
|
||||||
|
kwargs: {}
|
||||||
|
segments:
|
||||||
|
train: [2016-01-04, 2025-09-01]
|
||||||
|
valid: [2025-09-03, 2026-01-03]
|
||||||
|
test: [2026-01-04, 2026-08-10]
|
||||||
|
|
||||||
|
record:
|
||||||
|
- class: SignalRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs: {}
|
||||||
|
- class: SigAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
ana_long_short: true
|
||||||
|
ann_scaler: 252
|
||||||
|
- class: PortAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
config:
|
||||||
|
strategy:
|
||||||
|
class: TopkDropoutStrategy
|
||||||
|
module_path: qlib.contrib.strategy
|
||||||
|
kwargs:
|
||||||
|
signal: "<PRED>"
|
||||||
|
topk: 10
|
||||||
|
n_drop: 2
|
||||||
|
only_tradable: true
|
||||||
|
risk_degree: 0.95
|
||||||
|
backtest:
|
||||||
|
start_time: 2026-01-04
|
||||||
|
end_time: 2026-08-10
|
||||||
|
account: 1000000
|
||||||
|
benchmark: SPY
|
||||||
|
exchange_kwargs:
|
||||||
|
codes: "{{ UNIVERSE }}"
|
||||||
|
deal_price: $close
|
||||||
|
freq: day
|
||||||
|
open_cost: 0.0005
|
||||||
|
close_cost: 0.0015
|
||||||
|
min_cost: 5.0
|
||||||
|
risk_analysis_freq: 1d
|
||||||
+22
-20
@@ -1,22 +1,23 @@
|
|||||||
# -----------------------------------------------------------------------------
|
# -----------------------------------------------------------------------------
|
||||||
# VARIANT C (generic + moments): keeps the winning generic-only 19-field set
|
# EXP 18 - Risk-limit control: reference model + TopkDropout baseline (A).
|
||||||
# (jump,har,trend,hurst,signature) and adds the NEW generic moment families the
|
#
|
||||||
# engine now exposes:
|
# Signal/model identical to the reference (tac-rd-rank-ensemble-isolated,
|
||||||
# - realized skewness / kurtosis (sp_rskew_5, sp_rskew_22, sp_rkurt_5, sp_rkurt_22)
|
# run 0cea66d9...): RankICEnsembleLGBModel (parallel, 5 seeds) on the 50-ETF
|
||||||
# - downside semi-variance + ratios (sp_dsv_1/5/22, sp_dsv_ratio_1/5/22)
|
# SP-5d panel, test 2026-01-04..2026-08-10. This workflow reproduces the
|
||||||
# - signed max moves (sp_max_up, sp_max_down)
|
# unconstrained TopkDropout baseline net-of-cost so the risk-limited variant
|
||||||
# - RV autocorr / vol-of-vol (sp_rv_ac1, sp_rv_cv_22)
|
# (same pred, liquidity/size/concentration caps) can be compared 1:1.
|
||||||
# - longer-lag signature terms (sp_sig_level2_*_5)
|
#
|
||||||
# Drops the model-specific ou/hmm families (they scored high in importance but
|
# The risk_limits spec itself is applied via rd_backtest / rd_strategy_targets
|
||||||
# hurt the rank dimension in the all-24 run). Same panel/model as baseline.
|
# (tool-level param, not a YAML key); this run records the unconstrained
|
||||||
|
# baseline that the limit A/B is measured against.
|
||||||
#
|
#
|
||||||
# Run:
|
# Run:
|
||||||
# rd_run_workflow config_path=experiments/workflows/ablate_generic_moments.yaml \
|
# rd_run_workflow config_path=experiments/workflows/exp18-risk-limit/a_baseline.yaml \
|
||||||
# experiment_name=tac-rd-moments
|
# experiment_name=tac-rd-risk-limit
|
||||||
# -----------------------------------------------------------------------------
|
# -----------------------------------------------------------------------------
|
||||||
{%- set LAKE = TAC_LAKE_DIR %}
|
{%- set LAKE = TAC_LAKE_DIR %}
|
||||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||||
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_max_up,sp_max_down,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_rv_ac1,sp_rv_cv_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead,sp_sig_level2_lead_lag_5,sp_sig_level2_lag_lead_5,sp_rskew_5,sp_rskew_22,sp_rkurt_5,sp_rkurt_22,sp_dsv_1,sp_dsv_5,sp_dsv_22,sp_dsv_ratio_1,sp_dsv_ratio_5,sp_dsv_ratio_22" %}
|
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||||
|
|
||||||
qlib_init:
|
qlib_init:
|
||||||
provider_uri: "{{ LAKE }}"
|
provider_uri: "{{ LAKE }}"
|
||||||
@@ -46,12 +47,12 @@ qlib_init:
|
|||||||
module_path: qlib.workflow.expm
|
module_path: qlib.workflow.expm
|
||||||
kwargs:
|
kwargs:
|
||||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||||
default_exp_name: "tac-rd-moments"
|
default_exp_name: "tac-rd-risk-limit"
|
||||||
|
|
||||||
task:
|
task:
|
||||||
model:
|
model:
|
||||||
class: RankICLGBModel
|
class: RankICEnsembleLGBModel
|
||||||
module_path: tac_qlib.contrib.model.rank_gbdt
|
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||||
kwargs:
|
kwargs:
|
||||||
loss: mse
|
loss: mse
|
||||||
learning_rate: 0.02
|
learning_rate: 0.02
|
||||||
@@ -66,7 +67,8 @@ task:
|
|||||||
subsample_freq: 1
|
subsample_freq: 1
|
||||||
reg_alpha: 0.1
|
reg_alpha: 0.1
|
||||||
reg_lambda: 1.0
|
reg_lambda: 1.0
|
||||||
seed: 42
|
seeds: "42,7,2026,99,123"
|
||||||
|
parallel: 5
|
||||||
|
|
||||||
dataset:
|
dataset:
|
||||||
class: DatasetH
|
class: DatasetH
|
||||||
@@ -78,8 +80,8 @@ task:
|
|||||||
kwargs:
|
kwargs:
|
||||||
instruments: "{{ UNIVERSE }}"
|
instruments: "{{ UNIVERSE }}"
|
||||||
start_time: 2015-01-03
|
start_time: 2015-01-03
|
||||||
end_time: 2026-08-10
|
end_time: 2026-08-14
|
||||||
fit_start_time: 2015-01-03
|
fit_start_time: 2016-01-04
|
||||||
fit_end_time: 2025-09-01
|
fit_end_time: 2025-09-01
|
||||||
freq: day
|
freq: day
|
||||||
lake_root: "{{ LAKE }}"
|
lake_root: "{{ LAKE }}"
|
||||||
@@ -98,7 +100,7 @@ task:
|
|||||||
- class: Fillna
|
- class: Fillna
|
||||||
kwargs: {}
|
kwargs: {}
|
||||||
segments:
|
segments:
|
||||||
train: [2015-01-03, 2025-09-01]
|
train: [2016-01-04, 2025-09-01]
|
||||||
valid: [2025-09-03, 2026-01-03]
|
valid: [2025-09-03, 2026-01-03]
|
||||||
test: [2026-01-04, 2026-08-10]
|
test: [2026-01-04, 2026-08-10]
|
||||||
|
|
||||||
@@ -0,0 +1,135 @@
|
|||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
|
||||||
|
#
|
||||||
|
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
|
||||||
|
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
|
||||||
|
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
|
||||||
|
#
|
||||||
|
# Run:
|
||||||
|
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
|
||||||
|
# experiment_name=tac-rd-risk-limit
|
||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
{%- set LAKE = TAC_LAKE_DIR %}
|
||||||
|
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||||
|
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||||
|
|
||||||
|
qlib_init:
|
||||||
|
provider_uri: "{{ LAKE }}"
|
||||||
|
region: us
|
||||||
|
expression_cache: null
|
||||||
|
dataset_cache: null
|
||||||
|
|
||||||
|
calendar_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
instrument_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
markets: {}
|
||||||
|
feature_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
|
||||||
|
exp_manager:
|
||||||
|
class: MLflowExpManager
|
||||||
|
module_path: qlib.workflow.expm
|
||||||
|
kwargs:
|
||||||
|
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||||
|
default_exp_name: "tac-rd-risk-limit"
|
||||||
|
|
||||||
|
task:
|
||||||
|
model:
|
||||||
|
class: RankICEnsembleLGBModel
|
||||||
|
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||||
|
kwargs:
|
||||||
|
loss: mse
|
||||||
|
learning_rate: 0.02
|
||||||
|
num_leaves: 31
|
||||||
|
n_estimators: 3000
|
||||||
|
num_boost_round: 3000
|
||||||
|
early_stopping_rounds: 200
|
||||||
|
min_data_in_leaf: 20
|
||||||
|
lambda_l2: 0.5
|
||||||
|
colsample_bytree: 0.8
|
||||||
|
subsample: 0.8
|
||||||
|
subsample_freq: 1
|
||||||
|
reg_alpha: 0.1
|
||||||
|
reg_lambda: 1.0
|
||||||
|
seeds: "42,7,2026,99,123"
|
||||||
|
parallel: 5
|
||||||
|
|
||||||
|
dataset:
|
||||||
|
class: DatasetH
|
||||||
|
module_path: qlib.data.dataset
|
||||||
|
kwargs:
|
||||||
|
handler:
|
||||||
|
class: TACHandler
|
||||||
|
module_path: tac_qlib.contrib.data.handler
|
||||||
|
kwargs:
|
||||||
|
instruments: "{{ UNIVERSE }}"
|
||||||
|
start_time: 2015-01-03
|
||||||
|
end_time: 2026-08-14
|
||||||
|
fit_start_time: 2016-01-04
|
||||||
|
fit_end_time: 2025-09-01
|
||||||
|
freq: day
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||||
|
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||||
|
infer_processors:
|
||||||
|
- class: DropAllNaN
|
||||||
|
kwargs: {}
|
||||||
|
- class: ProcessInf
|
||||||
|
kwargs: {}
|
||||||
|
- class: CSRankNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: ZScoreNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: Fillna
|
||||||
|
kwargs: {}
|
||||||
|
segments:
|
||||||
|
train: [2016-01-04, 2025-09-01]
|
||||||
|
valid: [2025-09-03, 2026-01-03]
|
||||||
|
test: [2026-01-04, 2026-08-10]
|
||||||
|
|
||||||
|
record:
|
||||||
|
- class: SignalRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs: {}
|
||||||
|
- class: SigAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
ana_long_short: true
|
||||||
|
ann_scaler: 252
|
||||||
|
- class: PortAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
config:
|
||||||
|
strategy:
|
||||||
|
class: TopkDropoutStrategy
|
||||||
|
module_path: qlib.contrib.strategy
|
||||||
|
kwargs:
|
||||||
|
signal: "<PRED>"
|
||||||
|
topk: 10
|
||||||
|
n_drop: 2
|
||||||
|
only_tradable: true
|
||||||
|
risk_degree: 0.95
|
||||||
|
backtest:
|
||||||
|
start_time: 2026-01-04
|
||||||
|
end_time: 2026-08-10
|
||||||
|
account: 1000000
|
||||||
|
benchmark: SPY
|
||||||
|
exchange_kwargs:
|
||||||
|
codes: "{{ UNIVERSE }}"
|
||||||
|
deal_price: $close
|
||||||
|
freq: day
|
||||||
|
open_cost: 0.0005
|
||||||
|
close_cost: 0.0015
|
||||||
|
min_cost: 5.0
|
||||||
|
risk_analysis_freq: 1d
|
||||||
@@ -0,0 +1,135 @@
|
|||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
|
||||||
|
#
|
||||||
|
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
|
||||||
|
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
|
||||||
|
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
|
||||||
|
#
|
||||||
|
# Run:
|
||||||
|
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
|
||||||
|
# experiment_name=tac-rd-risk-limit
|
||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
{%- set LAKE = TAC_LAKE_DIR %}
|
||||||
|
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||||
|
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||||
|
|
||||||
|
qlib_init:
|
||||||
|
provider_uri: "{{ LAKE }}"
|
||||||
|
region: us
|
||||||
|
expression_cache: null
|
||||||
|
dataset_cache: null
|
||||||
|
|
||||||
|
calendar_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
instrument_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
markets: {}
|
||||||
|
feature_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
|
||||||
|
exp_manager:
|
||||||
|
class: MLflowExpManager
|
||||||
|
module_path: qlib.workflow.expm
|
||||||
|
kwargs:
|
||||||
|
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||||
|
default_exp_name: "tac-rd-risk-limit"
|
||||||
|
|
||||||
|
task:
|
||||||
|
model:
|
||||||
|
class: RankICEnsembleLGBModel
|
||||||
|
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||||
|
kwargs:
|
||||||
|
loss: mse
|
||||||
|
learning_rate: 0.02
|
||||||
|
num_leaves: 31
|
||||||
|
n_estimators: 3000
|
||||||
|
num_boost_round: 3000
|
||||||
|
early_stopping_rounds: 200
|
||||||
|
min_data_in_leaf: 20
|
||||||
|
lambda_l2: 0.5
|
||||||
|
colsample_bytree: 0.8
|
||||||
|
subsample: 0.8
|
||||||
|
subsample_freq: 1
|
||||||
|
reg_alpha: 0.1
|
||||||
|
reg_lambda: 1.0
|
||||||
|
seeds: "42,7,2026,99,123"
|
||||||
|
parallel: 1
|
||||||
|
|
||||||
|
dataset:
|
||||||
|
class: DatasetH
|
||||||
|
module_path: qlib.data.dataset
|
||||||
|
kwargs:
|
||||||
|
handler:
|
||||||
|
class: TACHandler
|
||||||
|
module_path: tac_qlib.contrib.data.handler
|
||||||
|
kwargs:
|
||||||
|
instruments: "{{ UNIVERSE }}"
|
||||||
|
start_time: 2015-01-03
|
||||||
|
end_time: 2026-08-14
|
||||||
|
fit_start_time: 2016-01-04
|
||||||
|
fit_end_time: 2025-09-01
|
||||||
|
freq: day
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||||
|
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||||
|
infer_processors:
|
||||||
|
- class: DropAllNaN
|
||||||
|
kwargs: {}
|
||||||
|
- class: ProcessInf
|
||||||
|
kwargs: {}
|
||||||
|
- class: CSRankNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: ZScoreNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: Fillna
|
||||||
|
kwargs: {}
|
||||||
|
segments:
|
||||||
|
train: [2016-01-04, 2025-09-01]
|
||||||
|
valid: [2025-09-03, 2026-01-03]
|
||||||
|
test: [2026-01-04, 2026-08-10]
|
||||||
|
|
||||||
|
record:
|
||||||
|
- class: SignalRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs: {}
|
||||||
|
- class: SigAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
ana_long_short: true
|
||||||
|
ann_scaler: 252
|
||||||
|
- class: PortAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
config:
|
||||||
|
strategy:
|
||||||
|
class: TopkDropoutStrategy
|
||||||
|
module_path: qlib.contrib.strategy
|
||||||
|
kwargs:
|
||||||
|
signal: "<PRED>"
|
||||||
|
topk: 10
|
||||||
|
n_drop: 2
|
||||||
|
only_tradable: true
|
||||||
|
risk_degree: 0.95
|
||||||
|
backtest:
|
||||||
|
start_time: 2026-01-04
|
||||||
|
end_time: 2026-08-10
|
||||||
|
account: 1000000
|
||||||
|
benchmark: SPY
|
||||||
|
exchange_kwargs:
|
||||||
|
codes: "{{ UNIVERSE }}"
|
||||||
|
deal_price: $close
|
||||||
|
freq: day
|
||||||
|
open_cost: 0.0005
|
||||||
|
close_cost: 0.0015
|
||||||
|
min_cost: 5.0
|
||||||
|
risk_analysis_freq: 1d
|
||||||
@@ -0,0 +1,135 @@
|
|||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
|
||||||
|
#
|
||||||
|
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
|
||||||
|
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
|
||||||
|
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
|
||||||
|
#
|
||||||
|
# Run:
|
||||||
|
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
|
||||||
|
# experiment_name=tac-rd-risk-limit
|
||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
{%- set LAKE = TAC_LAKE_DIR %}
|
||||||
|
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||||
|
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||||
|
|
||||||
|
qlib_init:
|
||||||
|
provider_uri: "{{ LAKE }}"
|
||||||
|
region: us
|
||||||
|
expression_cache: null
|
||||||
|
dataset_cache: null
|
||||||
|
|
||||||
|
calendar_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
instrument_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
markets: {}
|
||||||
|
feature_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
|
||||||
|
exp_manager:
|
||||||
|
class: MLflowExpManager
|
||||||
|
module_path: qlib.workflow.expm
|
||||||
|
kwargs:
|
||||||
|
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||||
|
default_exp_name: "tac-rd-risk-limit"
|
||||||
|
|
||||||
|
task:
|
||||||
|
model:
|
||||||
|
class: RankICEnsembleLGBModel
|
||||||
|
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||||
|
kwargs:
|
||||||
|
loss: mse
|
||||||
|
learning_rate: 0.02
|
||||||
|
num_leaves: 31
|
||||||
|
n_estimators: 3000
|
||||||
|
num_boost_round: 3000
|
||||||
|
early_stopping_rounds: 200
|
||||||
|
min_data_in_leaf: 20
|
||||||
|
lambda_l2: 0.5
|
||||||
|
colsample_bytree: 0.8
|
||||||
|
subsample: 0.8
|
||||||
|
subsample_freq: 1
|
||||||
|
reg_alpha: 0.1
|
||||||
|
reg_lambda: 1.0
|
||||||
|
seeds: "42,7"
|
||||||
|
parallel: 2
|
||||||
|
|
||||||
|
dataset:
|
||||||
|
class: DatasetH
|
||||||
|
module_path: qlib.data.dataset
|
||||||
|
kwargs:
|
||||||
|
handler:
|
||||||
|
class: TACHandler
|
||||||
|
module_path: tac_qlib.contrib.data.handler
|
||||||
|
kwargs:
|
||||||
|
instruments: "{{ UNIVERSE }}"
|
||||||
|
start_time: 2015-01-03
|
||||||
|
end_time: 2026-08-14
|
||||||
|
fit_start_time: 2016-01-04
|
||||||
|
fit_end_time: 2025-09-01
|
||||||
|
freq: day
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||||
|
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||||
|
infer_processors:
|
||||||
|
- class: DropAllNaN
|
||||||
|
kwargs: {}
|
||||||
|
- class: ProcessInf
|
||||||
|
kwargs: {}
|
||||||
|
- class: CSRankNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: ZScoreNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: Fillna
|
||||||
|
kwargs: {}
|
||||||
|
segments:
|
||||||
|
train: [2016-01-04, 2025-09-01]
|
||||||
|
valid: [2025-09-03, 2026-01-03]
|
||||||
|
test: [2026-01-04, 2026-08-10]
|
||||||
|
|
||||||
|
record:
|
||||||
|
- class: SignalRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs: {}
|
||||||
|
- class: SigAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
ana_long_short: true
|
||||||
|
ann_scaler: 252
|
||||||
|
- class: PortAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
config:
|
||||||
|
strategy:
|
||||||
|
class: TopkDropoutStrategy
|
||||||
|
module_path: qlib.contrib.strategy
|
||||||
|
kwargs:
|
||||||
|
signal: "<PRED>"
|
||||||
|
topk: 10
|
||||||
|
n_drop: 2
|
||||||
|
only_tradable: true
|
||||||
|
risk_degree: 0.95
|
||||||
|
backtest:
|
||||||
|
start_time: 2026-01-04
|
||||||
|
end_time: 2026-08-10
|
||||||
|
account: 1000000
|
||||||
|
benchmark: SPY
|
||||||
|
exchange_kwargs:
|
||||||
|
codes: "{{ UNIVERSE }}"
|
||||||
|
deal_price: $close
|
||||||
|
freq: day
|
||||||
|
open_cost: 0.0005
|
||||||
|
close_cost: 0.0015
|
||||||
|
min_cost: 5.0
|
||||||
|
risk_analysis_freq: 1d
|
||||||
@@ -0,0 +1,136 @@
|
|||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
|
||||||
|
#
|
||||||
|
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
|
||||||
|
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
|
||||||
|
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
|
||||||
|
#
|
||||||
|
# Run:
|
||||||
|
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
|
||||||
|
# experiment_name=tac-rd-risk-limit
|
||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
{%- set LAKE = TAC_LAKE_DIR %}
|
||||||
|
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||||
|
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||||
|
|
||||||
|
qlib_init:
|
||||||
|
provider_uri: "{{ LAKE }}"
|
||||||
|
region: us
|
||||||
|
expression_cache: null
|
||||||
|
dataset_cache: null
|
||||||
|
|
||||||
|
calendar_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
instrument_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
markets: {}
|
||||||
|
feature_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
|
||||||
|
exp_manager:
|
||||||
|
class: MLflowExpManager
|
||||||
|
module_path: qlib.workflow.expm
|
||||||
|
kwargs:
|
||||||
|
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||||
|
default_exp_name: "tac-rd-risk-limit"
|
||||||
|
|
||||||
|
task:
|
||||||
|
model:
|
||||||
|
class: RankICEnsembleLGBModel
|
||||||
|
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||||
|
kwargs:
|
||||||
|
loss: mse
|
||||||
|
learning_rate: 0.02
|
||||||
|
num_leaves: 31
|
||||||
|
n_estimators: 3000
|
||||||
|
num_boost_round: 3000
|
||||||
|
early_stopping_rounds: 200
|
||||||
|
min_data_in_leaf: 20
|
||||||
|
lambda_l2: 0.5
|
||||||
|
colsample_bytree: 0.8
|
||||||
|
subsample: 0.8
|
||||||
|
subsample_freq: 1
|
||||||
|
reg_alpha: 0.1
|
||||||
|
reg_lambda: 1.0
|
||||||
|
seeds: "42,7,2026,99,123"
|
||||||
|
parallel: 5
|
||||||
|
|
||||||
|
dataset:
|
||||||
|
class: DatasetH
|
||||||
|
module_path: qlib.data.dataset
|
||||||
|
kwargs:
|
||||||
|
handler:
|
||||||
|
class: TACHandler
|
||||||
|
module_path: tac_qlib.contrib.data.handler
|
||||||
|
kwargs:
|
||||||
|
instruments: "{{ UNIVERSE }}"
|
||||||
|
start_time: 2015-01-03
|
||||||
|
end_time: 2026-08-14
|
||||||
|
fit_start_time: 2016-01-04
|
||||||
|
fit_end_time: 2025-09-01
|
||||||
|
freq: day
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||||
|
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||||
|
infer_processors:
|
||||||
|
- class: DropAllNaN
|
||||||
|
kwargs: {}
|
||||||
|
- class: ProcessInf
|
||||||
|
kwargs: {}
|
||||||
|
- class: CSRankNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: ZScoreNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: Fillna
|
||||||
|
kwargs: {}
|
||||||
|
segments:
|
||||||
|
train: [2016-01-04, 2025-09-01]
|
||||||
|
valid: [2025-09-03, 2026-01-03]
|
||||||
|
test: [2026-01-04, 2026-08-10]
|
||||||
|
|
||||||
|
record:
|
||||||
|
- class: SignalRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs: {}
|
||||||
|
- class: SigAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
ana_long_short: true
|
||||||
|
ann_scaler: 252
|
||||||
|
- class: PortAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
config:
|
||||||
|
strategy:
|
||||||
|
class: MomentumGateTopk
|
||||||
|
module_path: tac_qlib.contrib.strategy.momentum_gate
|
||||||
|
kwargs:
|
||||||
|
signal: "<PRED>"
|
||||||
|
topk: 10
|
||||||
|
n_drop: 2
|
||||||
|
min_momentum: 0.0
|
||||||
|
only_tradable: true
|
||||||
|
risk_degree: 0.95
|
||||||
|
backtest:
|
||||||
|
start_time: 2026-01-04
|
||||||
|
end_time: 2026-08-10
|
||||||
|
account: 1000000
|
||||||
|
benchmark: SPY
|
||||||
|
exchange_kwargs:
|
||||||
|
codes: "{{ UNIVERSE }}"
|
||||||
|
deal_price: $close
|
||||||
|
freq: day
|
||||||
|
open_cost: 0.0005
|
||||||
|
close_cost: 0.0015
|
||||||
|
min_cost: 5.0
|
||||||
|
risk_analysis_freq: 1d
|
||||||
@@ -0,0 +1,138 @@
|
|||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
|
||||||
|
#
|
||||||
|
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
|
||||||
|
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
|
||||||
|
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
|
||||||
|
#
|
||||||
|
# Run:
|
||||||
|
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
|
||||||
|
# experiment_name=tac-rd-risk-limit
|
||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
{%- set LAKE = TAC_LAKE_DIR %}
|
||||||
|
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||||
|
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||||
|
|
||||||
|
qlib_init:
|
||||||
|
provider_uri: "{{ LAKE }}"
|
||||||
|
region: us
|
||||||
|
expression_cache: null
|
||||||
|
dataset_cache: null
|
||||||
|
|
||||||
|
calendar_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
instrument_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
markets: {}
|
||||||
|
feature_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
|
||||||
|
exp_manager:
|
||||||
|
class: MLflowExpManager
|
||||||
|
module_path: qlib.workflow.expm
|
||||||
|
kwargs:
|
||||||
|
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||||
|
default_exp_name: "tac-rd-risk-limit"
|
||||||
|
|
||||||
|
task:
|
||||||
|
model:
|
||||||
|
class: RankICEnsembleLGBModel
|
||||||
|
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||||
|
kwargs:
|
||||||
|
loss: mse
|
||||||
|
learning_rate: 0.02
|
||||||
|
num_leaves: 31
|
||||||
|
n_estimators: 3000
|
||||||
|
num_boost_round: 3000
|
||||||
|
early_stopping_rounds: 200
|
||||||
|
min_data_in_leaf: 20
|
||||||
|
lambda_l2: 0.5
|
||||||
|
colsample_bytree: 0.8
|
||||||
|
subsample: 0.8
|
||||||
|
subsample_freq: 1
|
||||||
|
reg_alpha: 0.1
|
||||||
|
reg_lambda: 1.0
|
||||||
|
seeds: "42,7,2026,99,123"
|
||||||
|
parallel: 5
|
||||||
|
|
||||||
|
dataset:
|
||||||
|
class: DatasetH
|
||||||
|
module_path: qlib.data.dataset
|
||||||
|
kwargs:
|
||||||
|
handler:
|
||||||
|
class: TACHandler
|
||||||
|
module_path: tac_qlib.contrib.data.handler
|
||||||
|
kwargs:
|
||||||
|
instruments: "{{ UNIVERSE }}"
|
||||||
|
start_time: 2015-01-03
|
||||||
|
end_time: 2026-08-14
|
||||||
|
fit_start_time: 2016-01-04
|
||||||
|
fit_end_time: 2025-09-01
|
||||||
|
freq: day
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||||
|
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||||
|
infer_processors:
|
||||||
|
- class: DropAllNaN
|
||||||
|
kwargs: {}
|
||||||
|
- class: ProcessInf
|
||||||
|
kwargs: {}
|
||||||
|
- class: CSRankNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: ZScoreNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: Fillna
|
||||||
|
kwargs: {}
|
||||||
|
segments:
|
||||||
|
train: [2016-01-04, 2025-09-01]
|
||||||
|
valid: [2025-09-03, 2026-01-03]
|
||||||
|
test: [2026-01-04, 2026-08-10]
|
||||||
|
|
||||||
|
record:
|
||||||
|
- class: SignalRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs: {}
|
||||||
|
- class: SigAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
ana_long_short: true
|
||||||
|
ann_scaler: 252
|
||||||
|
- class: PortAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
config:
|
||||||
|
strategy:
|
||||||
|
class: HmmRiskTopk
|
||||||
|
module_path: tac_qlib.contrib.strategy.hmm_risk
|
||||||
|
kwargs:
|
||||||
|
signal: "<PRED>"
|
||||||
|
topk: 10
|
||||||
|
n_drop: 2
|
||||||
|
hmm_pause_pct: 0.70
|
||||||
|
drawdown_pause_pct: 8.0
|
||||||
|
liquidity_floor_adv: 5000000
|
||||||
|
only_tradable: true
|
||||||
|
risk_degree: 0.95
|
||||||
|
backtest:
|
||||||
|
start_time: 2026-01-04
|
||||||
|
end_time: 2026-08-10
|
||||||
|
account: 1000000
|
||||||
|
benchmark: SPY
|
||||||
|
exchange_kwargs:
|
||||||
|
codes: "{{ UNIVERSE }}"
|
||||||
|
deal_price: $close
|
||||||
|
freq: day
|
||||||
|
open_cost: 0.0005
|
||||||
|
close_cost: 0.0015
|
||||||
|
min_cost: 5.0
|
||||||
|
risk_analysis_freq: 1d
|
||||||
@@ -0,0 +1,137 @@
|
|||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
|
||||||
|
#
|
||||||
|
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
|
||||||
|
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
|
||||||
|
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
|
||||||
|
#
|
||||||
|
# Run:
|
||||||
|
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
|
||||||
|
# experiment_name=tac-rd-risk-limit
|
||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
{%- set LAKE = TAC_LAKE_DIR %}
|
||||||
|
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||||
|
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
||||||
|
|
||||||
|
qlib_init:
|
||||||
|
provider_uri: "{{ LAKE }}"
|
||||||
|
region: us
|
||||||
|
expression_cache: null
|
||||||
|
dataset_cache: null
|
||||||
|
|
||||||
|
calendar_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
instrument_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
markets: {}
|
||||||
|
feature_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
|
||||||
|
exp_manager:
|
||||||
|
class: MLflowExpManager
|
||||||
|
module_path: qlib.workflow.expm
|
||||||
|
kwargs:
|
||||||
|
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||||
|
default_exp_name: "tac-rd-risk-limit"
|
||||||
|
|
||||||
|
task:
|
||||||
|
model:
|
||||||
|
class: RankICEnsembleLGBModel
|
||||||
|
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||||
|
kwargs:
|
||||||
|
loss: mse
|
||||||
|
learning_rate: 0.02
|
||||||
|
num_leaves: 31
|
||||||
|
n_estimators: 3000
|
||||||
|
num_boost_round: 3000
|
||||||
|
early_stopping_rounds: 200
|
||||||
|
min_data_in_leaf: 20
|
||||||
|
lambda_l2: 0.5
|
||||||
|
colsample_bytree: 0.8
|
||||||
|
subsample: 0.8
|
||||||
|
subsample_freq: 1
|
||||||
|
reg_alpha: 0.1
|
||||||
|
reg_lambda: 1.0
|
||||||
|
seeds: "42,7,2026,99,123"
|
||||||
|
weight_mode: rolling_ic
|
||||||
|
rolling_ic_window: 21
|
||||||
|
parallel: 5
|
||||||
|
|
||||||
|
dataset:
|
||||||
|
class: DatasetH
|
||||||
|
module_path: qlib.data.dataset
|
||||||
|
kwargs:
|
||||||
|
handler:
|
||||||
|
class: TACHandler
|
||||||
|
module_path: tac_qlib.contrib.data.handler
|
||||||
|
kwargs:
|
||||||
|
instruments: "{{ UNIVERSE }}"
|
||||||
|
start_time: 2015-01-03
|
||||||
|
end_time: 2026-08-14
|
||||||
|
fit_start_time: 2016-01-04
|
||||||
|
fit_end_time: 2025-09-01
|
||||||
|
freq: day
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||||
|
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||||
|
infer_processors:
|
||||||
|
- class: DropAllNaN
|
||||||
|
kwargs: {}
|
||||||
|
- class: ProcessInf
|
||||||
|
kwargs: {}
|
||||||
|
- class: CSRankNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: ZScoreNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: Fillna
|
||||||
|
kwargs: {}
|
||||||
|
segments:
|
||||||
|
train: [2016-01-04, 2025-09-01]
|
||||||
|
valid: [2025-09-03, 2026-01-03]
|
||||||
|
test: [2026-01-04, 2026-08-10]
|
||||||
|
|
||||||
|
record:
|
||||||
|
- class: SignalRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs: {}
|
||||||
|
- class: SigAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
ana_long_short: true
|
||||||
|
ann_scaler: 252
|
||||||
|
- class: PortAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
config:
|
||||||
|
strategy:
|
||||||
|
class: TopkDropoutStrategy
|
||||||
|
module_path: qlib.contrib.strategy
|
||||||
|
kwargs:
|
||||||
|
signal: "<PRED>"
|
||||||
|
topk: 10
|
||||||
|
n_drop: 2
|
||||||
|
only_tradable: true
|
||||||
|
risk_degree: 0.95
|
||||||
|
backtest:
|
||||||
|
start_time: 2026-01-04
|
||||||
|
end_time: 2026-08-10
|
||||||
|
account: 1000000
|
||||||
|
benchmark: SPY
|
||||||
|
exchange_kwargs:
|
||||||
|
codes: "{{ UNIVERSE }}"
|
||||||
|
deal_price: $close
|
||||||
|
freq: day
|
||||||
|
open_cost: 0.0005
|
||||||
|
close_cost: 0.0015
|
||||||
|
min_cost: 5.0
|
||||||
|
risk_analysis_freq: 1d
|
||||||
@@ -0,0 +1,135 @@
|
|||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
# EXP 20 - R1: 2-seed ensemble (seeds 42,7), TopkDropout baseline.
|
||||||
|
#
|
||||||
|
# Runtime cut: 2 seeds instead of 5. Everything else identical to the reference
|
||||||
|
# (test 2026-01-04..2026-08-10, SPY, costs 5bp/15bp). Measures whether the
|
||||||
|
# 2-seed ensemble keeps the reference quality at ~2/5 the training time.
|
||||||
|
#
|
||||||
|
# Run:
|
||||||
|
# rd_run_workflow config_path=experiments/workflows/exp20-risk-limit-improve/r1_2seed.yaml \
|
||||||
|
# experiment_name=tac-rd-risk-limit
|
||||||
|
# -----------------------------------------------------------------------------
|
||||||
|
{%- set LAKE = TAC_LAKE_DIR %}
|
||||||
|
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
||||||
|
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead,sma_3,ema_3" %}
|
||||||
|
|
||||||
|
qlib_init:
|
||||||
|
provider_uri: "{{ LAKE }}"
|
||||||
|
region: us
|
||||||
|
expression_cache: null
|
||||||
|
dataset_cache: null
|
||||||
|
|
||||||
|
calendar_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
instrument_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
markets: {}
|
||||||
|
feature_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
|
||||||
|
exp_manager:
|
||||||
|
class: MLflowExpManager
|
||||||
|
module_path: qlib.workflow.expm
|
||||||
|
kwargs:
|
||||||
|
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
||||||
|
default_exp_name: "tac-rd-risk-limit"
|
||||||
|
|
||||||
|
task:
|
||||||
|
model:
|
||||||
|
class: RankICEnsembleLGBModel
|
||||||
|
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||||
|
kwargs:
|
||||||
|
loss: mse
|
||||||
|
learning_rate: 0.02
|
||||||
|
num_leaves: 31
|
||||||
|
n_estimators: 3000
|
||||||
|
num_boost_round: 3000
|
||||||
|
early_stopping_rounds: 200
|
||||||
|
min_data_in_leaf: 20
|
||||||
|
lambda_l2: 0.5
|
||||||
|
colsample_bytree: 0.8
|
||||||
|
subsample: 0.8
|
||||||
|
subsample_freq: 1
|
||||||
|
reg_alpha: 0.1
|
||||||
|
reg_lambda: 1.0
|
||||||
|
seeds: "42,7,2026,99,123"
|
||||||
|
parallel: 5
|
||||||
|
|
||||||
|
dataset:
|
||||||
|
class: DatasetH
|
||||||
|
module_path: qlib.data.dataset
|
||||||
|
kwargs:
|
||||||
|
handler:
|
||||||
|
class: TACHandler
|
||||||
|
module_path: tac_qlib.contrib.data.handler
|
||||||
|
kwargs:
|
||||||
|
instruments: "{{ UNIVERSE }}"
|
||||||
|
start_time: 2015-01-03
|
||||||
|
end_time: 2026-08-14
|
||||||
|
fit_start_time: 2016-01-04
|
||||||
|
fit_end_time: 2025-09-01
|
||||||
|
freq: day
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||||
|
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
||||||
|
infer_processors:
|
||||||
|
- class: DropAllNaN
|
||||||
|
kwargs: {}
|
||||||
|
- class: ProcessInf
|
||||||
|
kwargs: {}
|
||||||
|
- class: CSRankNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: ZScoreNorm
|
||||||
|
kwargs: {}
|
||||||
|
- class: Fillna
|
||||||
|
kwargs: {}
|
||||||
|
segments:
|
||||||
|
train: [2016-01-04, 2025-09-01]
|
||||||
|
valid: [2025-09-03, 2026-01-03]
|
||||||
|
test: [2026-01-04, 2026-08-10]
|
||||||
|
|
||||||
|
record:
|
||||||
|
- class: SignalRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs: {}
|
||||||
|
- class: SigAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
ana_long_short: true
|
||||||
|
ann_scaler: 252
|
||||||
|
- class: PortAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
config:
|
||||||
|
strategy:
|
||||||
|
class: TopkDropoutStrategy
|
||||||
|
module_path: qlib.contrib.strategy
|
||||||
|
kwargs:
|
||||||
|
signal: "<PRED>"
|
||||||
|
topk: 10
|
||||||
|
n_drop: 2
|
||||||
|
only_tradable: true
|
||||||
|
risk_degree: 0.95
|
||||||
|
backtest:
|
||||||
|
start_time: 2026-01-04
|
||||||
|
end_time: 2026-08-10
|
||||||
|
account: 1000000
|
||||||
|
benchmark: SPY
|
||||||
|
exchange_kwargs:
|
||||||
|
codes: "{{ UNIVERSE }}"
|
||||||
|
deal_price: $close
|
||||||
|
freq: day
|
||||||
|
open_cost: 0.0005
|
||||||
|
close_cost: 0.0015
|
||||||
|
min_cost: 5.0
|
||||||
|
risk_analysis_freq: 1d
|
||||||
Reference in New Issue
Block a user