Compare commits

..
33 changed files with 1671 additions and 359 deletions
+21 -17
View File
@@ -1,27 +1,31 @@
# TradeAC custom-qlib-code snapshot (auto-generated) # TradeAC custom-qlib-code snapshot (auto-generated)
# parent repo HEAD : f9ef005e9aa05e546c11eef760046648cf5a6334 # parent repo HEAD : 70589e3766800c984c1e3e2e1e69892d1f10b3c1
# tac-qlib/tac_qlib/contrib # tac-qlib/tac_qlib/contrib
# tac-qlib/tac_qlib/data # tac-qlib/tac_qlib/data
# per-file hashes (git hash-object): # per-file hashes (git hash-object):
1b6298c4a5652f2e863cbdc385a1014a570fcd59 tac-qlib/tac_qlib/contrib/__init__.py 1b6298c4a5652f2e863cbdc385a1014a570fcd59 tac-qlib/tac_qlib/contrib/__init__.py
b8112569f9b2537c45b6535e1a505a207878d322 tac-qlib/tac_qlib/contrib/__pycache__/__init__.cpython-312.pyc b419ee55ed455a1c45423d1c9025ca5cc0a98576 tac-qlib/tac_qlib/contrib/__pycache__/__init__.cpython-312.pyc
c76a9f17f680e74eea766eff27f7624359749ed6 tac-qlib/tac_qlib/contrib/data/__init__.py c76a9f17f680e74eea766eff27f7624359749ed6 tac-qlib/tac_qlib/contrib/data/__init__.py
8d5333ebd2b44165c50cba639ca2d4ac3fc7cfec tac-qlib/tac_qlib/contrib/data/__pycache__/__init__.cpython-312.pyc 2f6c67620aa2f9e6aaaef3369361d9b3eac3d6ca tac-qlib/tac_qlib/contrib/data/__pycache__/__init__.cpython-312.pyc
18cb37c0354184c49fa2e598396d7df0634cce0f tac-qlib/tac_qlib/contrib/data/__pycache__/handler.cpython-312.pyc fdd5923a70a399e8680913593ff111641947898e tac-qlib/tac_qlib/contrib/data/__pycache__/handler.cpython-312.pyc
871ff1e163c29261f140c3f53d42a41e6504c779 tac-qlib/tac_qlib/contrib/data/handler.py 0dd25ef161c6e0f15eafc84886e7e1381deb38c3 tac-qlib/tac_qlib/contrib/data/handler.py
b151d139a0dcde87d74b21e7c4b729176ba5c39b tac-qlib/tac_qlib/contrib/model/__init__.py b151d139a0dcde87d74b21e7c4b729176ba5c39b tac-qlib/tac_qlib/contrib/model/__init__.py
ab958203f33a99d12c7d923b6efb435189231666 tac-qlib/tac_qlib/contrib/model/__pycache__/__init__.cpython-312.pyc 08dec87ccdf6bb5d2cf611ca3032a4280aaab8cf tac-qlib/tac_qlib/contrib/model/__pycache__/__init__.cpython-312.pyc
9dc36de7e343073b7d511349ee5aede086c38f94 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_ensemble.cpython-312.pyc 6fb61946ea9a83dfb560de3717f5fbf482c4c00e tac-qlib/tac_qlib/contrib/model/__pycache__/rank_ensemble.cpython-312.pyc
9f9014ddd9bce37490061312d51e8e6fe540fec4 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_gbdt.cpython-312.pyc 3e80f2e08b661ddd2f58ffe5a6196063fa41ae51 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_gbdt.cpython-312.pyc
d3f051f3a8650c42fedc7b367b966f7c74fb5789 tac-qlib/tac_qlib/contrib/model/rank_ensemble.py d3f051f3a8650c42fedc7b367b966f7c74fb5789 tac-qlib/tac_qlib/contrib/model/rank_ensemble.py
ccfe7d554989aa7f3e5a2128ae663e51b2207149 tac-qlib/tac_qlib/contrib/model/rank_gbdt.py d03e6611338918d4aac5eea4adf26f85a3763652 tac-qlib/tac_qlib/contrib/model/rank_gbdt.py
4afcf9058231111c412925f4c4b84e81d656db87 tac-qlib/tac_qlib/contrib/strategy/__init__.py c4ef84ffda2a611262412fe1127689c667f3d0c1 tac-qlib/tac_qlib/contrib/strategy/__init__.py
74e5ecbbbb20bb71fd5cd083383de4ce88476712 tac-qlib/tac_qlib/contrib/strategy/__pycache__/__init__.cpython-312.pyc 6ad10c2ebe37c16417e67c7aeb731ad1fcb6da2f tac-qlib/tac_qlib/contrib/strategy/__pycache__/__init__.cpython-312.pyc
afaf562aeaa12cebc8529cd916153252e7e3c38a tac-qlib/tac_qlib/contrib/strategy/__pycache__/optimal_stop.cpython-312.pyc 8d684b3216b040071d9ee4fa920a0e0c7486d278 tac-qlib/tac_qlib/contrib/strategy/__pycache__/optimal_stop.cpython-312.pyc
896ef74ae47bcd1ed388e1e5d9c8d70c28097fe9 tac-qlib/tac_qlib/contrib/strategy/kelly_dropout.py
79aaad9e39fcc740a773f4f63c512ce1086cfde0 tac-qlib/tac_qlib/contrib/strategy/optimal_stop.py 79aaad9e39fcc740a773f4f63c512ce1086cfde0 tac-qlib/tac_qlib/contrib/strategy/optimal_stop.py
5b9acfb4340111b204249add7760bd53c6ae03f1 tac-qlib/tac_qlib/contrib/strategy/regime_gate.py
aa1ee880d52ceb5821d65973962099c2254f710a tac-qlib/tac_qlib/contrib/strategy/top_bottom.py
fe60bacdfedd48617863be31f24b7c7daebfac5a tac-qlib/tac_qlib/contrib/strategy/weekly_rebalance.py
92e6e90eb0cd0a25142034560f27adb6b705b1a8 tac-qlib/tac_qlib/data/__init__.py 92e6e90eb0cd0a25142034560f27adb6b705b1a8 tac-qlib/tac_qlib/data/__init__.py
0ed1ead6c1314a3f25784d453e54a15a8a04baaa tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc 7c4e6c345fad1978efe8860c0d977d0c02d6f8d9 tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc
9609782800944c45b78bb58eaa7b51ba1b7f8f43 tac-qlib/tac_qlib/data/__pycache__/config.cpython-312.pyc 99e602392d51663cb06d5c425000b1ed1e5a916b tac-qlib/tac_qlib/data/__pycache__/config.cpython-312.pyc
a85628d71d12cfe5b18b1c884c5d829c89594579 tac-qlib/tac_qlib/data/__pycache__/providers.cpython-312.pyc 020dcdcf288e4832c8cf2386351f78d5ceb4fe13 tac-qlib/tac_qlib/data/__pycache__/providers.cpython-312.pyc
686d36f6d101c547491ca866aa143aa542e17518 tac-qlib/tac_qlib/data/config.py 53c9007a928841fd3c3b08450f9a6520ce1ac091 tac-qlib/tac_qlib/data/config.py
d9f839be30026f337754a3f015425a8efdbe8e2a tac-qlib/tac_qlib/data/providers.py 8d0644f6f0d1efb94798ed444cc73e63b643459b tac-qlib/tac_qlib/data/providers.py
+30 -12
View File
@@ -64,9 +64,13 @@ def check_transform_proc(proc_l, fit_start_time, fit_end_time):
def get_common_feature_fields(lake_root=None, market="US", timeframe="1d") -> List[str]: def get_common_feature_fields(lake_root=None, market="US", timeframe="1d") -> List[str]:
"""Discover ta-lib columns present in *every* features parquet file of the lake. """Discover feature columns present in *every* feature file of the lake.
Returns sorted field names (without the ``$`` prefix). Empty if no features are persisted. Walks the `family=ta|sp` partition layout (plus any legacy flat files).
TA and SP columns are disjoint by construction, so the common set is
computed per family (columns shared by all symbol files of that family),
then the per-family results are unioned. Returns sorted field names
(without the ``$`` prefix). Empty if no features are persisted.
""" """
cfg = LakeConfig(lake_root, market) cfg = LakeConfig(lake_root, market)
feat_dir = cfg.features_dir(timeframe) feat_dir = cfg.features_dir(timeframe)
@@ -74,16 +78,30 @@ def get_common_feature_fields(lake_root=None, market="US", timeframe="1d") -> Li
return [] return []
import pyarrow.parquet as pq import pyarrow.parquet as pq
common = None def _family_common(fam_dir: Path) -> set:
for p in sorted(feat_dir.glob("symbol=*.parquet")): common = None
try: for p in sorted(fam_dir.glob("symbol=*.parquet")):
cols = set(pq.read_schema(p).names) - set(NON_FEATURE_COLUMNS) try:
except Exception: # pragma: no cover - skip unreadable files cols = set(pq.read_schema(p).names) - set(NON_FEATURE_COLUMNS)
continue except Exception: # pragma: no cover - skip unreadable files
common = cols if common is None else (common & cols) continue
if not common: common = cols if common is None else (common & cols)
break if not common:
return sorted(common) if common else [] break
return common or set()
common: set = set()
# family tier: features/market=*/timeframe=*/family=*/symbol=*.parquet
for fam in ("ta", "sp"):
fam_dir = feat_dir / f"family={fam}"
if fam_dir.is_dir():
common |= _family_common(fam_dir)
# legacy flat: features/market=*/timeframe=*/symbol=*.parquet
if (feat_dir / "family=ta").exists() or (feat_dir / "family=sp").exists():
pass # family layout already covered
else:
common |= _family_common(feat_dir)
return sorted(common)
class DropAllNaN(processor_module.Processor): class DropAllNaN(processor_module.Processor):
@@ -53,23 +53,61 @@ from qlib.workflow import R
__all__ = ["RankICLGBModel", "rankic_feval"] __all__ = ["RankICLGBModel", "rankic_feval"]
def _group_averaged_rank(values: np.ndarray, gid: np.ndarray, offs: np.ndarray) -> np.ndarray:
"""Averaged (tie-corrected) rank of ``values`` within each group, vectorized.
``gid`` maps each row to its group id; ``offs`` holds the cumulative row
offsets so that group ``i`` occupies rows ``[offs[i], offs[i+1])``. Returns
the same result as ``pandas.Series.rank(method='average')`` applied per
group, but in one pass (``np.lexsort`` is the only non-linear step).
"""
n = len(values)
order = np.lexsort((values, gid))
ord_rank = np.empty(n, dtype=np.float64)
ord_rank[order] = np.arange(n, dtype=np.float64) - offs[gid[order]] + 1.0
sg = gid[order]
sv = values[order]
newblock = np.empty(n, dtype=bool)
newblock[0] = True
newblock[1:] = (sg[1:] != sg[:-1]) | (sv[1:] != sv[:-1])
blockid = np.cumsum(newblock) - 1
block_mean = np.bincount(blockid, weights=ord_rank[order]) / np.bincount(blockid)
out = np.empty(n)
out[order] = block_mean[blockid]
return out
def _per_day_spearman(preds: np.ndarray, labels: np.ndarray, group: np.ndarray) -> float: def _per_day_spearman(preds: np.ndarray, labels: np.ndarray, group: np.ndarray) -> float:
"""Mean per-day Spearman rank correlation of preds vs labels. """Mean per-day Spearman rank correlation of preds vs labels.
``group`` holds the number of rows of each trading day (query group), in ``group`` holds the number of rows of each trading day (query group), in
order. Days with <3 valid rows or a constant pred/label are skipped. order. Days with <3 valid rows or a constant pred/label are skipped.
Vectorized: per-day Spearman == Pearson of the per-day rank transforms,
and the Pearson moments (``sum``, ``sum`` of products/squares) aggregate
over each day with ``np.bincount``. Runs ~10x faster than the per-day
``pd.Series.rank()`` loop that preceded it — this feval is invoked on the
train and valid panels every boosting round, per seed.
""" """
if group is None or len(group) == 0: if group is None or len(group) == 0:
return 0.0 return 0.0
offs = np.concatenate([[0], np.cumsum(group.astype(int))]) offs = np.concatenate([[0], np.cumsum(group.astype(int))])
vals = [] gid = np.repeat(np.arange(len(group)), group.astype(int))
for i in range(len(group)): rp = _group_averaged_rank(preds, gid, offs)
s = slice(offs[i], offs[i + 1]) rl = _group_averaged_rank(labels, gid, offs)
p, l = preds[s], labels[s] n_g = group.astype(float)
if len(p) < 3 or np.std(p) == 0 or np.std(l) == 0: s_p = np.bincount(gid, weights=rp)
continue s_l = np.bincount(gid, weights=rl)
vals.append(np.corrcoef(pd.Series(p).rank(), pd.Series(l).rank())[0, 1]) s_pl = np.bincount(gid, weights=rp * rl)
return float(np.mean(vals)) if vals else 0.0 s_pp = np.bincount(gid, weights=rp * rp)
s_ll = np.bincount(gid, weights=rl * rl)
cov = n_g * s_pl - s_p * s_l
var_p = n_g * s_pp - s_p ** 2
var_l = n_g * s_ll - s_l ** 2
denom = np.sqrt(var_p * var_l)
valid = (n_g >= 3) & (denom > 0)
corr = np.where(valid, cov / np.where(denom == 0, 1, denom), 0.0)
return float(corr[valid].mean()) if valid.any() else 0.0
def rankic_feval(preds, dataset): def rankic_feval(preds, dataset):
@@ -1,3 +1,13 @@
from .kelly_dropout import FractionalKellyDropoutStrategy # noqa: F401
from .optimal_stop import OptimalStopControl # noqa: F401 from .optimal_stop import OptimalStopControl # noqa: F401
from .regime_gate import RegimeGateDropoutStrategy # noqa: F401
from .top_bottom import TopBottomDropoutStrategy # noqa: F401
from .weekly_rebalance import WeeklyRebalanceDropoutStrategy # noqa: F401
__all__ = ["OptimalStopControl"] __all__ = [
"OptimalStopControl",
"FractionalKellyDropoutStrategy",
"WeeklyRebalanceDropoutStrategy",
"TopBottomDropoutStrategy",
"RegimeGateDropoutStrategy",
]
@@ -0,0 +1,201 @@
"""Fractional-Kelly dropout strategy for cross-sectional signals.
Sizing rule variant of ``qlib.contrib.strategy.signal_strategy.TopkDropoutStrategy``:
the topk/n_drop SELECTION is identical to the reference, but the buy size is
proportional to the score MAGNITUDE (edge) instead of equal-weight, capped at a
fraction ``cap_frac`` of the equal-weight notional so a single name cannot
over-concentrate the book.
``cap_frac`` is the fraction of the equal-weight per-name notional that a top
signal can deploy at most (e.g. 0.5 = at most half the equal-weight size).
Names whose score is below the median of the buy set get a proportionally
smaller slice; the residual stays in cash (that is the point of the rule:
throw away less edge per name, deploy less capital when conviction is low).
"""
from __future__ import annotations
from typing import List
import numpy as np
import pandas as pd
from qlib.backtest import Order
from qlib.backtest.decision import OrderDir, TradeDecisionWO
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
__all__ = ["FractionalKellyDropoutStrategy"]
DEFAULT_CAP_FRAC = 0.5
class FractionalKellyDropoutStrategy(TopkDropoutStrategy):
"""TopkDropout selection with score-magnitude (fractional-Kelly) sizing.
Parameters
----------
topk, n_drop, method_sell, method_buy, hold_thresh, only_tradable,
forbid_all_trade_at_limit : same as ``TopkDropoutStrategy``.
cap_frac : max buy notional as a fraction of the equal-weight notional.
"""
def __init__(self, *, topk, n_drop, cap_frac: float = DEFAULT_CAP_FRAC, **kwargs):
super().__init__(topk=topk, n_drop=n_drop, **kwargs)
self.cap_frac = cap_frac
def generate_trade_decision(self, execute_result=None):
import copy
trade_step = self.trade_calendar.get_trade_step()
trade_start_time, trade_end_time = self.trade_calendar.get_step_time(trade_step)
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(trade_step, shift=1)
pred_score = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
if isinstance(pred_score, pd.DataFrame):
pred_score = pred_score.iloc[:, 0]
if pred_score is None:
return TradeDecisionWO([], self)
if self.only_tradable:
def get_first_n(li, n, reverse=False):
cur_n = 0
res = []
for si in reversed(li) if reverse else li:
if self.trade_exchange.is_stock_tradable(
stock_id=si, start_time=trade_start_time, end_time=trade_end_time
):
res.append(si)
cur_n += 1
if cur_n >= n:
break
return res[::-1] if reverse else res
def get_last_n(li, n):
return get_first_n(li, n, reverse=True)
def filter_stock(li):
return [
si
for si in li
if self.trade_exchange.is_stock_tradable(
stock_id=si, start_time=trade_start_time, end_time=trade_end_time
)
]
else:
def get_first_n(li, n):
return list(li)[:n]
def get_last_n(li, n):
return list(li)[-n:]
def filter_stock(li):
return li
current_temp: "object" = copy.deepcopy(self.trade_position)
sell_order_list: List[Order] = []
buy_order_list: List[Order] = []
cash = current_temp.get_cash()
current_stock_list = current_temp.get_stock_list()
last = pred_score.reindex(current_stock_list).sort_values(ascending=False).index
if self.method_buy == "top":
today = get_first_n(
pred_score[~pred_score.index.isin(last)].sort_values(ascending=False).index,
self.n_drop + self.topk - len(last),
)
elif self.method_buy == "random":
topk_candi = get_first_n(pred_score.sort_values(ascending=False).index, self.topk)
candi = list(filter(lambda x: x not in last, topk_candi))
n = self.n_drop + self.topk - len(last)
try:
today = np.random.choice(candi, n, replace=False)
except ValueError:
today = candi
else:
raise NotImplementedError(f"This type of input is not supported")
comb = pred_score.reindex(last.union(pd.Index(today))).sort_values(ascending=False).index
if self.method_sell == "bottom":
sell = last[last.isin(get_last_n(comb, self.n_drop))]
elif self.method_sell == "random":
candi = filter_stock(last)
try:
sell = pd.Index(np.random.choice(candi, self.n_drop, replace=False) if len(last) else [])
except ValueError:
sell = candi
else:
raise NotImplementedError(f"This type of input is not supported")
buy = today[: len(sell) + self.topk - len(last)]
for code in current_stock_list:
if not self.trade_exchange.is_stock_tradable(
stock_id=code,
start_time=trade_start_time,
end_time=trade_end_time,
direction=None if self.forbid_all_trade_at_limit else OrderDir.SELL,
):
continue
if code in sell:
time_per_step = self.trade_calendar.get_freq()
if current_temp.get_stock_count(code, bar=time_per_step) < self.hold_thresh:
continue
sell_amount = current_temp.get_stock_amount(code=code)
sell_order = Order(
stock_id=code,
amount=sell_amount,
start_time=trade_start_time,
end_time=trade_end_time,
direction=Order.SELL,
)
if self.trade_exchange.check_order(sell_order):
sell_order_list.append(sell_order)
trade_val, trade_cost, trade_price = self.trade_exchange.deal_order(
sell_order, position=current_temp
)
cash += trade_val - trade_cost
if len(buy) == 0:
return TradeDecisionWO(sell_order_list, self)
# ---- fractional-Kelly sizing --------------------------------------
# equal-weight notional (reference baseline)
eq_notional = cash * self.risk_degree / len(buy)
buy_scores = pred_score.reindex(buy).astype(float)
lo, hi = buy_scores.min(), buy_scores.max()
if hi == lo:
w = pd.Series(1.0, index=buy_scores.index)
else:
w = (buy_scores - lo) / (hi - lo) # [0,1] edge magnitude
w = w.clip(lower=0.0)
w_max = w.max()
w = w / w_max if w_max > 0 else w # max == 1.0
for code in buy:
if not self.trade_exchange.is_stock_tradable(
stock_id=code,
start_time=trade_start_time,
end_time=trade_end_time,
direction=None if self.forbid_all_trade_at_limit else OrderDir.BUY,
):
continue
buy_price = self.trade_exchange.get_deal_price(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time, direction=OrderDir.BUY
)
notional = eq_notional * min(self.cap_frac, float(w.get(code, 0.0)))
buy_amount = notional / buy_price
factor = self.trade_exchange.get_factor(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time
)
buy_amount = self.trade_exchange.round_amount_by_trade_unit(buy_amount, factor)
buy_order = Order(
stock_id=code,
amount=buy_amount,
start_time=trade_start_time,
end_time=trade_end_time,
direction=Order.BUY,
)
buy_order_list.append(buy_order)
return TradeDecisionWO(sell_order_list + buy_order_list, self)
@@ -0,0 +1,231 @@
"""HMM-regime overlay TopkDropout strategy.
Regime-gate overlay on ``qlib.contrib.strategy.signal_strategy.TopkDropoutStrategy``:
selection and sizing are identical to the reference, but a name is only BOUGHT
(entry gate) when its per-symbol HMM regime posterior ``sp_hmm_p_regime1`` on
the signal date is >= ``regime_threshold``; otherwise it is held in cash instead
of being opened.
The regime posterior is read from the lake feature provider on the fly via
``qlib.data.D.features`` (field ``$sp_hmm_p_regime1``) for the signal window, so
no regime column needs to enter the model's ``feature_fields`` — the gate is a
pure overlay (book ch.01: regime flags regressed as model features, survived
only as an overlay). The HMM itself was fit with ``fit_end=<train end>`` when
the lake features were backfilled, so there is no lookahead.
Names already held are NOT force-sold when the regime turns unfavourable
(entry gate only, matching the queue-10 design).
"""
from __future__ import annotations
from typing import List
import numpy as np
import pandas as pd
from qlib.backtest import Order
from qlib.backtest.decision import OrderDir, TradeDecisionWO
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
try:
from qlib.data import D
except ImportError: # pragma: no cover - qlib always present in this stack
D = None
__all__ = ["RegimeGateDropoutStrategy"]
DEFAULT_REGIME_THRESHOLD = 0.5
REGIME_FIELD = "$sp_hmm_p_regime1"
class RegimeGateDropoutStrategy(TopkDropoutStrategy):
"""TopkDropout with an HMM-regime entry gate on buy candidates.
Parameters
----------
topk, n_drop, method_sell, method_buy, hold_thresh, only_tradable,
forbid_all_trade_at_limit : same as ``TopkDropoutStrategy``.
regime_threshold : minimum ``sp_hmm_p_regime1`` posterior required to open a
new position (default 0.5).
"""
def __init__(self, *, topk, n_drop, regime_threshold: float = DEFAULT_REGIME_THRESHOLD, **kwargs):
super().__init__(topk=topk, n_drop=n_drop, **kwargs)
self.regime_threshold = regime_threshold
def _regime_for(self, codes, pred_start, pred_end) -> pd.Series:
"""Return {code: sp_hmm_p_regime1} for the signal window (last day)."""
if D is None:
return pd.Series(dtype=float)
try:
df = D.features(list(codes), [REGIME_FIELD], start_time=pred_start, end_time=pred_end, freq="day")
except Exception: # noqa: BLE001 - a regime read failure should gate open, not crash
return pd.Series(dtype=float)
if df is None or len(df) == 0:
return pd.Series(dtype=float)
# df index is MultiIndex (datetime, instrument); take the last day's values
df = df.reset_index()
ts_col = "datetime" if "datetime" in df.columns else df.columns[0]
sym_col = "instrument" if "instrument" in df.columns else df.columns[1]
last_ts = df[ts_col].max()
last = df[df[ts_col] == last_ts]
out = {}
for _, row in last.iterrows():
sym = str(row[sym_col]).split("/")[-1].upper()
val = row.iloc[-1]
out[sym] = float(val) if val == val else np.nan
return pd.Series(out)
def generate_trade_decision(self, execute_result=None):
import copy
trade_step = self.trade_calendar.get_trade_step()
trade_start_time, trade_end_time = self.trade_calendar.get_step_time(trade_step)
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(trade_step, shift=1)
pred_score = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
if isinstance(pred_score, pd.DataFrame):
pred_score = pred_score.iloc[:, 0]
if pred_score is None:
return TradeDecisionWO([], self)
if self.only_tradable:
def get_first_n(li, n, reverse=False):
cur_n = 0
res = []
for si in reversed(li) if reverse else li:
if self.trade_exchange.is_stock_tradable(
stock_id=si, start_time=trade_start_time, end_time=trade_end_time
):
res.append(si)
cur_n += 1
if cur_n >= n:
break
return res[::-1] if reverse else res
def get_last_n(li, n):
return get_first_n(li, n, reverse=True)
def filter_stock(li):
return [
si
for si in li
if self.trade_exchange.is_stock_tradable(
stock_id=si, start_time=trade_start_time, end_time=trade_end_time
)
]
else:
def get_first_n(li, n):
return list(li)[:n]
def get_last_n(li, n):
return list(li)[-n:]
def filter_stock(li):
return li
current_temp: "object" = copy.deepcopy(self.trade_position)
sell_order_list: List[Order] = []
buy_order_list: List[Order] = []
cash = current_temp.get_cash()
current_stock_list = current_temp.get_stock_list()
last = pred_score.reindex(current_stock_list).sort_values(ascending=False).index
if self.method_buy == "top":
today = get_first_n(
pred_score[~pred_score.index.isin(last)].sort_values(ascending=False).index,
self.n_drop + self.topk - len(last),
)
elif self.method_buy == "random":
topk_candi = get_first_n(pred_score.sort_values(ascending=False).index, self.topk)
candi = list(filter(lambda x: x not in last, topk_candi))
n = self.n_drop + self.topk - len(last)
try:
today = np.random.choice(candi, n, replace=False)
except ValueError:
today = candi
else:
raise NotImplementedError(f"This type of input is not supported")
comb = pred_score.reindex(last.union(pd.Index(today))).sort_values(ascending=False).index
if self.method_sell == "bottom":
sell = last[last.isin(get_last_n(comb, self.n_drop))]
elif self.method_sell == "random":
candi = filter_stock(last)
try:
sell = pd.Index(np.random.choice(candi, self.n_drop, replace=False) if len(last) else [])
except ValueError:
sell = candi
else:
raise NotImplementedError(f"This type of input is not supported")
buy = today[: len(sell) + self.topk - len(last)]
# ---- regime gate -----------------------------------------------------
if buy:
regime = self._regime_for(buy, pred_start_time, pred_end_time)
gated = [c for c in buy if regime.get(c, np.nan) >= self.regime_threshold]
else:
gated = []
for code in current_stock_list:
if not self.trade_exchange.is_stock_tradable(
stock_id=code,
start_time=trade_start_time,
end_time=trade_end_time,
direction=None if self.forbid_all_trade_at_limit else OrderDir.SELL,
):
continue
if code in sell:
time_per_step = self.trade_calendar.get_freq()
if current_temp.get_stock_count(code, bar=time_per_step) < self.hold_thresh:
continue
sell_amount = current_temp.get_stock_amount(code=code)
sell_order = Order(
stock_id=code,
amount=sell_amount,
start_time=trade_start_time,
end_time=trade_end_time,
direction=Order.SELL,
)
if self.trade_exchange.check_order(sell_order):
sell_order_list.append(sell_order)
trade_val, trade_cost, trade_price = self.trade_exchange.deal_order(
sell_order, position=current_temp
)
cash += trade_val - trade_cost
if len(gated) == 0:
return TradeDecisionWO(sell_order_list, self)
value = cash * self.risk_degree / len(gated)
for code in gated:
if not self.trade_exchange.is_stock_tradable(
stock_id=code,
start_time=trade_start_time,
end_time=trade_end_time,
direction=None if self.forbid_all_trade_at_limit else OrderDir.BUY,
):
continue
buy_price = self.trade_exchange.get_deal_price(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time, direction=OrderDir.BUY
)
buy_amount = value / buy_price
factor = self.trade_exchange.get_factor(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time
)
buy_amount = self.trade_exchange.round_amount_by_trade_unit(buy_amount, factor)
buy_order = Order(
stock_id=code,
amount=buy_amount,
start_time=trade_start_time,
end_time=trade_end_time,
direction=Order.BUY,
)
buy_order_list.append(buy_order)
return TradeDecisionWO(sell_order_list + buy_order_list, self)
@@ -0,0 +1,169 @@
"""Market-neutral top/bottom long-short strategy for cross-sectional signals.
Captures the cross-sectional long-short spread net of costs: buys the top-ranked
``topk`` names and shorts the bottom-ranked ``topk`` names, equal-weight per
side, sized to ``risk_degree`` of total value per side. Rebalances daily to the
current rank (dropout-free: the book converges to the latest top/bottom sets).
The long and short legs use equal notional per side (gross exposure ~2x
``risk_degree`` of NAV, i.e. approximately market neutral before transaction
costs). Benchmark neutrality (SPY beta ~ 0) is the secondary sanity metric.
"""
from __future__ import annotations
from typing import List
import copy
import pandas as pd
from qlib.backtest import Order
from qlib.backtest.decision import OrderDir, TradeDecisionWO
from qlib.contrib.strategy.signal_strategy import BaseSignalStrategy
__all__ = ["TopBottomDropoutStrategy"]
DEFAULT_SHORT_LEG = True
DEFAULT_REBALANCE_DAILY = True
class TopBottomDropoutStrategy(BaseSignalStrategy):
"""Long top-k / short bottom-k equal-weight market-neutral book.
Parameters
----------
topk : number of names on each side (long top-k and short bottom-k).
short_leg : whether to open the short side (if False, long-only topk).
rebalance_daily : if True rebalance to current rank every day; else keep
positions and only refresh on score changes (dropout-style).
risk_degree : fraction of total value deployed per side.
"""
def __init__(
self,
*,
topk: int = 10,
short_leg: bool = DEFAULT_SHORT_LEG,
rebalance_daily: bool = DEFAULT_REBALANCE_DAILY,
**kwargs,
):
super().__init__(**kwargs)
self.topk = topk
self.short_leg = short_leg
self.rebalance_daily = rebalance_daily
self._prev_longs = set()
self._prev_shorts = set()
def generate_trade_decision(self, execute_result=None):
trade_step = self.trade_calendar.get_trade_step()
trade_start_time, trade_end_time = self.trade_calendar.get_step_time(trade_step)
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(trade_step, shift=1)
pred_score = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
if isinstance(pred_score, pd.DataFrame):
pred_score = pred_score.iloc[:, 0]
if pred_score is None or len(pred_score) == 0:
return TradeDecisionWO([], self)
# rank all names; topk longs and topk shorts
ranked = pred_score.sort_values(ascending=False)
longs = list(ranked.index[: self.topk])
shorts = list(ranked.index[-self.topk :]) if self.short_leg else []
current_temp: "object" = copy.deepcopy(self.trade_position)
current_codes = set(current_temp.get_stock_list())
holdings = {c: current_temp for c in current_codes if abs(current_temp.get_stock_amount(c)) > 1e-6}
sell_orders: List[Order] = []
buy_orders: List[Order] = []
def _tradable(code, direction):
try:
return self.trade_exchange.is_stock_tradable(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time, direction=direction
)
except TypeError:
return self.trade_exchange.is_stock_tradable(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time
)
# determine target set (long/short)
target_longs = set(longs)
target_shorts = set(shorts)
# close positions not in the target book
for code in list(holdings):
if code in target_longs or code in target_shorts:
continue
amt = abs(current_temp.get_stock_amount(code))
o = Order(
stock_id=code,
amount=amt,
start_time=trade_start_time,
end_time=trade_end_time,
direction=Order.SELL if code in target_longs else Order.SELL,
)
if self.trade_exchange.check_order(o):
sell_orders.append(o)
self.trade_exchange.deal_order(o, position=current_temp)
# equal-weight notional per side
total_value = current_temp.get_cash()
for code, pos in holdings.items():
if code in target_longs or code in target_shorts:
mark = self.trade_exchange.get_deal_price(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time, direction=Order.SELL
)
if mark is not None and mark == mark:
total_value += abs(current_temp.get_stock_amount(code)) * mark
side_notional = total_value * self.risk_degree / max(1, self.topk)
for code in longs:
if code in holdings and abs(current_temp.get_stock_amount(code)) > 1e-6:
continue
px = self.trade_exchange.get_deal_price(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time, direction=Order.BUY
)
if px is None or px != px or px <= 0:
continue
amount = side_notional / px
factor = self.trade_exchange.get_factor(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time
)
amount = self.trade_exchange.round_amount_by_trade_unit(amount, factor)
o = Order(
stock_id=code,
amount=amount,
start_time=trade_start_time,
end_time=trade_end_time,
direction=Order.BUY,
)
if self.trade_exchange.check_order(o):
buy_orders.append(o)
if self.short_leg:
for code in shorts:
if code in holdings and abs(current_temp.get_stock_amount(code)) > 1e-6:
continue
px = self.trade_exchange.get_deal_price(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time, direction=Order.SELL
)
if px is None or px != px or px <= 0:
continue
amount = side_notional / px
factor = self.trade_exchange.get_factor(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time
)
amount = self.trade_exchange.round_amount_by_trade_unit(amount, factor)
o = Order(
stock_id=code,
amount=amount,
start_time=trade_start_time,
end_time=trade_end_time,
direction=Order.SELL,
)
if self.trade_exchange.check_order(o):
sell_orders.append(o)
return TradeDecisionWO(sell_orders + buy_orders, self)
@@ -0,0 +1,202 @@
"""Weekly-rebalance TopkDropout strategy.
Turnover-reduction variant of ``qlib.contrib.strategy.signal_strategy.TopkDropoutStrategy``:
the topk/n_drop selection and sizing are identical to the reference, but the
target book is recomputed only on the first trading day of each ISO week; on the
other days the strategy issues NO orders (holds the book untouched).
The weekly cadence is derived from the qlib trade calendar: a rebalance happens
when the current trade step's date belongs to a different ISO ``(year, week)``
than the previous trade step. ``hold_band_pct`` (default 0) optionally skips
tiny rebalances: when a name's existing position differs from the new target by
less than this fraction, no order is generated for it.
"""
from __future__ import annotations
from typing import List
import numpy as np
import pandas as pd
from qlib.backtest import Order
from qlib.backtest.decision import OrderDir, TradeDecisionWO
from qlib.contrib.strategy.signal_strategy import TopkDropoutStrategy
__all__ = ["WeeklyRebalanceDropoutStrategy"]
DEFAULT_HOLD_BAND_PCT = 0.0
class WeeklyRebalanceDropoutStrategy(TopkDropoutStrategy):
"""TopkDropout rebalanced once per ISO week; holds otherwise.
Parameters
----------
topk, n_drop, method_sell, method_buy, hold_thresh, only_tradable,
forbid_all_trade_at_limit : same as ``TopkDropoutStrategy``.
hold_band_pct : skip order for a name whose deviation from target weight is
below this fraction of the target (no-trade buffer band).
"""
def __init__(self, *, topk, n_drop, hold_band_pct: float = DEFAULT_HOLD_BAND_PCT, **kwargs):
super().__init__(topk=topk, n_drop=n_drop, **kwargs)
self.hold_band_pct = hold_band_pct
@staticmethod
def _iso_week(ts) -> tuple:
return (ts.year, ts.week)
def generate_trade_decision(self, execute_result=None):
import copy
trade_step = self.trade_calendar.get_trade_step()
trade_start_time, trade_end_time = self.trade_calendar.get_step_time(trade_step)
cur_week = self._iso_week(trade_start_time)
prev_week = getattr(self, "_last_week", None)
self._last_week = cur_week
if prev_week is not None and prev_week == cur_week:
# not the first trading day of this ISO week -> hold
return TradeDecisionWO([], self)
pred_start_time, pred_end_time = self.trade_calendar.get_step_time(trade_step, shift=1)
pred_score = self.signal.get_signal(start_time=pred_start_time, end_time=pred_end_time)
if isinstance(pred_score, pd.DataFrame):
pred_score = pred_score.iloc[:, 0]
if pred_score is None:
return TradeDecisionWO([], self)
if self.only_tradable:
def get_first_n(li, n, reverse=False):
cur_n = 0
res = []
for si in reversed(li) if reverse else li:
if self.trade_exchange.is_stock_tradable(
stock_id=si, start_time=trade_start_time, end_time=trade_end_time
):
res.append(si)
cur_n += 1
if cur_n >= n:
break
return res[::-1] if reverse else res
def get_last_n(li, n):
return get_first_n(li, n, reverse=True)
def filter_stock(li):
return [
si
for si in li
if self.trade_exchange.is_stock_tradable(
stock_id=si, start_time=trade_start_time, end_time=trade_end_time
)
]
else:
def get_first_n(li, n):
return list(li)[:n]
def get_last_n(li, n):
return list(li)[-n:]
def filter_stock(li):
return li
current_temp: "object" = copy.deepcopy(self.trade_position)
sell_order_list: List[Order] = []
buy_order_list: List[Order] = []
cash = current_temp.get_cash()
current_stock_list = current_temp.get_stock_list()
last = pred_score.reindex(current_stock_list).sort_values(ascending=False).index
if self.method_buy == "top":
today = get_first_n(
pred_score[~pred_score.index.isin(last)].sort_values(ascending=False).index,
self.n_drop + self.topk - len(last),
)
elif self.method_buy == "random":
topk_candi = get_first_n(pred_score.sort_values(ascending=False).index, self.topk)
candi = list(filter(lambda x: x not in last, topk_candi))
n = self.n_drop + self.topk - len(last)
try:
today = np.random.choice(candi, n, replace=False)
except ValueError:
today = candi
else:
raise NotImplementedError(f"This type of input is not supported")
comb = pred_score.reindex(last.union(pd.Index(today))).sort_values(ascending=False).index
if self.method_sell == "bottom":
sell = last[last.isin(get_last_n(comb, self.n_drop))]
elif self.method_sell == "random":
candi = filter_stock(last)
try:
sell = pd.Index(np.random.choice(candi, self.n_drop, replace=False) if len(last) else [])
except ValueError:
sell = candi
else:
raise NotImplementedError(f"This type of input is not supported")
buy = today[: len(sell) + self.topk - len(last)]
for code in current_stock_list:
if not self.trade_exchange.is_stock_tradable(
stock_id=code,
start_time=trade_start_time,
end_time=trade_end_time,
direction=None if self.forbid_all_trade_at_limit else OrderDir.SELL,
):
continue
if code in sell:
time_per_step = self.trade_calendar.get_freq()
if current_temp.get_stock_count(code, bar=time_per_step) < self.hold_thresh:
continue
sell_amount = current_temp.get_stock_amount(code=code)
sell_order = Order(
stock_id=code,
amount=sell_amount,
start_time=trade_start_time,
end_time=trade_end_time,
direction=Order.SELL,
)
if self.trade_exchange.check_order(sell_order):
sell_order_list.append(sell_order)
trade_val, trade_cost, trade_price = self.trade_exchange.deal_order(
sell_order, position=current_temp
)
cash += trade_val - trade_cost
if len(buy) == 0:
return TradeDecisionWO(sell_order_list, self)
value = cash * self.risk_degree / len(buy)
for code in buy:
if not self.trade_exchange.is_stock_tradable(
stock_id=code,
start_time=trade_start_time,
end_time=trade_end_time,
direction=None if self.forbid_all_trade_at_limit else OrderDir.BUY,
):
continue
buy_price = self.trade_exchange.get_deal_price(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time, direction=OrderDir.BUY
)
buy_amount = value / buy_price
factor = self.trade_exchange.get_factor(
stock_id=code, start_time=trade_start_time, end_time=trade_end_time
)
buy_amount = self.trade_exchange.round_amount_by_trade_unit(buy_amount, factor)
buy_order = Order(
stock_id=code,
amount=buy_amount,
start_time=trade_start_time,
end_time=trade_end_time,
direction=Order.BUY,
)
buy_order_list.append(buy_order)
return TradeDecisionWO(sell_order_list + buy_order_list, self)
+29 -2
View File
@@ -6,10 +6,11 @@ The lake is a hive-partitioned parquet store (see ``tac-engine/skills/tradeac-la
├── market=US/ ├── market=US/
│ └── timeframe=1d/ │ └── timeframe=1d/
│ └── symbol=AAPL.parquet # OHLCV bars: t, date, o, h, l, c, v, n, vw │ └── symbol=AAPL.parquet # OHLCV bars: t, date, o, h, l, c, v, n, vw
├── features/ # ta-lib indicators, wide format ├── features/ # indicators, wide format, family tier
│ └── market=US/ │ └── market=US/
│ └── timeframe=1d/ │ └── timeframe=1d/
│ └── symbol=AAPL.parquet # t, sma_5, sma_20, rsi_14, ... │ ├── family=ta/symbol=AAPL.parquet # t, sma_5, sma_20, rsi_14, ...
│ └── family=sp/symbol=AAPL.parquet # t, sp_ou_*, sp_hmm_*, ...
├── calendar.parquet # trading days per market ├── calendar.parquet # trading days per market
├── coverage.parquet # per (market,timeframe,symbol) loaded windows ├── coverage.parquet # per (market,timeframe,symbol) loaded windows
└── symbols.parquet # asset master └── symbols.parquet # asset master
@@ -107,8 +108,34 @@ class LakeConfig:
return self.lake_root / "features" / f"market={self.market}" / f"timeframe={timeframe}" return self.lake_root / "features" / f"market={self.market}" / f"timeframe={timeframe}"
def features_path(self, timeframe: str, symbol: str) -> Path: def features_path(self, timeframe: str, symbol: str) -> Path:
# Legacy flat path (no family tier). Prefer `load_features` which
# resolves the family=ta|sp partition layout.
return self.features_dir(timeframe) / f"symbol={str(symbol).upper()}.parquet" return self.features_dir(timeframe) / f"symbol={str(symbol).upper()}.parquet"
def load_features(self, timeframe: str, symbol: str) -> pd.DataFrame:
"""All feature columns for a symbol, merging the `family=ta` and
`family=sp` partitions by timestamp. Returns an empty frame when no
feature files exist (legacy flat layout falls back transparently)."""
sym = str(symbol).upper()
frames = []
for family in ("ta", "sp"):
p = self.features_dir(timeframe) / f"family={family}" / f"symbol={sym}.parquet"
if p.exists():
frames.append(pd.read_parquet(p))
if not frames:
flat = self.features_dir(timeframe) / f"symbol={sym}.parquet"
if flat.exists():
return pd.read_parquet(flat)
return pd.DataFrame()
if len(frames) == 1:
return frames[0]
merged = frames[0]
for extra in frames[1:]:
merged = merged.merge(extra, on="t", how="outer", suffixes=("", "_dup"))
for c in [c for c in merged.columns if c.endswith("_dup")]:
merged = merged.drop(columns=c)
return merged
def calendar_path(self) -> Path: def calendar_path(self) -> Path:
return self.lake_root / "calendar.parquet" return self.lake_root / "calendar.parquet"
+1 -2
View File
@@ -173,8 +173,7 @@ class LakeFeatureProvider(FeatureProvider):
def _load_feature_df(self, instrument: str, timeframe: str) -> pd.DataFrame: def _load_feature_df(self, instrument: str, timeframe: str) -> pd.DataFrame:
key = (instrument, timeframe) key = (instrument, timeframe)
if key not in self._feature_cache: if key not in self._feature_cache:
p = self.cfg.features_path(timeframe, instrument) self._feature_cache[key] = self.cfg.load_features(timeframe, instrument)
self._feature_cache[key] = pd.read_parquet(p) if p.exists() else pd.DataFrame()
return self._feature_cache[key] return self._feature_cache[key]
@staticmethod @staticmethod
+69
View File
@@ -0,0 +1,69 @@
# TradeAC Experiment Queue — Series 2 (Q12+)
**Purpose.** The next pre-registered batch of experiments, continuing Series 1
(Q01–Q11, exp 33–43, all executed and folded into `book/CLAIMS.md` /
`book/EVIDENCE.md`). Each entry targets a still-unproven `HYPOTHESIS` from the
book or an open question flagged in `CLAIMS.md`/`book/README.md`, and follows the
Series-1 discipline: one variable changed vs the exp-26 reference, acceptance
fixed BEFORE the run, sequential execution, trace-first, verify-then-close.
**Reference / control (MUST reproduce first).** exp 26 (`21afc6af…`, mlflow exp
25) is the campaign baseline; exp 39 (Q07, weekly rebalance) is the best
construction. Reference config is byte-reproduced in `workflows/exp26/` on the
`exp/26-…` branch and in this dir's `workflows/*.yaml`.
| Config element | exp-26 reference value |
|---|---|
| Universe | 50-ETF panel (`UNIVERSE` below) |
| Features | compact stochastic 25-field set (no ou/hmm/moments/garch) |
| Label | `Ref($close,-6)/Ref($close,-1)-1` (5d) |
| Model | `RankICEnsembleLGBModel`, seeds `42,7,2026,99,123`, lr 0.02, leaves 31, 3000 rounds, ES 200 |
| Segments | train 2016-01-04..2025-09-01 / valid 2025-09-03..2026-01-03 / test 2026-01-04..2026-08-10 |
| Strategy | TopkDropout, topk 10, n_drop 1, risk_degree 0.95 |
| Costs | open 0.0005 / close 0.0015 / min $5, deal $close, SPY benchmark, $1M |
**Reference metrics to beat (EVIDENCE#015):** net_ann +2.13%, net_IR 0.21, gross
+7.02%, maxDD −7.69%, RankIC 0.0663, RankICIR 0.2545, L/S Sharpe 4.54. Weekly
(Q07, EVIDENCE#028): net +12.51%, IR 1.24, maxDD −4.13%, ~1.1pp cost drag.
## The queue (ordered by value × feasibility)
| ID | Title / hypothesis | Change vs reference (ONE var) | Acceptance | Config | Ready? |
|----|--------------------|-------------------------------|------------|--------|--------|
| Q12 | **22d label + weekly recompute** — the untested combo: Q05's label edge (IC 0.097, RankIC 0.117) with Q07's cost relief | label → 22d AND strategy → weekly (two coupled, explicitly pre-registered) | net_IR > 0.5, net_ann > +5%, cost drag ≤ 2pp | `workflows/q12_label22d_weekly.yaml` | ✅ |
| Q13 | **Weekly rebalance reproduction on a 2nd window** — Q07 was a single OOS window; reproduce on test 2025-01-02..2025-12-31 before promoting to a live round | segments only (shifted) | net_IR > 0.21, net_ann > +2.13% on the new window | `workflows/q13_weekly_second_window.yaml` | ✅ |
| Q14 | **Out-of-universe validation** — compact stochastic set generalizes off the 50-ETF panel to a single-stock universe | universe → 30 liquid single names | RankIC > 0.03, ICIR > 0.15, net IR > 0 on stocks | `workflows/q14_out_of_universe.yaml` | ⚠️ needs stock-lake backfill (see design) |
| Q15 | **5-seed vs single-model clean A/B** — seed-count claim (exp 12 idea, re-validated exp 22–24, never a clean A/B) | seeds → 1 (`2026`) | single-model RankIC/IR < 5-seed ref; net_IR ≥ 0.21 acceptable if ≥ single | `workflows/q15_single_seed.yaml` | ✅ |
| Q16 | **HMM family added as features** — settles "dropping model-specific (ou,hmm) improves signal" (exp 25 tested OU; hmm-as-feature untested) | features += `sp_hmm_p_regime1,sp_hmm_state` | no improvement: RankIC ≤ 0.0663, net_IR ≤ 0.21 | `workflows/q16_hmm_features.yaml` | ✅ |
| Q17 | **Realized-moments family added** — settles "moment/volatility families regress" (exp 11 idea, never clean A/B) | features += `sp_rskew_5,sp_rskew_22,sp_rkurt_5,sp_rkurt_22,sp_dsv_5,sp_dsv_22` | no improvement: RankIC ≤ 0.0663, net_IR ≤ 0.21 | `workflows/q17_moments_features.yaml` | ✅ |
| Q18 | **OptimalStopControl clean re-test** — exp 13/14 claim (TopkDropout > stop-control) never re-tested post-reset | strategy → `OptimalStopControl` (exp-13 params) | TopkDropout net_IR ≥ stop-control net_IR; document cost drag | `workflows/q18_optstop.yaml` | ✅ (module verified in venv) |
| Q19 | **Martingale / variance-ratio study close-out** — exp 19 never closed; VR<1 at 5–20d on clean lake | ad-hoc script (no qrun) | VR stats + drift decomposition on 50-ETF panel | `designs/q19_martingale_vr.md` | ✅ script |
| Q20 | **Effective independent names (≈4)** — eigenvalue analysis on clean-lake covariance | ad-hoc script | eigenvalue spectrum + effective-rank count | `designs/q20_effective_names.md` | ✅ script |
### Deferred (methodology / infra, P3)
- Purged / walk-forward CV (was queue's old Q12) — methodology, not an alpha lever.
- PSI-based drift-aware retraining cadence — needs a drift-gate module + a retrain decision rule.
- No-trade buffer band / notional-vs-qty sizing — siblings of Q12/Q13; queue only if weekly reproduces.
- Macro/drift overlays (SPY>200d regime gate, momentum tilt) — needs new data pipeline.
## Execution protocol (per queued run)
1. **Validate the lake first** (`validate_lake_dataset` + `rd_status`) — clean-lake lesson: silent NaN-drops and hollow coverage invalidate a run. Q14 additionally requires backfilling the single-stock universe (bars + sp/ta features, full range, explicit `start`/`end`).
2. **Trace before running** (`rd_trace_start` with the hypothesis as `rational`, fresh `experiment_name`, `evolved_from=auto`).
3. **Run** `rd_run_workflow config_path=<abs path to the queue YAML> experiment_name=<fresh name>` — `wait=false`, poll `rd_exp_get_run` until `FINISHED`.
4. **Verify against acceptance** via `rd_exp_result` (headline + backtest risk).
5. **Finish the trace** (`rd_trace_finish` with `metrics` + `evaluation`), snapshot any changed contrib modules.
6. **Report to the book** — PROVE/REFUTE → update `book/CLAIMS.md` + `book/EVIDENCE.md`.
Sequential execution only (concurrent runs hang — chat-ideas.md ops lesson). Any
custom strategy/module changed here must be copied into the venv site-packages
snapshot before `rd_run_workflow` can import it (see `/app/AGENTS.md`). As of
2026-08-20 `WeeklyRebalanceDropoutStrategy` and `OptimalStopControl` are verified
in sync with the venv snapshot; the lake already persists the `sp_hmm_*` and
`sp_moments` families on the 50-ETF panel.
## Provenance
Mined 2026-08-20 from `book/CLAIMS.md`, `book/EVIDENCE.md`, `book/README.md`,
`book/references/chat-ideas.md`, and Series-1 `queue/` (Q01–Q11, executed exp
33–43). Reference numbers are post-clean-lake (exp 21+).
+26
View File
@@ -0,0 +1,26 @@
# QUEUE-19 — Martingale / variance-ratio study close-out (no qrun)
**Status:** QUEUED · **Priority:** P2 · **Effort:** ad-hoc script under `book/data/`
## Hypothesis (settle)
Assets are submartingales long-horizon / mean-reverting short-horizon
(`VR < 1` at 5–20d). CLAIMS.md marks this HYPOTHESIS (chat-derived martingale
study; exp 19 was opened but never closed). It is a market-structure claim, not a
trading claim — settle it with a clean-lake script, then close exp 19 or open a
scripted EVIDENCE entry.
## Method (persist everything under `book/data/evidence/q19-vr/`)
1. Load the 50-ETF panel 1d bars from the lake for 2015-01-01..2026-08-19.
2. Compute the Lo–MacKinlay variance ratio at horizons 5 / 10 / 20d per symbol,
with heteroskedasticity-robust z-stats.
3. Report: per-horizon VR distribution, fraction of symbols with VR < 1 and the
z-significance, pooled drift vs daily variance (submartingale check).
4. Cross-check the pooled `sp_trend_slope_5` regression beta claim (β ≈ −0.53,
t ≈ −24) on the clean lake.
5. Write `VR_stats.csv` + a one-page summary into the evidence dir.
## Acceptance
- VR < 1 at 5–20d for a material fraction of the panel with |z| > 2 → supports
the mean-reversion HYPOTHESIS; else mark REFUTED or REFERENCED.
- The result updates CLAIMS.md's "Assets are submartingales…" row and closes the
exp-19 open thread.
+22
View File
@@ -0,0 +1,22 @@
# QUEUE-20 — Effective independent names in the 50-ETF book (no qrun)
**Status:** QUEUED · **Priority:** P2 · **Effort:** ad-hoc script under `book/data/`
## Hypothesis (settle)
The 50-ETF book has only ~4 effective independent names (CLAIMS.md HYPOTHESIS,
chat-derived eigenvalue analysis, pre-reset). This is a concentration/diversification
claim with direct sizing relevance; verify it on the clean lake.
## Method (persist everything under `book/data/evidence/q20-effective-names/`)
1. Load the 50-ETF panel 1d returns from the lake for the test window 2026-01-04..2026-08-10.
2. Standardize returns; compute the correlation matrix and its eigendecomposition.
3. Count eigenvalues above the Marchenko–Pastur bound (N=50, T≈150) and report the
cumulative-variance share of the top k components.
4. Effective-rank measures: participation ratio `(Σλ)² / Σλ²` and cumulative 80%
variance count.
5. Write `eigenanalysis.csv` + a one-page summary.
## Acceptance
- If effective rank ≈ 4 (top-4 explain ~80%+ variance), the concentration claim is
PROVEN and feeds chapter 08 sizing guidance (why topk 10→20 adds no breadth).
- If effective rank is much larger, mark the claim REFUTED.
+105
View File
@@ -0,0 +1,105 @@
# QUEUE-12 — Long-horizon label (22d) + weekly recompute construction.
# Untested combination from book/CLAIMS.md open questions: Q05 (exp 37) proved the
# 22d label has the strongest signal (IC 0.097, RankIC 0.117) but daily turnover
# killed the book (net -4.60%); Q07 (exp 39) proved weekly recompute is the cost
# lever (net +12.51%). Hypothesis: pairing them monetizes the label edge.
# Change vs exp-26 reference: label 5d -> 22d AND strategy -> WeeklyRebalanceDropoutStrategy.
# Acceptance: net_IR > 0.5, net_ann > +5%, cost drag <= 2pp.
# Run: rd_run_workflow config_path=<repo>/experiments/queue/workflows/q12_label22d_weekly.yaml \
# experiment_name=tac-rd-q12-label22d-weekly
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-q12-label22d-weekly" }
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-10
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-23)/Ref($close,-1)-1"
feature_fields: "{{ FEATURES }}"
infer_processors:
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: ProcessInf, kwargs: {} }
- { class: CSRankNorm, kwargs: {} }
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: Fillna, kwargs: {} }
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: WeeklyRebalanceDropoutStrategy
module_path: tac_qlib.contrib.strategy.weekly_rebalance
kwargs: { signal: "<PRED>", topk: 10, n_drop: 1, only_tradable: true, risk_degree: 0.95 }
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -0,0 +1,106 @@
# QUEUE-13 — Weekly rebalance reproduction on a second OOS window.
# Q07 (exp 39) proved weekly recompute on test 2026-01-04..2026-08-10 (net +12.51%,
# IR 1.24) but that is a single OOS window. Before promoting the weekly construction
# to a live round, reproduce it on a disjoint window: test 2025-01-02..2025-12-31
# with train/valid shifted to end 2024.
# Change vs exp-26 reference: segments shifted only (train ends 2024-08, test = 2025);
# strategy is the SAME weekly recompute as exp 39. Label stays 5d.
# Acceptance: net_IR > 0.21 AND net_ann > +2.13% on the 2025 window.
# Run: rd_run_workflow config_path=<repo>/experiments/queue/workflows/q13_weekly_second_window.yaml \
# experiment_name=tac-rd-q13-weekly-second-window
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-q13-weekly-second-window" }
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2025-12-31
fit_start_time: 2016-01-04
fit_end_time: 2024-08-30
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "{{ FEATURES }}"
infer_processors:
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2024-08-30" } }
- { class: ProcessInf, kwargs: {} }
- { class: CSRankNorm, kwargs: {} }
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2024-08-30" } }
- { class: Fillna, kwargs: {} }
segments:
train: [2016-01-04, 2024-08-30]
valid: [2024-09-03, 2024-12-31]
test: [2025-01-02, 2025-12-31]
record:
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: WeeklyRebalanceDropoutStrategy
module_path: tac_qlib.contrib.strategy.weekly_rebalance
kwargs: { signal: "<PRED>", topk: 10, n_drop: 1, only_tradable: true, risk_degree: 0.95 }
backtest:
start_time: 2025-01-02
end_time: 2025-12-31
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
+107
View File
@@ -0,0 +1,107 @@
# QUEUE-14 — Out-of-universe validation: compact stochastic set on single-stock names.
# The 50-ETF panel results (compact feature set, RankIC 0.0663) are panel-specific;
# book/CLAIMS.md marks "generalizes to other universes" HYPOTHESIS - TODO(evidence-needed).
# Change vs exp-26 reference: universe -> 30 liquid US single-stock names.
# PREREQUISITE: backfill lake bars + sp/ta features for these symbols (full range,
# explicit start/end) — the stock panel currently has only ~180d of data (2025-12-01+).
# Backfill: get_lake_bars symbols=... start=2000-01-03 then
# get_lake_sp symbol=<s> start=2000-01-03 end=<today> fit_end=<train-end> persist=true
# Acceptance: RankIC > 0.03, ICIR > 0.15, net IR > 0 on the stock universe.
# Run: rd_run_workflow config_path=<repo>/experiments/queue/workflows/q14_out_of_universe.yaml \
# experiment_name=tac-rd-q14-out-of-universe
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "AAPL,MSFT,NVDA,AMZN,GOOGL,META,TSLA,AVGO,AMD,JPM,UNH,PG,JNJ,MA,V,WMT,DIS,HD,KO,PEP,BAC,XOM,MCD,ABBV,COST,CRM,NFLX,ORCL,IBM,T" %}
{%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-q14-out-of-universe" }
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-10
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "{{ FEATURES }}"
infer_processors:
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: ProcessInf, kwargs: {} }
- { class: CSRankNorm, kwargs: {} }
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: Fillna, kwargs: {} }
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs: { signal: "<PRED>", topk: 10, n_drop: 1, only_tradable: true, risk_degree: 0.95 }
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
@@ -1,52 +1,38 @@
# ----------------------------------------------------------------------------- # QUEUE-15 — 5-seed vs single-model clean A/B on the compact stochastic set.
# ABLATION B (generic-only): same panel/model as the baseline, but feature # CLAIMS.md HYPOTHESIS: "5-seed RankIC ensemble raises performance vs single model
# fields restricted to the model-free / generic stochastic-process families # on ablated set" — pre-clean-lake exp 12 idea, re-validated directionally by exp
# (jump,har,trend,hurst,signature). Drops the model-specific ou (OU/AR-1 # 22–24, never a clean A/B post-reset. Seed count is load-bearing (exp 28: 2<5).
# half-life) and hmm (2-state regime) families to test whether the generic # Change vs exp-26 reference: seeds "42,7,2026,99,123" -> single seed "2026".
# families alone dominate the rank dimension. # Acceptance: single-model RankIC < 0.0663, net_IR < 0.21 (ensemble beats single).
# # Run: rd_run_workflow config_path=<repo>/experiments/queue/workflows/q15_single_seed.yaml \
# Run: # experiment_name=tac-rd-q15-single-seed
# rd_run_workflow config_path=tac-qlib/workflows/ablate_generic_only_sp_fields.yaml \
# experiment_name=tac-rd-rank-ablate
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %} {%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %} {%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %} {%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init: qlib_init:
provider_uri: "{{ LAKE }}" provider_uri: "{{ LAKE }}"
region: us region: us
expression_cache: null expression_cache: null
dataset_cache: null dataset_cache: null
calendar_provider: calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US }
lake_root: "{{ LAKE }}"
market: US
instrument_provider: instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider: feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US }
lake_root: "{{ LAKE }}"
market: US
exp_manager: exp_manager:
class: MLflowExpManager class: MLflowExpManager
module_path: qlib.workflow.expm module_path: qlib.workflow.expm
kwargs: kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-q15-single-seed" }
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-rank-ablate"
task: task:
model: model:
class: RankICLGBModel class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_gbdt module_path: tac_qlib.contrib.model.rank_ensemble
kwargs: kwargs:
loss: mse loss: mse
learning_rate: 0.02 learning_rate: 0.02
@@ -61,7 +47,7 @@ task:
subsample_freq: 1 subsample_freq: 1
reg_alpha: 0.1 reg_alpha: 0.1
reg_lambda: 1.0 reg_lambda: 1.0
seed: 42 seeds: "2026"
dataset: dataset:
class: DatasetH class: DatasetH
@@ -74,38 +60,27 @@ task:
instruments: "{{ UNIVERSE }}" instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03 start_time: 2015-01-03
end_time: 2026-08-10 end_time: 2026-08-10
fit_start_time: 2015-01-03 fit_start_time: 2016-01-04
fit_end_time: 2025-09-01 fit_end_time: 2025-09-01
freq: day freq: day
lake_root: "{{ LAKE }}" lake_root: "{{ LAKE }}"
market: US market: US
label: "Ref($close,-6)/Ref($close,-1)-1" label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}" feature_fields: "{{ FEATURES }}"
infer_processors: infer_processors:
- class: DropAllNaN - { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
kwargs: {} - { class: ProcessInf, kwargs: {} }
- class: ProcessInf - { class: CSRankNorm, kwargs: {} }
kwargs: {} - { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- class: CSRankNorm - { class: Fillna, kwargs: {} }
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments: segments:
train: [2015-01-03, 2025-09-01] train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03] valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10] test: [2026-01-04, 2026-08-10]
record: record:
- class: SignalRecord - { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
module_path: qlib.workflow.record_temp - { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord - class: PortAnaRecord
module_path: qlib.workflow.record_temp module_path: qlib.workflow.record_temp
kwargs: kwargs:
@@ -113,12 +88,7 @@ task:
strategy: strategy:
class: TopkDropoutStrategy class: TopkDropoutStrategy
module_path: qlib.contrib.strategy module_path: qlib.contrib.strategy
kwargs: kwargs: { signal: "<PRED>", topk: 10, n_drop: 1, only_tradable: true, risk_degree: 0.95 }
signal: "<PRED>"
topk: 10
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest: backtest:
start_time: 2026-01-04 start_time: 2026-01-04
end_time: 2026-08-10 end_time: 2026-08-10
@@ -131,4 +101,4 @@ task:
open_cost: 0.0005 open_cost: 0.0005
close_cost: 0.0015 close_cost: 0.0015
min_cost: 5.0 min_cost: 5.0
risk_analysis_freq: 1d risk_analysis_freq: 1d
@@ -1,51 +1,39 @@
# ----------------------------------------------------------------------------- # QUEUE-16 — HMM family added as model features to the compact set.
# ABLATION A (baseline): LightGBM with RankIC early-stopping on the 50-ETF SP-5d # CLAIMS.md HYPOTHESIS: "Dropping model-specific feature families (ou, hmm)
# panel, using ALL 24 sp_* feature columns (ou,hmm,jump,har,trend,hurst, # improves the rank signal" — exp 25 cleanly tested OU (adding it hurts: IC 0.0511->0.0343);
# signature). Copy of the canonical workflow_lgb_sp5d_rankic.yaml with a # hmm-as-features has NOT been clean A/B'd post-reset (exp 42 tested hmm as an entry
# distinct experiment name so the ablation runs are isolated. # GATE overlay, refuted). This run adds the hmm family columns to the compact set.
# # Change vs exp-26 reference: features += sp_hmm_p_regime1, sp_hmm_state.
# Run: # Acceptance (prune-hypothesis): no improvement — RankIC <= 0.0663, net_IR <= 0.21.
# rd_run_workflow config_path=tac-qlib/workflows/ablate_baseline_all_sp_fields.yaml \ # Run: rd_run_workflow config_path=<repo>/experiments/queue/workflows/q16_hmm_features.yaml \
# experiment_name=tac-rd-rank-ablate # experiment_name=tac-rd-q16-hmm-features
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %} {%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %} {%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %} {%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead,sp_hmm_p_regime1,sp_hmm_state" %}
qlib_init: qlib_init:
provider_uri: "{{ LAKE }}" provider_uri: "{{ LAKE }}"
region: us region: us
expression_cache: null expression_cache: null
dataset_cache: null dataset_cache: null
calendar_provider: calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US }
lake_root: "{{ LAKE }}"
market: US
instrument_provider: instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider: feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US }
lake_root: "{{ LAKE }}"
market: US
exp_manager: exp_manager:
class: MLflowExpManager class: MLflowExpManager
module_path: qlib.workflow.expm module_path: qlib.workflow.expm
kwargs: kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-q16-hmm-features" }
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-rank-ablate"
task: task:
model: model:
class: RankICLGBModel class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_gbdt module_path: tac_qlib.contrib.model.rank_ensemble
kwargs: kwargs:
loss: mse loss: mse
learning_rate: 0.02 learning_rate: 0.02
@@ -60,7 +48,7 @@ task:
subsample_freq: 1 subsample_freq: 1
reg_alpha: 0.1 reg_alpha: 0.1
reg_lambda: 1.0 reg_lambda: 1.0
seed: 42 seeds: "42,7,2026,99,123"
dataset: dataset:
class: DatasetH class: DatasetH
@@ -73,38 +61,27 @@ task:
instruments: "{{ UNIVERSE }}" instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03 start_time: 2015-01-03
end_time: 2026-08-10 end_time: 2026-08-10
fit_start_time: 2015-01-03 fit_start_time: 2016-01-04
fit_end_time: 2025-09-01 fit_end_time: 2025-09-01
freq: day freq: day
lake_root: "{{ LAKE }}" lake_root: "{{ LAKE }}"
market: US market: US
label: "Ref($close,-6)/Ref($close,-1)-1" label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}" feature_fields: "{{ FEATURES }}"
infer_processors: infer_processors:
- class: DropAllNaN - { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
kwargs: {} - { class: ProcessInf, kwargs: {} }
- class: ProcessInf - { class: CSRankNorm, kwargs: {} }
kwargs: {} - { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- class: CSRankNorm - { class: Fillna, kwargs: {} }
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments: segments:
train: [2015-01-03, 2025-09-01] train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03] valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10] test: [2026-01-04, 2026-08-10]
record: record:
- class: SignalRecord - { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
module_path: qlib.workflow.record_temp - { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord - class: PortAnaRecord
module_path: qlib.workflow.record_temp module_path: qlib.workflow.record_temp
kwargs: kwargs:
@@ -112,12 +89,7 @@ task:
strategy: strategy:
class: TopkDropoutStrategy class: TopkDropoutStrategy
module_path: qlib.contrib.strategy module_path: qlib.contrib.strategy
kwargs: kwargs: { signal: "<PRED>", topk: 10, n_drop: 1, only_tradable: true, risk_degree: 0.95 }
signal: "<PRED>"
topk: 10
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest: backtest:
start_time: 2026-01-04 start_time: 2026-01-04
end_time: 2026-08-10 end_time: 2026-08-10
@@ -130,4 +102,4 @@ task:
open_cost: 0.0005 open_cost: 0.0005
close_cost: 0.0015 close_cost: 0.0015
min_cost: 5.0 min_cost: 5.0
risk_analysis_freq: 1d risk_analysis_freq: 1d
@@ -1,53 +1,34 @@
# ----------------------------------------------------------------------------- # QUEUE-17 — Realized-moments family added to the compact set.
# EXP 18 - Risk-limit control: reference model + TopkDropout baseline (A). # CLAIMS.md HYPOTHESIS: "Adding moment/volatility families regresses the signal"
# # (idea: pre-clean-lake exp 11). M1 momentum bundle (exp 29) and M3 GARCH (exp 31)
# Signal/model identical to the reference (tac-rd-rank-ensemble-isolated, # were refuted post-reset; the realized-moments family (sp_rskew/sp_rkurt/sp_dsv)
# run 0cea66d9...): RankICEnsembleLGBModel (parallel, 5 seeds) on the 50-ETF # has NOT been clean A/B'd. This run adds the moments columns to the compact set.
# SP-5d panel, test 2026-01-04..2026-08-10. This workflow reproduces the # Change vs exp-26 reference: features += sp_rskew_5,sp_rskew_22,sp_rkurt_5,sp_rkurt_22,sp_dsv_5,sp_dsv_22.
# unconstrained TopkDropout baseline net-of-cost so the risk-limited variant # Acceptance (prune-hypothesis): no improvement — RankIC <= 0.0663, net_IR <= 0.21.
# (same pred, liquidity/size/concentration caps) can be compared 1:1. # Run: rd_run_workflow config_path=<repo>/experiments/queue/workflows/q17_moments_features.yaml \
# # experiment_name=tac-rd-q17-moments-features
# The risk_limits spec itself is applied via rd_backtest / rd_strategy_targets
# (tool-level param, not a YAML key); this run records the unconstrained
# baseline that the limit A/B is measured against.
#
# Run:
# rd_run_workflow config_path=experiments/workflows/exp18-risk-limit/a_baseline.yaml \
# experiment_name=tac-rd-risk-limit
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %} {%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %} {%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %} {%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead,sp_rskew_5,sp_rskew_22,sp_rkurt_5,sp_rkurt_22,sp_dsv_5,sp_dsv_22" %}
qlib_init: qlib_init:
provider_uri: "{{ LAKE }}" provider_uri: "{{ LAKE }}"
region: us region: us
expression_cache: null expression_cache: null
dataset_cache: null dataset_cache: null
calendar_provider: calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US }
lake_root: "{{ LAKE }}"
market: US
instrument_provider: instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider: feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: kwargs: { lake_root: "{{ LAKE }}", market: US }
lake_root: "{{ LAKE }}"
market: US
exp_manager: exp_manager:
class: MLflowExpManager class: MLflowExpManager
module_path: qlib.workflow.expm module_path: qlib.workflow.expm
kwargs: kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-q17-moments-features" }
uri: "sqlite:///{{ LAKE }}/mlruns.db"
default_exp_name: "tac-rd-risk-limit"
task: task:
model: model:
@@ -68,7 +49,6 @@ task:
reg_alpha: 0.1 reg_alpha: 0.1
reg_lambda: 1.0 reg_lambda: 1.0
seeds: "42,7,2026,99,123" seeds: "42,7,2026,99,123"
parallel: 5
dataset: dataset:
class: DatasetH class: DatasetH
@@ -80,39 +60,28 @@ task:
kwargs: kwargs:
instruments: "{{ UNIVERSE }}" instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03 start_time: 2015-01-03
end_time: 2026-08-14 end_time: 2026-08-10
fit_start_time: 2016-01-04 fit_start_time: 2016-01-04
fit_end_time: 2025-09-01 fit_end_time: 2025-09-01
freq: day freq: day
lake_root: "{{ LAKE }}" lake_root: "{{ LAKE }}"
market: US market: US
label: "Ref($close,-6)/Ref($close,-1)-1" label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}" feature_fields: "{{ FEATURES }}"
infer_processors: infer_processors:
- class: DropAllNaN - { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
kwargs: {} - { class: ProcessInf, kwargs: {} }
- class: ProcessInf - { class: CSRankNorm, kwargs: {} }
kwargs: {} - { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- class: CSRankNorm - { class: Fillna, kwargs: {} }
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments: segments:
train: [2016-01-04, 2025-09-01] train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03] valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10] test: [2026-01-04, 2026-08-10]
record: record:
- class: SignalRecord - { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
module_path: qlib.workflow.record_temp - { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord - class: PortAnaRecord
module_path: qlib.workflow.record_temp module_path: qlib.workflow.record_temp
kwargs: kwargs:
@@ -120,12 +89,7 @@ task:
strategy: strategy:
class: TopkDropoutStrategy class: TopkDropoutStrategy
module_path: qlib.contrib.strategy module_path: qlib.contrib.strategy
kwargs: kwargs: { signal: "<PRED>", topk: 10, n_drop: 1, only_tradable: true, risk_degree: 0.95 }
signal: "<PRED>"
topk: 10
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest: backtest:
start_time: 2026-01-04 start_time: 2026-01-04
end_time: 2026-08-10 end_time: 2026-08-10
@@ -138,4 +102,4 @@ task:
open_cost: 0.0005 open_cost: 0.0005
close_cost: 0.0015 close_cost: 0.0015
min_cost: 5.0 min_cost: 5.0
risk_analysis_freq: 1d risk_analysis_freq: 1d
+106
View File
@@ -0,0 +1,106 @@
# QUEUE-18 — OptimalStopControl clean re-test vs TopkDropout (exp 13/14 claim).
# CLAIMS.md HYPOTHESIS: "TopkDropout beats stochastic-control OptimalStopControl on
# the ensemble signal" — exp 13/14 were pre-clean-lake; never re-tested post-reset.
# Same compact signal as the exp-26 reference; ONLY the strategy changes to
# OptimalStopControl with exp-13 params (entry 0.85 / exit 0.7 / hold 10 / sl -0.08).
# PREREQUISITE: tac_qlib/contrib/strategy/optimal_stop.py must be synced to the venv
# site-packages snapshot before running (see /app/AGENTS.md).
# Acceptance: TopkDropout net_IR >= stop-control net_IR; document cost drag of both.
# Run: rd_run_workflow config_path=<repo>/experiments/queue/workflows/q18_optstop.yaml \
# experiment_name=tac-rd-q18-optstop
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-q18-optstop" }
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-10
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "{{ FEATURES }}"
infer_processors:
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: ProcessInf, kwargs: {} }
- { class: CSRankNorm, kwargs: {} }
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: Fillna, kwargs: {} }
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: OptimalStopControl
module_path: tac_qlib.contrib.strategy.optimal_stop
kwargs: { signal: "<PRED>", topk: 10, entry_pct: 0.85, exit_pct: 0.7, max_hold_days: 10, min_hold_days: 2, sl: -0.08 }
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
+107
View File
@@ -0,0 +1,107 @@
# QUEUE-21 — Long-horizon label (10d) + weekly recompute construction.
# Untested combination from book/CLAIMS.md open questions: Q04 (exp 36) proved the
# 10d label has strong signal (IC 0.093, RankIC 0.096, L/S Sharpe 5.89) but daily
# turnover killed the book (net -9.92%); Q07 (exp 39) proved weekly recompute is
# the cost lever (net +12.51%). Q12 already tested 22d+weekly and failed (net -4.88%),
# so the 22d label's problem is not just turnover. Hypothesis: the 10d label's edge
# survives weekly rebalance because it captures a shorter, more actionable horizon.
# Change vs exp-26 reference: label 5d -> 10d AND strategy -> WeeklyRebalanceDropoutStrategy.
# Acceptance: net_IR > 0.5, net_ann > +5%, cost drag <= 2pp.
# Run: rd_run_workflow config_path=<repo>/experiments/queue/workflows/q21_label10d_weekly.yaml \
# experiment_name=tac-rd-q21-label10d-weekly
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs: { lake_root: "{{ LAKE }}", market: US }
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-q21-label10d-weekly" }
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-10
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-11)/Ref($close,-1)-1"
feature_fields: "{{ FEATURES }}"
infer_processors:
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: ProcessInf, kwargs: {} }
- { class: CSRankNorm, kwargs: {} }
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
- { class: Fillna, kwargs: {} }
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: WeeklyRebalanceDropoutStrategy
module_path: tac_qlib.contrib.strategy.weekly_rebalance
kwargs: { signal: "<PRED>", topk: 10, n_drop: 1, only_tradable: true, risk_degree: 0.95 }
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d
-141
View File
@@ -1,141 +0,0 @@
# -----------------------------------------------------------------------------
# ISOLATION: multi-seed RankIC ensemble, ablate-B generic-only feature set.
#
# Isolates the ensemble effect on the SP-5d rank signal. Same panel, segments,
# history (full backfilled 2016+) and feature set as the exp-9 ablate-B winner
# (generic-only sp_* families: jump,har,trend,hurst,signature), but replaces the
# single RankICLGBModel with a 5-seed RankICEnsembleLGBModel (42,7,2026,99,123)
# that averages per-day predictions.
#
# Differs from exp-15 (tac-rd-rank-ensemble, mlflow exp 15) ONLY by dropping the
# TA subset (rsi_14,roc_10,macd_hist,willr_14,atr_14) and the inter-asset xr_*
# features, so any change vs exp-15 is attributable to the feature set alone,
# and any change vs exp-9 is attributable to the ensemble + full history alone.
#
# Run:
# rd_run_workflow config_path=experiments/workflows/exp12_isolation_ensemble.yaml \
# experiment_name=tac-rd-rank-ensemble-isolated
# -----------------------------------------------------------------------------
{%- set LAKE = TAC_LAKE_DIR %}
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
qlib_init:
provider_uri: "{{ LAKE }}"
region: us
expression_cache: null
dataset_cache: null
calendar_provider:
class: tac_qlib.data.providers.LakeCalendarProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
instrument_provider:
class: tac_qlib.data.providers.LakeInstrumentProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
markets: {}
feature_provider:
class: tac_qlib.data.providers.LakeFeatureProvider
kwargs:
lake_root: "{{ LAKE }}"
market: US
exp_manager:
class: MLflowExpManager
module_path: qlib.workflow.expm
kwargs:
uri: "sqlite:///mlruns.db"
default_exp_name: "tac-rd-rank-ensemble-isolated"
task:
model:
class: RankICEnsembleLGBModel
module_path: tac_qlib.contrib.model.rank_ensemble
kwargs:
loss: mse
learning_rate: 0.02
num_leaves: 31
n_estimators: 3000
num_boost_round: 3000
early_stopping_rounds: 200
min_data_in_leaf: 20
lambda_l2: 0.5
colsample_bytree: 0.8
subsample: 0.8
subsample_freq: 1
reg_alpha: 0.1
reg_lambda: 1.0
seeds: "42,7,2026,99,123"
dataset:
class: DatasetH
module_path: qlib.data.dataset
kwargs:
handler:
class: TACHandler
module_path: tac_qlib.contrib.data.handler
kwargs:
instruments: "{{ UNIVERSE }}"
start_time: 2015-01-03
end_time: 2026-08-14
fit_start_time: 2016-01-04
fit_end_time: 2025-09-01
freq: day
lake_root: "{{ LAKE }}"
market: US
label: "Ref($close,-6)/Ref($close,-1)-1"
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
infer_processors:
- class: DropAllNaN
kwargs: {}
- class: ProcessInf
kwargs: {}
- class: CSRankNorm
kwargs: {}
- class: ZScoreNorm
kwargs: {}
- class: Fillna
kwargs: {}
segments:
train: [2016-01-04, 2025-09-01]
valid: [2025-09-03, 2026-01-03]
test: [2026-01-04, 2026-08-10]
record:
- class: SignalRecord
module_path: qlib.workflow.record_temp
kwargs: {}
- class: SigAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
ana_long_short: true
ann_scaler: 252
- class: PortAnaRecord
module_path: qlib.workflow.record_temp
kwargs:
config:
strategy:
class: TopkDropoutStrategy
module_path: qlib.contrib.strategy
kwargs:
signal: "<PRED>"
topk: 10
n_drop: 2
only_tradable: true
risk_degree: 0.95
backtest:
start_time: 2026-01-04
end_time: 2026-08-10
account: 1000000
benchmark: SPY
exchange_kwargs:
codes: "{{ UNIVERSE }}"
deal_price: $close
freq: day
open_cost: 0.0005
close_cost: 0.0015
min_cost: 5.0
risk_analysis_freq: 1d