start experiment 62 (exp/62-scheduled-algo-retrain-on-2026-08-25-tac)

This commit is contained in:
zhaoli
2026-08-26 12:01:18 +00:00
parent 1142cc9eef
commit 1c0c8f5ecd
14 changed files with 191 additions and 18 deletions
+14 -14
View File
@@ -1,34 +1,34 @@
# TradeAC custom-qlib-code snapshot (auto-generated)
# parent repo HEAD : ee5aa8e73c5286860d3ec1dfa6001bf0f8b6690d
# parent repo HEAD : 1142cc9eefde88739f33f34aa96af7b182320d96
# tac-qlib/tac_qlib/contrib
# tac-qlib/tac_qlib/data
# per-file hashes (git hash-object):
1b6298c4a5652f2e863cbdc385a1014a570fcd59 tac-qlib/tac_qlib/contrib/__init__.py
e21663470e4ab108691392c3f0bcff0a3e7fe459 tac-qlib/tac_qlib/contrib/__pycache__/__init__.cpython-312.pyc
6c57851807631dfa1a525f87538a1b0a495fd7b2 tac-qlib/tac_qlib/contrib/__pycache__/__init__.cpython-312.pyc
2224424d0ff193be4f55d1b791f8fce89439c5d2 tac-qlib/tac_qlib/contrib/backtest/__init__.py
0bf40dee440ddbded357d7bbb4efc67c62c4b084 tac-qlib/tac_qlib/contrib/backtest/tradeac_exchange.py
c76a9f17f680e74eea766eff27f7624359749ed6 tac-qlib/tac_qlib/contrib/data/__init__.py
a5c6952e9c8bab2c83fd8030416999726aea73d9 tac-qlib/tac_qlib/contrib/data/__pycache__/__init__.cpython-312.pyc
39f148bef96fcce8f548d74c6d338eccdc4145e3 tac-qlib/tac_qlib/contrib/data/__pycache__/handler.cpython-312.pyc
fc3d01d530d4e84b3386614b6ac94324e811920e tac-qlib/tac_qlib/contrib/data/handler.py
1acd2cb845eac1bcee54450004a4af36484544ed tac-qlib/tac_qlib/contrib/data/__pycache__/__init__.cpython-312.pyc
2ef18965f77e8955580334d2edc09bd381204355 tac-qlib/tac_qlib/contrib/data/__pycache__/handler.cpython-312.pyc
3bba0f1696e4ab4b3deebec3f31f269b2e713899 tac-qlib/tac_qlib/contrib/data/handler.py
b151d139a0dcde87d74b21e7c4b729176ba5c39b tac-qlib/tac_qlib/contrib/model/__init__.py
f76257db439b030d84ecea2b8ae67079bae4dc6f tac-qlib/tac_qlib/contrib/model/__pycache__/__init__.cpython-312.pyc
858f5fcbe285642569c214d5803edbae1e5dfa1e tac-qlib/tac_qlib/contrib/model/__pycache__/rank_ensemble.cpython-312.pyc
7de125f7f263c554b3337138c760d44394b4824a tac-qlib/tac_qlib/contrib/model/__pycache__/rank_gbdt.cpython-312.pyc
c975d2b978f2cc08a388a5d921938704a3dd592d tac-qlib/tac_qlib/contrib/model/__pycache__/__init__.cpython-312.pyc
009ebd83c5156ca3d7039a112e0d277dd416ca86 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_ensemble.cpython-312.pyc
1716b680b5623394229f7600ad4c81ad07fa6a2b tac-qlib/tac_qlib/contrib/model/__pycache__/rank_gbdt.cpython-312.pyc
d3f051f3a8650c42fedc7b367b966f7c74fb5789 tac-qlib/tac_qlib/contrib/model/rank_ensemble.py
d03e6611338918d4aac5eea4adf26f85a3763652 tac-qlib/tac_qlib/contrib/model/rank_gbdt.py
184f80da8edf944bad3c8fb4d4d3d189bf4f082b tac-qlib/tac_qlib/contrib/strategy/__init__.py
6893c2097f0de110909f3f5076ac1c32ac5b742b tac-qlib/tac_qlib/contrib/strategy/__pycache__/__init__.cpython-312.pyc
423ec5cd15b273c119ba3b1d7dc7467fc6ba41c4 tac-qlib/tac_qlib/contrib/strategy/__pycache__/long_short.cpython-312.pyc
b6b3238699337d559868cb5f8e0c3e900e5da0cf tac-qlib/tac_qlib/contrib/strategy/__pycache__/optimal_stop.cpython-312.pyc
9c9f7743970b1a3827bb72768bb6e8be03040759 tac-qlib/tac_qlib/contrib/strategy/__pycache__/__init__.cpython-312.pyc
6e38a7fa8584b80410ccc88e5feff228a7ece38b tac-qlib/tac_qlib/contrib/strategy/__pycache__/long_short.cpython-312.pyc
f983d5c2472cd16ef9f14a240674ec0a7f41e81c tac-qlib/tac_qlib/contrib/strategy/__pycache__/optimal_stop.cpython-312.pyc
896ef74ae47bcd1ed388e1e5d9c8d70c28097fe9 tac-qlib/tac_qlib/contrib/strategy/kelly_dropout.py
9090fc6dfbd339f2f4df4b0c9b87f400ecb5c9d5 tac-qlib/tac_qlib/contrib/strategy/long_short.py
79aaad9e39fcc740a773f4f63c512ce1086cfde0 tac-qlib/tac_qlib/contrib/strategy/optimal_stop.py
5b9acfb4340111b204249add7760bd53c6ae03f1 tac-qlib/tac_qlib/contrib/strategy/regime_gate.py
fe60bacdfedd48617863be31f24b7c7daebfac5a tac-qlib/tac_qlib/contrib/strategy/weekly_rebalance.py
92e6e90eb0cd0a25142034560f27adb6b705b1a8 tac-qlib/tac_qlib/data/__init__.py
fa1090eb6f17b1b7835038e35fac86334294d165 tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc
48629a1f51bd4058036ce78d79ed49454982c051 tac-qlib/tac_qlib/data/__pycache__/config.cpython-312.pyc
ad7c9b726c9c52773fa82031fcd681337be97fa8 tac-qlib/tac_qlib/data/__pycache__/providers.cpython-312.pyc
a0e969bd6504bb8d9220f4641cc01e960c3120e4 tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc
2f8c537d11135155539276bee342d087aad8743e tac-qlib/tac_qlib/data/__pycache__/config.cpython-312.pyc
53cf7828c425f6b5032b206b93a238607111a6ed tac-qlib/tac_qlib/data/__pycache__/providers.cpython-312.pyc
1953fb2a6371525db7f7b0e1c9dfbf3492d82110 tac-qlib/tac_qlib/data/config.py
8d0644f6f0d1efb94798ed444cc73e63b643459b tac-qlib/tac_qlib/data/providers.py
+177 -4
View File
@@ -15,6 +15,9 @@ import os
from inspect import getfullargspec
from typing import List, Optional, Tuple, Union
import numpy as np
import pandas as pd
from qlib.data.dataset import processor as processor_module
from qlib.data.dataset.handler import DataHandlerLP
from qlib.utils import get_callable_kwargs
@@ -144,6 +147,174 @@ class DropAllNaN(processor_module.Processor):
return df
class BenchResidual(processor_module.Processor):
"""Subtract a benchmark instrument's forward return from the label, per datetime.
Turns the training target from an absolute-return rank into a *residual* rank:
``r_i - r_bench`` is ranked cross-sectionally by the downstream ``CSRankNorm`` /
``CSZScoreNorm`` processors instead of ``r_i`` alone. Must be inserted BEFORE any
per-date normalization so the ranking itself is computed on residual returns
(ordering flips exactly where the benchmark trends).
Stateless: ``fit`` is a no-op and the benchmark forward return is recomputed from
the lake parquet on first ``__call__``. Rows whose benchmark value is missing are
left untouched. Accepts ``fit_start_time``/``fit_end_time`` (ignored) so
``check_transform_proc`` can inject the fit window uniformly.
NOTE: under any cross-sectional normalization downstream (``CSRankNorm`` /
``CSZScoreNorm``) this processor is a mathematical no-op: subtracting the same
per-date constant preserves ranks, and z-scoring absorbs constant shifts. Use
``BenchBetaResidual`` for a target that actually reorders.
"""
def __init__(
self,
benchmark="SPY",
fields_group="label",
lake_root=None,
market="US",
timeframe=None,
freq="day",
fit_start_time=None,
fit_end_time=None,
):
self.benchmark = benchmark
self.fields_group = fields_group
self.lake_root = lake_root
self.market = market
self.timeframe = timeframe or timeframe_for_freq(freq)
self.fit_start_time = fit_start_time
self.fit_end_time = fit_end_time
self._bench_label = None
def _load_bench_label(self):
if self._bench_label is not None:
return self._bench_label
cfg = LakeConfig(self.lake_root, self.market)
p = cfg.bar_path(self.timeframe, self.benchmark)
if not p.exists():
raise FileNotFoundError(f"BenchResidual: benchmark bar file not found: {p}")
df = pd.read_parquet(p)
s = pd.Series(df["c"].astype(float).values, index=pd.to_datetime(df["t"])).sort_index()
s.index = s.index.normalize()
# mirror Ref($close,-6)/Ref($close,-1)-1 on the benchmark's own calendar
bench_label = s.shift(-6) / s.shift(-1) - 1
self._bench_label = bench_label[~bench_label.index.duplicated(keep="last")]
return self._bench_label
def fit(self, df=None):
return self
def __call__(self, df):
bl = self._load_bench_label()
cols = processor_module.get_group_columns(df, self.fields_group)
dt = df.index.get_level_values("datetime")
aligned = bl.reindex(pd.DatetimeIndex(dt.unique())).reindex(dt)
mask = aligned.notna().values
out = df.copy()
for c in cols:
vals = df[c].values
res = vals.copy()
res[mask] = np.asarray(vals[mask], dtype=float) - aligned[mask].values
out[c] = res
return out
class BenchBetaResidual(processor_module.Processor):
"""Residualize the label against a beta-scaled benchmark move: ``r_i - b_i * r_bench``.
Unlike a plain constant subtraction (see ``BenchResidual``), the name-specific rolling
beta ``b_i`` makes this survive cross-sectional normalization: in up-weeks high-beta
names lose rank, in down-weeks they gain — exactly the relative structure an absolute-
return ranking hides.
Beta is estimated from *past* data only (rolling ``window`` trading days of daily close
returns of each instrument vs the benchmark, both read up to and including ``t``), so
no lookahead enters the target. The benchmark leg uses the same horizon as the label
expression (``Ref($close,-6)/Ref($close,-1)-1`` by default via ``horizon``/``base``,
matching the yaml's 6-day label). Rows with missing beta or benchmark values keep
their raw label.
Requires ``$close`` to be present in the feature group (it always is for TACHandler).
Stateless; accepts ``fit_start_time``/``fit_end_time`` (ignored) for uniform kwargs
injection. Must be inserted BEFORE any per-date normalization processor.
"""
def __init__(
self,
benchmark="SPY",
fields_group="label",
lake_root=None,
market="US",
timeframe=None,
freq="day",
window=63,
horizon=6,
base=1,
feature_field="$close",
fit_start_time=None,
fit_end_time=None,
):
self.benchmark = benchmark
self.fields_group = fields_group
self.lake_root = lake_root
self.market = market
self.timeframe = timeframe or timeframe_for_freq(freq)
self.window = int(window)
self.horizon = int(horizon)
self.base = int(base)
self.feature_field = feature_field
self.fit_start_time = fit_start_time
self.fit_end_time = fit_end_time
self._bench = None
def _load_bench_close(self):
if self._bench is not None:
return self._bench
cfg = LakeConfig(self.lake_root, self.market)
p = cfg.bar_path(self.timeframe, self.benchmark)
if not p.exists():
raise FileNotFoundError(f"BenchBetaResidual: benchmark bar file not found: {p}")
df = pd.read_parquet(p)
s = pd.Series(df["c"].astype(float).values, index=pd.to_datetime(df["t"])).sort_index()
s.index = s.index.normalize()
self._bench = s[~s.index.duplicated(keep="last")]
return self._bench
def fit(self, df=None):
return self
def __call__(self, df):
bench = self._load_bench_close()
# benchmark forward return over the same horizon as the label expression
fwd = bench.shift(-(self.base + self.horizon - 1)) / bench.shift(-self.base) - 1
px_col = ("feature", self.feature_field)
if px_col not in df.columns:
raise KeyError(f"BenchBetaResidual: {self.feature_field} not found in features")
px = df[px_col].unstack("instrument").sort_index()
rets = px / px.shift(1) - 1
bret = bench.reindex(px.index).pct_change()
# rolling beta per instrument using data <= t (no lookahead)
cov = rets.rolling(self.window, min_periods=max(10, self.window // 2)).cov(bret)
var = bret.rolling(self.window, min_periods=max(10, self.window // 2)).var()
beta = cov.div(var, axis=0)
contrib = beta.mul(fwd.reindex(px.index), axis=0)
cols = list(processor_module.get_group_columns(df, self.fields_group))
out = df.copy()
for c in cols:
lab = df[c].unstack("instrument").reindex(px.index)
resid = lab - contrib.where(contrib.notna() & lab.notna(), 0.0)
new_vals = resid.stack()
new_vals.index.names = df.index.names
# residual where available, raw label otherwise (e.g. beta warm-up rows)
out[c] = new_vals.reindex(out.index).fillna(df[c])
return out
class TACHandler(DataHandlerLP):
"""DataHandlerLP backed by the TradeAC parquet lake.
@@ -246,10 +417,12 @@ class TACHandler(DataHandlerLP):
return get_common_feature_fields(lake_root, market, timeframe_for_freq(freq))
__all__ = ["TACHandler", "DropAllNaN", "get_common_feature_fields"]
__all__ = ["TACHandler", "DropAllNaN", "BenchResidual", "BenchBetaResidual", "get_common_feature_fields"]
# Make `DropAllNaN` resolvable by bare name from processor configs (e.g. the default
# ``infer_processors`` and workflow yamls that reference it without a ``module_path``),
# mirroring how qlib registers its own processors in ``qlib.data.dataset.processor``.
# Make `DropAllNaN`/`BenchResidual`/`BenchBetaResidual` resolvable by bare name from processor
# configs (e.g. the default ``infer_processors`` and workflow yamls that reference them without a
# ``module_path``), mirroring how qlib registers its own processors in ``qlib.data.dataset.processor``.
processor_module.DropAllNaN = DropAllNaN
processor_module.BenchResidual = BenchResidual
processor_module.BenchBetaResidual = BenchBetaResidual