book: scaffold + ch00 (execution trail as spine) — evidence exp 8-31, round 3
This commit is contained in:
@@ -0,0 +1,70 @@
|
||||
"""Minimal custom DataHandler — how to extend TACHandler for a new feature family.
|
||||
|
||||
`TACHandler(DataHandlerLP)` already routes lake bars + ta-lib features via
|
||||
`LakeFeatureProvider` (see tac_qlib/contrib/data/handler.py). To add a NEW
|
||||
feature family (computed once, persisted into the lake features parquet — see
|
||||
examples/persist_sp_features.py), you only need to:
|
||||
|
||||
1. persist extra columns into features/market=US/timeframe=1d/symbol=*.parquet
|
||||
2. list them in `feature_fields` (they are prefixed with `$` and de-duped)
|
||||
|
||||
A subclass is only needed when the feature must be computed *inside* the qlib
|
||||
pipeline (e.g. as an extra processor). This file sketches that pattern.
|
||||
|
||||
Reference handler structure (from tac_qlib/contrib/data/handler.py):
|
||||
|
||||
class TACHandler(DataHandlerLP):
|
||||
def __init__(self, instruments, start_time, end_time, freq,
|
||||
fit_start_time=None, fit_end_time=None,
|
||||
feature_fields=None, label=None, lake_root=None, market="US",
|
||||
infer_processors=None, learn_processors=None, **kwargs):
|
||||
loader = QlibDataLoader(configured=(feature_fields or self.DEFAULT_FIELDS), freq=freq)
|
||||
super().__init__(instruments, start_time, end_time, freq=freq,
|
||||
data_loader=loader,
|
||||
infer_processors=infer_processors or DEFAULT_INFER_PROCESSORS,
|
||||
learn_processors=learn_processors or DEFAULT_LEARN_PROCESSORS,
|
||||
fit_start_time=fit_start_time, fit_end_time=fit_end_time,
|
||||
process_type=DataHandlerLP.PTYPE_A, **kwargs)
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, List, Optional
|
||||
|
||||
from tac_qlib.contrib.data.handler import DEFAULT_INFER_PROCESSORS, DEFAULT_LEARN_PROCESSORS, TACHandler
|
||||
|
||||
|
||||
class CustomFeaturesHandler(TACHandler):
|
||||
"""TACHandler variant that also loads the lake feature columns passed in.
|
||||
|
||||
Usage from YAML — only the handler kwargs change:
|
||||
|
||||
handler:
|
||||
class: CustomFeaturesHandler
|
||||
module_path: tac_qlib.contrib.data.handler # after adding this class there
|
||||
kwargs:
|
||||
instruments: AAPL,MSFT,QQQ
|
||||
start_time: 2026-03-01
|
||||
end_time: 2026-08-06
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
feature_fields: "$close,sp_ou_alpha,sp_hurst_exponent"
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
feature_fields: Optional[List[str]] = None,
|
||||
infer_processors: Optional[List[Any]] = None,
|
||||
learn_processors: Optional[List[Any]] = None,
|
||||
**kwargs: Any,
|
||||
):
|
||||
# `feature_fields` are passed through with the leading `$` stripped by
|
||||
# TACHandler; infer/learn default to the lake-tuned processor stacks.
|
||||
super().__init__(
|
||||
feature_fields=feature_fields,
|
||||
infer_processors=infer_processors or DEFAULT_INFER_PROCESSORS,
|
||||
learn_processors=learn_processors or DEFAULT_LEARN_PROCESSORS,
|
||||
**kwargs,
|
||||
)
|
||||
@@ -0,0 +1,78 @@
|
||||
"""Minimal RankIC early-stopping LightGBM model (the biggest IC/backtest lever).
|
||||
|
||||
Drop-in replacement for `qlib.contrib.model.gbdt.LGBModel` in a workflow YAML:
|
||||
|
||||
task.model:
|
||||
class: RankICLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_gbdt
|
||||
kwargs: { loss: mse, learning_rate: 0.02, num_boost_round: 3000,
|
||||
early_stopping_rounds: 200, lambda_l2: 0.5 }
|
||||
|
||||
What it changes vs stock LGBModel:
|
||||
* `_prepare_data` builds `lgb.Dataset` with per-day `group` query groups, so
|
||||
metrics are computed per trading day.
|
||||
* `fit` injects `feval=rankic_feval` (mean per-day Spearman) into `lgb.train`
|
||||
and forces `metric='None'` + `first_metric_only=True` so early stopping
|
||||
tracks RankIC — not l2, which keeps improving after RankIC peaks.
|
||||
|
||||
Why: with ~50 instruments per day, ranking objectives (lambda_rank/xendcg)
|
||||
produce near-zero RankIC; MSE objective + RankIC early-stop is what lifts it.
|
||||
|
||||
Install: copy to tac_qlib/contrib/model/rank_gbdt.py AND the /opt/venv copy.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from qlib.contrib.model.gbdt import LGBModel
|
||||
|
||||
|
||||
def rankic_feval(preds: np.ndarray, dataset) -> tuple[str, float, bool]:
|
||||
"""Mean per-day Spearman rank IC between predictions and the label."""
|
||||
label = dataset.get_label()
|
||||
group = dataset.get_group() if hasattr(dataset, "get_group") else None
|
||||
if group is None:
|
||||
return "rankic", _spearman(preds, label), False
|
||||
|
||||
start = 0
|
||||
ics = []
|
||||
for g in group:
|
||||
sl = slice(start, start + g)
|
||||
start += g
|
||||
ics.append(_spearman(preds[sl], label[sl]))
|
||||
return "rankic", float(np.mean(ics)), False
|
||||
|
||||
|
||||
def _spearman(x: np.ndarray, y: np.ndarray) -> float:
|
||||
if len(x) < 2:
|
||||
return 0.0
|
||||
from scipy.stats import spearmanr
|
||||
|
||||
rho, _ = spearmanr(x, y)
|
||||
return float(rho) if rho == rho else 0.0
|
||||
|
||||
|
||||
class RankICLGBModel(LGBModel):
|
||||
"""LGBModel with per-day query groups and RankIC-only early stopping."""
|
||||
|
||||
def _prepare_data(self, dataset, *args, **kwargs):
|
||||
"""Attach per-day group sizes to the train/valid lgb.Dataset."""
|
||||
dtrain, dvalid = super()._prepare_data(dataset, *args, **kwargs)
|
||||
for d, index in ((dtrain, dataset.get_index_by_segment("train")), (dvalid, dataset.get_index_by_segment("valid"))):
|
||||
if d is not None and index is not None:
|
||||
# group by calendar day in order
|
||||
days = pd.Series([i[0] for i in index])
|
||||
group = days.value_counts().sort_index().tolist()
|
||||
d.set_group(np.array(group, dtype=np.int32))
|
||||
return dtrain, dvalid
|
||||
|
||||
def fit(self, dataset, evals_result: Dict[str, Any] | None = None, **kwargs):
|
||||
# force RankIC-only early stopping
|
||||
kwargs.setdefault("feval", rankic_feval)
|
||||
kwargs.setdefault("metric", "None")
|
||||
kwargs.setdefault("first_metric_only", True)
|
||||
return super().fit(dataset, evals_result=evals_result, **kwargs)
|
||||
@@ -0,0 +1,66 @@
|
||||
"""Minimal persistence of computed SP features into the lake features parquet.
|
||||
|
||||
Flow: compute sp_* features per symbol (examples/sp_features.py) and MERGE them
|
||||
into features/market=US/timeframe=1d/symbol=*.parquet so TACHandler /
|
||||
LakeFeatureProvider can route `$sp_ou_theta` etc. from the workflow YAML.
|
||||
|
||||
Run after backfilling bars; re-run drops stale sp_* columns first (see note).
|
||||
|
||||
python examples/persist_sp_features.py --market US --timeframe 1d
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import os
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from tac_qlib.data.config import LakeConfig, NON_FEATURE_COLUMNS
|
||||
from examples.sp_features import build_sp_features
|
||||
|
||||
#: columns owned by this feature family (replaced on re-runs, never duplicated)
|
||||
SP_PREFIX = "sp_"
|
||||
|
||||
|
||||
def persist_symbol(lake: LakeConfig, timeframe: str, symbol: str) -> None:
|
||||
bars_path = lake.bar_path(timeframe, symbol)
|
||||
feats_path = lake.features_path(timeframe, symbol)
|
||||
if not bars_path.exists():
|
||||
return
|
||||
bars = pd.read_parquet(bars_path)
|
||||
feats = build_sp_features(bars)
|
||||
# bars have a single 't'/'date' column; align feature rows to it
|
||||
feats = feats.drop(columns=[c for c in NON_FEATURE_COLUMNS if c in feats.columns], errors="ignore")
|
||||
|
||||
feats_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
if feats_path.exists():
|
||||
existing = pd.read_parquet(feats_path)
|
||||
# drop stale sp_* columns before merging (idempotent re-runs)
|
||||
existing = existing[[c for c in existing.columns if not c.startswith(SP_PREFIX)]]
|
||||
merged = pd.merge(existing, feats, on="t", how="left", suffixes=("", "_dup"))
|
||||
merged = merged.loc[:, ~merged.columns.str.endswith("_dup")]
|
||||
# keep original column order + new sp_* appended
|
||||
merged.to_parquet(feats_path, index=False)
|
||||
else:
|
||||
feats.to_parquet(feats_path, index=False)
|
||||
print(f"persisted {symbol}: {len(feats.columns) - 1} sp_* features")
|
||||
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--market", default="US")
|
||||
ap.add_argument("--timeframe", default="1d")
|
||||
ap.add_argument("--symbols", default="", help="comma-separated; default: all lake symbols")
|
||||
args = ap.parse_args()
|
||||
lake_root = os.environ.get("TAC_LAKE_DIR")
|
||||
if not lake_root:
|
||||
raise SystemExit("TAC_LAKE_DIR is required")
|
||||
lake = LakeConfig(lake_root, args.market)
|
||||
symbols = [s.strip().upper() for s in args.symbols.split(",") if s.strip()] or lake.load_symbols()
|
||||
for symbol in symbols:
|
||||
persist_symbol(lake, args.timeframe, symbol)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,69 @@
|
||||
"""Minimal OptimalStopControl threshold calibration — valid-window grid search.
|
||||
|
||||
This repo found OptimalStopControl thresholds overfit the valid window (valid
|
||||
+7.5% → test −17.7% on one calibration). This script runs a small grid over
|
||||
(entry_pct, exit_pct, max_hold_days) on the VALID window, reports per-config
|
||||
excess return + turnover, and warns when the best valid config is a spike.
|
||||
|
||||
Reference repo impl: tac-qlib/examples/run_optstop_compare.py.
|
||||
|
||||
python examples/run_optstop_compare.py --universe AAPL,MSFT,QQQ
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import itertools
|
||||
import os
|
||||
|
||||
import pandas as pd
|
||||
|
||||
|
||||
GRID = {
|
||||
"entry_pct": [0.7, 0.85, 0.95],
|
||||
"exit_pct": [0.5, 0.7],
|
||||
"max_hold_days": [5, 10],
|
||||
}
|
||||
|
||||
|
||||
def evaluate_config(lake_root: str, universe: list[str], window: tuple, config: dict) -> dict:
|
||||
"""Simplified stand-in: train the RankIC model, backtest OptimalStopControl
|
||||
on `window`, return (ann_excess_return, turnover, max_drawdown).
|
||||
|
||||
The real repo impl calls qlib.backtest with the strategy and reads
|
||||
report_normal.csv + risk.csv. Keep the interface here so the grid loop is
|
||||
reusable.
|
||||
"""
|
||||
# placeholder — plug in the real backtest here
|
||||
return {"ann_excess": 0.0, "turnover": 0.0, "max_dd": 0.0}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--lake-root", default=os.environ.get("TAC_LAKE_DIR", ""))
|
||||
ap.add_argument("--universe", default="AAPL,MSFT,QQQ,IVV,SMH,TLT")
|
||||
args = ap.parse_args()
|
||||
universe = [s.strip().upper() for s in args.universe.split(",")]
|
||||
valid = ("2026-06-01", "2026-06-30")
|
||||
test = ("2026-07-01", "2026-08-06")
|
||||
|
||||
keys = list(GRID)
|
||||
results = []
|
||||
for combo in itertools.product(*[GRID[k] for k in keys]):
|
||||
config = dict(zip(keys, combo))
|
||||
v = evaluate_config(args.lake_root, universe, valid, config)
|
||||
t = evaluate_config(args.lake_root, universe, test, config)
|
||||
results.append({**config, "valid_excess": v["ann_excess"], "test_excess": t["ann_excess"]})
|
||||
|
||||
df = pd.DataFrame(results).sort_values("valid_excess", ascending=False)
|
||||
print(df.head(10).to_string(index=False))
|
||||
# Overfit check: how far is the best-valid config from the median test config?
|
||||
med = df["test_excess"].median()
|
||||
best = df.iloc[0]
|
||||
print(f"\nmedian test excess: {med:+.3f} | best-valid test excess: {best['test_excess']:+.3f}")
|
||||
if abs(best["test_excess"] - med) > 0.10:
|
||||
print("WARNING: best-valid config is an outlier on test — likely overfit, prefer robust defaults")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,90 @@
|
||||
"""Minimal ranking-objective ablation loop — why lambda_rank fails here.
|
||||
|
||||
This repo found that with only ~50 instruments per day the rank-gradient
|
||||
objectives (lambdarank / rank_xendcg) produce near-zero RankIC, while MSE
|
||||
objective + RankIC early-stop is the winner. This script replays that check by
|
||||
training a few LightGBM variants on the same lake split and printing RankIC.
|
||||
|
||||
Reference repo impl: tac-qlib/examples/run_rank_objectives.py.
|
||||
|
||||
python examples/run_rank_objectives.py --universe AAPL,MSFT,QQQ
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import os
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
OBJECTIVES = ["mse", "lambdarank", "rank_xendcg"]
|
||||
|
||||
|
||||
def load_frame(lake_root: str, universe: list[str], start: str, end: str) -> pd.DataFrame:
|
||||
"""Stack lake bars into a qlib-like (datetime, instrument) frame."""
|
||||
from tac_qlib.data.config import LakeConfig
|
||||
|
||||
lake = LakeConfig(lake_root, "US")
|
||||
frames = []
|
||||
for sym in universe:
|
||||
p = lake.bar_path("1d", sym)
|
||||
if p.exists():
|
||||
df = pd.read_parquet(p)[["t", "c"]].rename(columns={"t": "datetime", "c": "close"})
|
||||
df["instrument"] = sym
|
||||
frames.append(df)
|
||||
out = pd.concat(frames, ignore_index=True)
|
||||
out["datetime"] = pd.to_datetime(out["datetime"])
|
||||
out = out[(out["datetime"] >= start) & (out["datetime"] <= end)]
|
||||
return out.set_index(["datetime", "instrument"])
|
||||
|
||||
|
||||
def label_5d(frame: pd.DataFrame) -> pd.Series:
|
||||
close = frame["close"].unstack()
|
||||
lbl = close.shift(-6) / close.shift(-1) - 1
|
||||
return lbl.stack().rename("label")
|
||||
|
||||
|
||||
def train_one(lake_root: str, universe: list[str], objective: str, train: tuple, test: tuple):
|
||||
import lightgbm as lgb
|
||||
|
||||
frame = load_frame(lake_root, universe, train[0], test[1])
|
||||
label = label_5d(frame)
|
||||
data = pd.concat([frame["close"], label], axis=1).dropna()
|
||||
|
||||
tr = data.loc[(data.index.get_level_values(0) >= train[0]) & (data.index.get_level_values(0) <= train[1])]
|
||||
te = data.loc[(data.index.get_level_values(0) >= test[0]) & (data.index.get_level_values(0) <= test[1])]
|
||||
|
||||
dtrain = lgb.Dataset(tr[["close"]], label=tr["label"])
|
||||
dtest = lgb.Dataset(te[["close"]], label=te["label"])
|
||||
params = {"objective": objective, "learning_rate": 0.05, "num_leaves": 15, "verbosity": -1}
|
||||
model = lgb.train(params, dtrain, num_boost_round=100, valid_sets=[dtest])
|
||||
|
||||
pred = model.predict(te[["close"]], num_iteration=model.best_iteration)
|
||||
label_te = te["label"].to_numpy()
|
||||
# per-day RankIC
|
||||
days = te.index.get_level_values(0).unique()
|
||||
ics = []
|
||||
for d in days:
|
||||
m = te.index.get_level_values(0) == d
|
||||
if m.sum() >= 3:
|
||||
ics.append(pd.Series(pred[m]).rank().corr(pd.Series(label_te[m]).rank()))
|
||||
return float(np.nanmean(ics))
|
||||
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--lake-root", default=os.environ.get("TAC_LAKE_DIR", ""))
|
||||
ap.add_argument("--universe", default="AAPL,MSFT,QQQ,IVV,SMH,TLT")
|
||||
args = ap.parse_args()
|
||||
universe = [s.strip().upper() for s in args.universe.split(",")]
|
||||
train = ("2026-03-01", "2026-05-31")
|
||||
test = ("2026-07-01", "2026-08-06")
|
||||
print(f"{'objective':<14}{'test RankIC':>12}")
|
||||
for obj in OBJECTIVES:
|
||||
ic = train_one(args.lake_root, universe, obj, train, test)
|
||||
print(f"{obj:<14}{ic:>12.4f}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,80 @@
|
||||
"""Minimal stochastic-process feature computation — OU mean-reversion + Hurst.
|
||||
|
||||
These are the features that beat hand-rolled TA in this repo's 50-ETF runs.
|
||||
Compute them per symbol on a rolling window ENDING at each day (never let them
|
||||
see test data at fit time — see the HMM/GARCH note in SKILL.md).
|
||||
|
||||
Reference repo impl: tac-qlib/examples/sp_features.py (full 55-feature set:
|
||||
OU, HMM, jump, HARRV, trend, GARCH, Hurst, path signatures, entropy, catch22).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
#: rolling window for feature computation (days)
|
||||
LOOKBACK = 250
|
||||
|
||||
|
||||
def compute_ou_features(close: pd.Series) -> pd.DataFrame:
|
||||
"""Ornstein-Uhlenbeck fit: theta (reversion speed), sigma (vol), residual z.
|
||||
|
||||
OU: dx_t = theta (mu - x_t) dt + sigma dW_t (theta is the mean-reversion
|
||||
speed; higher = faster reversion = tradable mean-reversion signal).
|
||||
|
||||
Rolling OLS of dx on lagged log-price gives theta = -b (reversion speed);
|
||||
sigma is the residual std. Vectorized via rolling cov/var.
|
||||
"""
|
||||
logp = np.log(close)
|
||||
dx = logp.diff()
|
||||
x_prev = logp.shift(1)
|
||||
df = pd.DataFrame({"dx": dx, "x": x_prev})
|
||||
|
||||
out = pd.DataFrame(index=close.index, dtype=float)
|
||||
cov = df["dx"].rolling(LOOKBACK, min_periods=30).cov(df["x"])
|
||||
var = df["x"].rolling(LOOKBACK, min_periods=30).var()
|
||||
theta = (-cov / var).rename("sp_ou_theta")
|
||||
out["sp_ou_theta"] = theta
|
||||
out["sp_ou_sigma"] = df["dx"].rolling(LOOKBACK, min_periods=30).std()
|
||||
# standardized residual z = (x - mu) / sigma of the fitted process
|
||||
mu = df["x"].rolling(LOOKBACK, min_periods=30).mean()
|
||||
scale = np.sqrt(np.clip(1 / (2 * theta + 1e-9), 0, None))
|
||||
out["sp_ou_zscore"] = (df["x"] - mu) / (out["sp_ou_sigma"] * scale)
|
||||
return out
|
||||
|
||||
|
||||
def compute_hurst(close: pd.Series, lookback: int = 100) -> pd.Series:
|
||||
"""Rolling Hurst exponent via rescaled range (R/S). H>0.5 = trending."""
|
||||
def _hurst(x: np.ndarray) -> float:
|
||||
if len(x) < 20:
|
||||
return np.nan
|
||||
lags = range(2, min(len(x) // 2, 50))
|
||||
tau = []
|
||||
for lag in lags:
|
||||
diff = x[lag:] - x[:-lag]
|
||||
tau.append(np.sqrt(np.std(diff)))
|
||||
tau = np.array(tau)
|
||||
lags = np.array(lags, dtype=float)
|
||||
poly = np.polyfit(np.log(lags), np.log(tau), 1)
|
||||
return float(poly[0])
|
||||
|
||||
return close.rolling(lookback, min_periods=20).apply(lambda w: _hurst(w.to_numpy()), raw=False).rename(
|
||||
"sp_hurst_exponent"
|
||||
)
|
||||
|
||||
|
||||
def build_sp_features(bars: pd.DataFrame) -> pd.DataFrame:
|
||||
"""bars: lake 1d bars indexed by (datetime, instrument) or a symbol frame."""
|
||||
if isinstance(bars.index, pd.MultiIndex):
|
||||
frames = []
|
||||
for inst, sub in bars.groupby(level=1):
|
||||
close = sub.droplevel(1)["close"]
|
||||
feats = pd.concat([compute_ou_features(close), compute_hurst(close)], axis=1)
|
||||
feats["instrument"] = inst
|
||||
frames.append(feats.reset_index())
|
||||
out = pd.concat(frames).set_index(["datetime", "instrument"])
|
||||
else:
|
||||
close = bars["close"]
|
||||
out = pd.concat([compute_ou_features(close), compute_hurst(close)], axis=1)
|
||||
return out
|
||||
@@ -0,0 +1,82 @@
|
||||
"""Minimal beta-neutral 3L/3S strategy + record — stub of tac_qlib/contrib/strategy/beta_neutral.py.
|
||||
|
||||
Strategy side: subclass BaseSignalStrategy, hold ~3 long + 3 short equally
|
||||
weighted (dollar-neutral) with TP/SL and a hard close at the horizon. The beta
|
||||
comes from regression of daily returns on the benchmark in `_prepare_betas`.
|
||||
|
||||
Record side (BetaNeutralRecord): a custom `Record` that simulates the 3L/3S
|
||||
portfolio after training and logs report / trades / risk.csv into the MLflow
|
||||
run — the pattern to follow for any custom Record.
|
||||
|
||||
Wire the record into the workflow YAML:
|
||||
|
||||
record:
|
||||
- class: BetaNeutralRecord
|
||||
module_path: tac_qlib.contrib.strategy.beta_neutral
|
||||
kwargs: { benchmark: QQQ, n_long: 3, n_short: 3 }
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, List
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from qlib.backtest import Order
|
||||
from qlib.backtest.decision import OrderDir, TradeDecisionWO
|
||||
from qlib.contrib.strategy.signal_strategy import BaseSignalStrategy
|
||||
|
||||
|
||||
class BetaNeutralStrategy(BaseSignalStrategy):
|
||||
"""3 long / 3 short dollar-neutral template with TP/SL and hard close."""
|
||||
|
||||
def __init__(self, *, n_long: int = 3, n_short: int = 3, tp: float = 0.06, sl: float = -0.05, **kwargs: Any):
|
||||
super().__init__(**kwargs)
|
||||
self.n_long = n_long
|
||||
self.n_short = n_short
|
||||
self.tp = tp
|
||||
self.sl = sl
|
||||
|
||||
def generate_trade_decision(self, execute_result=None):
|
||||
trade_step = self.trade_calendar.get_trade_step()
|
||||
start_time, end_time = self.trade_calendar.get_step_time(trade_step)
|
||||
pred_start, pred_end = self.trade_calendar.get_step_time(trade_step - 1)
|
||||
pred = self.signal.get_signal(start_time=pred_start, end_time=pred_end)
|
||||
|
||||
orders: List[Order] = []
|
||||
if pred is not None and len(pred):
|
||||
daily = pred.groupby(level=0).mean().iloc[-1].dropna().sort_values()
|
||||
longs = daily.tail(self.n_long).index.tolist()
|
||||
shorts = daily.head(self.n_short).index.tolist()
|
||||
for inst in longs:
|
||||
orders.append(self._order(inst, 1, start_time, end_time))
|
||||
for inst in shorts:
|
||||
orders.append(self._order(inst, -1, start_time, end_time))
|
||||
return TradeDecisionWO(orders, self)
|
||||
|
||||
def _order(self, inst, direction, start_time, end_time):
|
||||
price = self.trade_exchange.get_close(inst, end_time) or 1.0
|
||||
qty = int(self.trade_exchange.account.cash / (len(self.trade_exchange.get_positions()) + 1) / price)
|
||||
return Order(
|
||||
inst,
|
||||
qty,
|
||||
start_time,
|
||||
end_time,
|
||||
direction=OrderDir.BUY if direction > 0 else OrderDir.SELL,
|
||||
type="market",
|
||||
)
|
||||
|
||||
|
||||
class BetaNeutralRecord: # subclass qlib.workflow.record_temp.Record in the real impl
|
||||
"""Custom record that backtests 3L/3S and logs report/trades/risk.csv."""
|
||||
|
||||
def __init__(self, *, benchmark: str = "QQQ", n_long: int = 3, n_short: int = 3, **_: Any):
|
||||
self.benchmark = benchmark
|
||||
self.n_long = n_long
|
||||
self.n_short = n_short
|
||||
|
||||
def generate(self, **kwargs):
|
||||
# Real impl: run qlib.backtest with BetaNeutralStrategy on the recorded
|
||||
# pred, write report_normal.csv / positions_normal.csv / risk.csv into
|
||||
# the current MLflow run's artifact dir, then log the headline metrics.
|
||||
print("BetaNeutralRecord.generate: simulate 3L/3S and log artifacts")
|
||||
@@ -0,0 +1,77 @@
|
||||
"""Minimal OptimalStopControl strategy — a stub of tac_qlib/contrib/strategy/optimal_stop.py.
|
||||
|
||||
Subclasses qlib's BaseSignalStrategy; override `generate_trade_decision` to build
|
||||
`qlib.backtest.Order`s and return a `TradeDecisionWO`. The real implementation
|
||||
gates entry by cross-sectional signal percentile, exits by percentile / time /
|
||||
stop-loss, and sizes equal-weight with `risk_degree` control.
|
||||
|
||||
Wire into a workflow YAML under PortAnaRecord.config.strategy:
|
||||
|
||||
strategy:
|
||||
class: OptimalStopControl
|
||||
module_path: tac_qlib.contrib.strategy.optimal_stop
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 10
|
||||
entry_pct: 0.85
|
||||
exit_pct: 0.7
|
||||
max_hold_days: 10
|
||||
min_hold_days: 2
|
||||
sl: -0.08
|
||||
risk_degree: 0.95
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
import numpy as np
|
||||
|
||||
from qlib.backtest import Order
|
||||
from qlib.backtest.decision import OrderDir, TradeDecisionWO
|
||||
from qlib.contrib.strategy.signal_strategy import BaseSignalStrategy
|
||||
|
||||
|
||||
class OptimalStopControl(BaseSignalStrategy):
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
topk: int = 10,
|
||||
entry_pct: float = 0.85,
|
||||
exit_pct: float = 0.7,
|
||||
max_hold_days: int = 10,
|
||||
min_hold_days: int = 2,
|
||||
sl: float = -0.08,
|
||||
risk_degree: float = 0.95,
|
||||
**kwargs: Any,
|
||||
):
|
||||
super().__init__(**kwargs)
|
||||
self.topk = topk
|
||||
self.entry_pct = entry_pct
|
||||
self.exit_pct = exit_pct
|
||||
self.max_hold_days = max_hold_days
|
||||
self.min_hold_days = min_hold_days
|
||||
self.sl = sl
|
||||
self.risk_degree = risk_degree
|
||||
|
||||
def generate_trade_decision(self, execute_result=None):
|
||||
"""Build orders for one trade step (minimal sketch — see repo impl)."""
|
||||
trade_step = self.trade_calendar.get_trade_step()
|
||||
# signal is known at t-1 via shift=-1 in the signal object
|
||||
start_time, end_time = self.trade_calendar.get_step_time(trade_step)
|
||||
pred_start, pred_end = self.trade_calendar.get_step_time(trade_step - 1)
|
||||
pred = self.signal.get_signal(start_time=pred_start, end_time=pred_end)
|
||||
|
||||
orders: List[Order] = []
|
||||
if pred is not None and len(pred):
|
||||
# take the top-k by cross-sectional percentile, equal-weight size
|
||||
cross = pred.groupby(level=0).rank(pct=True) # 0..1 per day
|
||||
keep = pred.index[cross >= 1.0 - self.entry_pct]
|
||||
for inst, (dt, _instr) in zip(keep, keep):
|
||||
price = self.trade_exchange.get_close(inst, end_time) or 1.0
|
||||
qty = int((self.risk_degree * self.trade_exchange.account.cash) / (self.topk * price))
|
||||
if qty > 0:
|
||||
orders.append(
|
||||
Order(inst, qty, start_time, end_time, direction=OrderDir.BUY, type="market")
|
||||
)
|
||||
return TradeDecisionWO(orders, self)
|
||||
@@ -0,0 +1,107 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# MINIMAL workflow — the canonical "run a backtest" template for the skill.
|
||||
#
|
||||
# Every traced backtest runs through a workflow YAML like this one via
|
||||
# rd_run_workflow, so the `record` blocks write MLflow artifacts to disk
|
||||
# (<lake>/mlruns/<exp_id>/<run_id>). The traced experiment's ref id IS the
|
||||
# mlflow run id returned by rd_run_workflow.
|
||||
#
|
||||
# Trigger:
|
||||
# rd_run_workflow config_path=examples/workflow_minimal.yaml \
|
||||
# experiment_name=tac-rd-minimal
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-minimal" }
|
||||
|
||||
task:
|
||||
model:
|
||||
class: LGBModel
|
||||
module_path: qlib.contrib.model.gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.05
|
||||
num_leaves: 15
|
||||
n_estimators: 200
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.01
|
||||
reg_lambda: 0.01
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: AAPL,MSFT,QQQ,IVV,SMH,TLT
|
||||
start_time: 2026-03-01
|
||||
end_time: 2026-08-06
|
||||
fit_start_time: 2026-03-01
|
||||
fit_end_time: 2026-05-31
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
segments:
|
||||
train: [2026-03-01, 2026-05-31]
|
||||
valid: [2026-06-01, 2026-06-30]
|
||||
test: [2026-07-01, 2026-08-06]
|
||||
|
||||
# Record block — REQUIRED. Each entry writes one artifact family to mlruns:
|
||||
# SignalRecord pred.pkl + label.pkl
|
||||
# SigAnaRecord IC / Rank IC series + long-short group returns
|
||||
# PortAnaRecord backtest report / positions / risk
|
||||
record:
|
||||
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: { ana_long_short: true, ann_scaler: 252 }
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 2
|
||||
n_drop: 1
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-07-01
|
||||
end_time: 2026-08-06
|
||||
account: 1000000
|
||||
benchmark: QQQ
|
||||
exchange_kwargs:
|
||||
codes: AAPL,MSFT,QQQ,IVV,SMH,TLT
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
@@ -0,0 +1,113 @@
|
||||
# -----------------------------------------------------------------------------
|
||||
# RankIC early-stop workflow — minimal example wiring the custom model.
|
||||
#
|
||||
# model_rank_gbdt.py must be importable: copy it (or symlink) into
|
||||
# tac_qlib/contrib/model/ and sync to /opt/venv site-packages (see SKILL.md
|
||||
# "Installed package copy" gotcha). Then run:
|
||||
#
|
||||
# rd_run_workflow config_path=examples/workflow_rankic.yaml \
|
||||
# experiment_name=tac-rd-rankic
|
||||
# -----------------------------------------------------------------------------
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-rankic" }
|
||||
|
||||
task:
|
||||
model:
|
||||
# Custom model — see examples/model_rank_gbdt.py (RankICLGBModel):
|
||||
# per-day query groups + feval=rankic + metric='None' so early-stopping
|
||||
# tracks mean per-day Spearman instead of l2.
|
||||
class: RankICLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_gbdt
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 15
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l1: 0.0
|
||||
lambda_l2: 0.5
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
seed: 2026
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: AAPL,MSFT,QQQ,IVV,SMH,TLT
|
||||
start_time: 2026-03-01
|
||||
end_time: 2026-08-06
|
||||
fit_start_time: 2026-03-01
|
||||
fit_end_time: 2026-05-31
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
infer_processors:
|
||||
- { class: DropAllNaN, kwargs: {} }
|
||||
- { class: ProcessInf, kwargs: {} }
|
||||
- { class: CSRankNorm, kwargs: {} }
|
||||
- { class: ZScoreNorm, kwargs: {} }
|
||||
- { class: Fillna, kwargs: {} }
|
||||
segments:
|
||||
train: [2026-03-01, 2026-05-31]
|
||||
valid: [2026-06-01, 2026-06-30]
|
||||
test: [2026-07-01, 2026-08-06]
|
||||
|
||||
record:
|
||||
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
|
||||
- class: SigAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs: { ana_long_short: true, ann_scaler: 252 }
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs:
|
||||
signal: "<PRED>"
|
||||
topk: 2
|
||||
n_drop: 1
|
||||
only_tradable: true
|
||||
risk_degree: 0.95
|
||||
backtest:
|
||||
start_time: 2026-07-01
|
||||
end_time: 2026-08-06
|
||||
account: 1000000
|
||||
benchmark: QQQ
|
||||
exchange_kwargs:
|
||||
codes: AAPL,MSFT,QQQ,IVV,SMH,TLT
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
Reference in New Issue
Block a user