book: scaffold + ch00 (execution trail as spine) — evidence exp 8-31, round 3

This commit is contained in:
TradeAC Book Agent
2026-08-18 22:35:23 +00:00
commit c93424e76c
83 changed files with 17676 additions and 0 deletions
+108
View File
@@ -0,0 +1,108 @@
"""Smoke tests for the tac_qlib contrib package (model/strategy).
Covers the pieces a workflow YAML resolves via ``module_path``:
- ``tac_qlib.contrib.model.rank_gbdt`` -> RankICLGBModel (+ rank feval)
- ``tac_qlib.contrib.strategy.optimal_stop`` -> OptimalStopControl
The strategy smoke test runs a real (tiny) daily backtest through qlib's
executor against the TradeAC lake. The model smoke test checks data
preparation (per-day query groups) + the rank feval without a full fit.
Run::
TAC_LAKE_DIR=/home/data/lake .venv/bin/python tests/test_contrib.py
"""
import os
import sys
import numpy as np
import pandas as pd
sys.path.insert(0, os.path.join(os.path.dirname(__file__), ".."))
LAKE_ROOT = os.environ["TAC_LAKE_DIR"]
CODES = ["AAPL", "MSFT", "NVDA", "GOOGL", "AMZN", "META"]
START, END = "2026-06-01", "2026-07-31"
def _signal(close: pd.DataFrame) -> pd.Series:
"""3-day momentum score indexed (datetime, instrument) covering [START, END]."""
mom = close.pct_change(3).stack()
mom.index = mom.index.set_names(["datetime", "instrument"])
return mom.dropna()
def main():
from tac_qlib.qlib_init import qlib_init
qlib_init(provider_uri=LAKE_ROOT, market="US", freq="day", log_level="WARN")
from qlib.data import D
close = D.features(CODES, ["$close"], START, END, freq="day")["$close"]
close = close.unstack("instrument")
sig = _signal(close)
assert len(sig) > 0, "empty synthetic signal"
print(f"[ok] synthetic signal: {len(sig)} rows, {sig.index.get_level_values(0).nunique()} days")
# ---- OptimalStopControl end-to-end ------------------------------------
from tac_qlib.contrib.strategy.optimal_stop import OptimalStopControl
from qlib.contrib.evaluate import backtest_daily
strat = OptimalStopControl(
signal=sig, topk=2, entry_pct=0.8, exit_pct=0.5,
max_hold_days=5, min_hold_days=1, sl=-0.05, notional=10_000.0,
)
report, positions = backtest_daily(
start_time=START, end_time=END, strategy=strat, account=1_000_000,
benchmark=None,
exchange_kwargs={"codes": CODES, "deal_price": "$close", "freq": "day",
"open_cost": 0.0005, "close_cost": 0.0015, "min_cost": 5.0},
)
assert isinstance(report, pd.DataFrame) and "return" in report and len(report) >= 5
assert not report["return"].isna().all()
print(f"[ok] OptimalStopControl backtest: {len(report)} days, "
f"end equity {float(report['return'].add(1).cumprod().iloc[-1]):.4f}")
# ---- RankICLGBModel: instantiate + _prepare_data (per-day groups) ------
from tac_qlib.contrib.data.handler import TACHandler
from qlib.data.dataset import DatasetH
h = TACHandler(
instruments=CODES, start_time=START, end_time=END,
fit_start_time=START, fit_end_time="2026-06-30", freq="day",
lake_root=LAKE_ROOT, market="US",
label="Ref($close,-6)/Ref($close,-1)-1",
)
ds = DatasetH(
handler=h,
segments={"train": (START, "2026-06-30"), "valid": ("2026-07-01", END)},
)
from tac_qlib.contrib.model.rank_gbdt import RankICLGBModel, rankic_feval
model = RankICLGBModel(loss="mse", learning_rate=0.05, num_leaves=7, n_estimators=50)
data = model._prepare_data(ds)
lgb_ds, names = list(zip(*data))
assert names == ("train", "valid")
groups = lgb_ds[0].get_group()
assert groups is not None and len(groups) >= 5, f"per-day query groups missing: {groups}"
# every group size == number of instruments that day
assert set(groups) <= {len(CODES), len(CODES) - 1}, f"unexpected group sizes {groups}"
print(f"[ok] RankICLGBModel._prepare_data: groups={groups[:5]}... (n_days={len(groups)})")
# rank feval on a hand-built lgb.Dataset
import lightgbm as lgb
y = np.array([1.0, 2.0, 3.0, 3.0, 2.0, 1.0])
preds = np.array([1.0, 2.0, 3.0, 3.0, 2.0, 1.0])
dv = lgb.Dataset(np.zeros((6, 2)), label=y, group=np.array([3, 3]))
name, value, higher = rankic_feval(preds, dv)
assert name == "rankic" and higher is True and abs(value - 1.0) < 1e-9
print(f"[ok] rankic_feval: {name}={value:.4f} (higher_is_better={higher})")
print("\nALL CONTRIB CHECKS PASSED")
if __name__ == "__main__":
main()
+106
View File
@@ -0,0 +1,106 @@
"""Smoke tests: qlib against the TradeAC lake (plain asserts, no pytest needed).
Run::
.venv/bin/python tests/test_lake_providers.py
"""
import os
import sys
import numpy as np
import pandas as pd
sys.path.insert(0, os.path.join(os.path.dirname(__file__), ".."))
# TAC_LAKE_DIR is mandatory (no default fallback). Fail fast if it is missing.
LAKE_ROOT = os.environ["TAC_LAKE_DIR"]
def main():
from tac_qlib.qlib_init import qlib_init
qlib_init(provider_uri=LAKE_ROOT, market="US", freq="day", log_level="WARN")
from qlib.data import D
from qlib.data.data import Cal, ExpressionD, Inst, DatasetD
# ---- calendar ---------------------------------------------------------
cal = Cal.calendar(freq="day")
assert isinstance(cal, (list, np.ndarray)) and len(cal) >= 100, f"calendar too small: {len(cal)}"
print(f"[ok] calendar: {len(cal)} trading days, {pd.Timestamp(cal[0]).date()} -> {pd.Timestamp(cal[-1]).date()}")
# ---- instruments ------------------------------------------------------
inst = Inst.list_instruments({"market": "all"}, start_time=cal[0], end_time=cal[-1], freq="day")
assert len(inst) >= 5, f"expected >=5 instruments, got {inst}"
print(f"[ok] instruments: {sorted(inst)}")
# ---- raw features -----------------------------------------------------
start, end = "2026-03-01", "2026-06-30"
df = D.features(sorted(inst)[:4], ["$close", "$volume", "$vwap"], start, end, freq="day")
assert not df.empty
assert df.columns.tolist() == ["$close", "$volume", "$vwap"]
assert not df["$close"].isna().all()
# index must be the (datetime, instrument) MultiIndex, sorted
assert isinstance(df.index, pd.MultiIndex)
assert df.index.names == [df.index.names[0], df.index.names[1]]
n_rows = len(df)
print(f"[ok] D.features: {len(df)} rows x {len(df.columns)} cols; close sample:\n{df['$close'].head(3)}")
# NaN for fields the lake does not store
df_unk = D.features(sorted(inst)[:2], ["$factor", "$change"], start, end, freq="day")
assert df_unk["$factor"].isna().all() and df_unk["$change"].isna().all()
print("[ok] unknown fields ($factor/$change) are all-NaN")
# ---- expression engine (Option A / qlib defaults) ---------------------
exp = "Ref($close,-2)/$close-1" # same default label as Alpha158
sym = sorted(inst)[0] # use a symbol that is actually in the lake
s = ExpressionD.expression(sym, exp, start_time=start, end_time=end, freq="day")
assert isinstance(s, pd.Series) and len(s) > 0
assert s.notna().any()
print(f"[ok] ExpressionD.expression: {len(s)} values, sample:\n{s.head(3)}")
# a full dataset can be materialised through the expression engine
df_ds = DatasetD.dataset(sorted(inst), [exp], start, end, freq="day")
assert isinstance(df_ds, pd.DataFrame) and len(df_ds) > 0
print(f"[ok] DatasetD.dataset: {df_ds.shape}")
# ---- TACHandler: feature discovery + DropAllNaN ------------------------
from tac_qlib.contrib.data.handler import TACHandler
h = TACHandler(
instruments=sorted(inst)[:6],
start_time="2026-03-01",
end_time="2026-08-06",
fit_start_time="2026-03-01",
fit_end_time="2026-05-31",
freq="day",
lake_root=LAKE_ROOT,
market="US",
)
# the lake's stoch_* columns are fully NaN -> they must be dropped by DropAllNaN
assert not any("stoch" in str(c) for c in h._infer.columns), h._infer.columns.tolist()
# train/valid/test must expose identical feature columns
from qlib.data.dataset import DatasetH
from qlib.data.dataset.handler import DataHandlerLP
ds = DatasetH(
handler=h,
segments={
"train": ("2026-03-01", "2026-05-31"),
"valid": ("2026-06-01", "2026-06-30"),
"test": ("2026-07-01", "2026-08-06"),
},
)
cols = {
seg: ds.prepare(segments=seg, col_set="feature", data_key=DataHandlerLP.DK_I).columns.tolist()
for seg in ("train", "valid", "test")
}
assert cols["train"] == cols["valid"] == cols["test"], cols
print(f"[ok] TACHandler: {len(cols['train'])} features, stoch dropped, segments aligned")
print("\nALL LAKE PROVIDER CHECKS PASSED")
if __name__ == "__main__":
main()