book: scaffold + ch00 (execution trail as spine) — evidence exp 8-31, round 3
This commit is contained in:
@@ -0,0 +1,108 @@
|
||||
"""Smoke tests for the tac_qlib contrib package (model/strategy).
|
||||
|
||||
Covers the pieces a workflow YAML resolves via ``module_path``:
|
||||
|
||||
- ``tac_qlib.contrib.model.rank_gbdt`` -> RankICLGBModel (+ rank feval)
|
||||
- ``tac_qlib.contrib.strategy.optimal_stop`` -> OptimalStopControl
|
||||
|
||||
The strategy smoke test runs a real (tiny) daily backtest through qlib's
|
||||
executor against the TradeAC lake. The model smoke test checks data
|
||||
preparation (per-day query groups) + the rank feval without a full fit.
|
||||
|
||||
Run::
|
||||
|
||||
TAC_LAKE_DIR=/home/data/lake .venv/bin/python tests/test_contrib.py
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(__file__), ".."))
|
||||
|
||||
LAKE_ROOT = os.environ["TAC_LAKE_DIR"]
|
||||
CODES = ["AAPL", "MSFT", "NVDA", "GOOGL", "AMZN", "META"]
|
||||
START, END = "2026-06-01", "2026-07-31"
|
||||
|
||||
|
||||
def _signal(close: pd.DataFrame) -> pd.Series:
|
||||
"""3-day momentum score indexed (datetime, instrument) covering [START, END]."""
|
||||
mom = close.pct_change(3).stack()
|
||||
mom.index = mom.index.set_names(["datetime", "instrument"])
|
||||
return mom.dropna()
|
||||
|
||||
|
||||
def main():
|
||||
from tac_qlib.qlib_init import qlib_init
|
||||
|
||||
qlib_init(provider_uri=LAKE_ROOT, market="US", freq="day", log_level="WARN")
|
||||
from qlib.data import D
|
||||
|
||||
close = D.features(CODES, ["$close"], START, END, freq="day")["$close"]
|
||||
close = close.unstack("instrument")
|
||||
sig = _signal(close)
|
||||
assert len(sig) > 0, "empty synthetic signal"
|
||||
print(f"[ok] synthetic signal: {len(sig)} rows, {sig.index.get_level_values(0).nunique()} days")
|
||||
|
||||
# ---- OptimalStopControl end-to-end ------------------------------------
|
||||
from tac_qlib.contrib.strategy.optimal_stop import OptimalStopControl
|
||||
from qlib.contrib.evaluate import backtest_daily
|
||||
|
||||
strat = OptimalStopControl(
|
||||
signal=sig, topk=2, entry_pct=0.8, exit_pct=0.5,
|
||||
max_hold_days=5, min_hold_days=1, sl=-0.05, notional=10_000.0,
|
||||
)
|
||||
report, positions = backtest_daily(
|
||||
start_time=START, end_time=END, strategy=strat, account=1_000_000,
|
||||
benchmark=None,
|
||||
exchange_kwargs={"codes": CODES, "deal_price": "$close", "freq": "day",
|
||||
"open_cost": 0.0005, "close_cost": 0.0015, "min_cost": 5.0},
|
||||
)
|
||||
assert isinstance(report, pd.DataFrame) and "return" in report and len(report) >= 5
|
||||
assert not report["return"].isna().all()
|
||||
print(f"[ok] OptimalStopControl backtest: {len(report)} days, "
|
||||
f"end equity {float(report['return'].add(1).cumprod().iloc[-1]):.4f}")
|
||||
|
||||
# ---- RankICLGBModel: instantiate + _prepare_data (per-day groups) ------
|
||||
from tac_qlib.contrib.data.handler import TACHandler
|
||||
from qlib.data.dataset import DatasetH
|
||||
|
||||
h = TACHandler(
|
||||
instruments=CODES, start_time=START, end_time=END,
|
||||
fit_start_time=START, fit_end_time="2026-06-30", freq="day",
|
||||
lake_root=LAKE_ROOT, market="US",
|
||||
label="Ref($close,-6)/Ref($close,-1)-1",
|
||||
)
|
||||
ds = DatasetH(
|
||||
handler=h,
|
||||
segments={"train": (START, "2026-06-30"), "valid": ("2026-07-01", END)},
|
||||
)
|
||||
from tac_qlib.contrib.model.rank_gbdt import RankICLGBModel, rankic_feval
|
||||
|
||||
model = RankICLGBModel(loss="mse", learning_rate=0.05, num_leaves=7, n_estimators=50)
|
||||
data = model._prepare_data(ds)
|
||||
lgb_ds, names = list(zip(*data))
|
||||
assert names == ("train", "valid")
|
||||
groups = lgb_ds[0].get_group()
|
||||
assert groups is not None and len(groups) >= 5, f"per-day query groups missing: {groups}"
|
||||
# every group size == number of instruments that day
|
||||
assert set(groups) <= {len(CODES), len(CODES) - 1}, f"unexpected group sizes {groups}"
|
||||
print(f"[ok] RankICLGBModel._prepare_data: groups={groups[:5]}... (n_days={len(groups)})")
|
||||
|
||||
# rank feval on a hand-built lgb.Dataset
|
||||
import lightgbm as lgb
|
||||
|
||||
y = np.array([1.0, 2.0, 3.0, 3.0, 2.0, 1.0])
|
||||
preds = np.array([1.0, 2.0, 3.0, 3.0, 2.0, 1.0])
|
||||
dv = lgb.Dataset(np.zeros((6, 2)), label=y, group=np.array([3, 3]))
|
||||
name, value, higher = rankic_feval(preds, dv)
|
||||
assert name == "rankic" and higher is True and abs(value - 1.0) < 1e-9
|
||||
print(f"[ok] rankic_feval: {name}={value:.4f} (higher_is_better={higher})")
|
||||
|
||||
print("\nALL CONTRIB CHECKS PASSED")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,106 @@
|
||||
"""Smoke tests: qlib against the TradeAC lake (plain asserts, no pytest needed).
|
||||
|
||||
Run::
|
||||
|
||||
.venv/bin/python tests/test_lake_providers.py
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(__file__), ".."))
|
||||
|
||||
# TAC_LAKE_DIR is mandatory (no default fallback). Fail fast if it is missing.
|
||||
LAKE_ROOT = os.environ["TAC_LAKE_DIR"]
|
||||
|
||||
|
||||
def main():
|
||||
from tac_qlib.qlib_init import qlib_init
|
||||
|
||||
qlib_init(provider_uri=LAKE_ROOT, market="US", freq="day", log_level="WARN")
|
||||
|
||||
from qlib.data import D
|
||||
from qlib.data.data import Cal, ExpressionD, Inst, DatasetD
|
||||
|
||||
# ---- calendar ---------------------------------------------------------
|
||||
cal = Cal.calendar(freq="day")
|
||||
assert isinstance(cal, (list, np.ndarray)) and len(cal) >= 100, f"calendar too small: {len(cal)}"
|
||||
print(f"[ok] calendar: {len(cal)} trading days, {pd.Timestamp(cal[0]).date()} -> {pd.Timestamp(cal[-1]).date()}")
|
||||
|
||||
# ---- instruments ------------------------------------------------------
|
||||
inst = Inst.list_instruments({"market": "all"}, start_time=cal[0], end_time=cal[-1], freq="day")
|
||||
assert len(inst) >= 5, f"expected >=5 instruments, got {inst}"
|
||||
print(f"[ok] instruments: {sorted(inst)}")
|
||||
|
||||
# ---- raw features -----------------------------------------------------
|
||||
start, end = "2026-03-01", "2026-06-30"
|
||||
df = D.features(sorted(inst)[:4], ["$close", "$volume", "$vwap"], start, end, freq="day")
|
||||
assert not df.empty
|
||||
assert df.columns.tolist() == ["$close", "$volume", "$vwap"]
|
||||
assert not df["$close"].isna().all()
|
||||
# index must be the (datetime, instrument) MultiIndex, sorted
|
||||
assert isinstance(df.index, pd.MultiIndex)
|
||||
assert df.index.names == [df.index.names[0], df.index.names[1]]
|
||||
n_rows = len(df)
|
||||
print(f"[ok] D.features: {len(df)} rows x {len(df.columns)} cols; close sample:\n{df['$close'].head(3)}")
|
||||
|
||||
# NaN for fields the lake does not store
|
||||
df_unk = D.features(sorted(inst)[:2], ["$factor", "$change"], start, end, freq="day")
|
||||
assert df_unk["$factor"].isna().all() and df_unk["$change"].isna().all()
|
||||
print("[ok] unknown fields ($factor/$change) are all-NaN")
|
||||
|
||||
# ---- expression engine (Option A / qlib defaults) ---------------------
|
||||
exp = "Ref($close,-2)/$close-1" # same default label as Alpha158
|
||||
sym = sorted(inst)[0] # use a symbol that is actually in the lake
|
||||
s = ExpressionD.expression(sym, exp, start_time=start, end_time=end, freq="day")
|
||||
assert isinstance(s, pd.Series) and len(s) > 0
|
||||
assert s.notna().any()
|
||||
print(f"[ok] ExpressionD.expression: {len(s)} values, sample:\n{s.head(3)}")
|
||||
|
||||
# a full dataset can be materialised through the expression engine
|
||||
df_ds = DatasetD.dataset(sorted(inst), [exp], start, end, freq="day")
|
||||
assert isinstance(df_ds, pd.DataFrame) and len(df_ds) > 0
|
||||
print(f"[ok] DatasetD.dataset: {df_ds.shape}")
|
||||
|
||||
# ---- TACHandler: feature discovery + DropAllNaN ------------------------
|
||||
from tac_qlib.contrib.data.handler import TACHandler
|
||||
|
||||
h = TACHandler(
|
||||
instruments=sorted(inst)[:6],
|
||||
start_time="2026-03-01",
|
||||
end_time="2026-08-06",
|
||||
fit_start_time="2026-03-01",
|
||||
fit_end_time="2026-05-31",
|
||||
freq="day",
|
||||
lake_root=LAKE_ROOT,
|
||||
market="US",
|
||||
)
|
||||
# the lake's stoch_* columns are fully NaN -> they must be dropped by DropAllNaN
|
||||
assert not any("stoch" in str(c) for c in h._infer.columns), h._infer.columns.tolist()
|
||||
# train/valid/test must expose identical feature columns
|
||||
from qlib.data.dataset import DatasetH
|
||||
from qlib.data.dataset.handler import DataHandlerLP
|
||||
|
||||
ds = DatasetH(
|
||||
handler=h,
|
||||
segments={
|
||||
"train": ("2026-03-01", "2026-05-31"),
|
||||
"valid": ("2026-06-01", "2026-06-30"),
|
||||
"test": ("2026-07-01", "2026-08-06"),
|
||||
},
|
||||
)
|
||||
cols = {
|
||||
seg: ds.prepare(segments=seg, col_set="feature", data_key=DataHandlerLP.DK_I).columns.tolist()
|
||||
for seg in ("train", "valid", "test")
|
||||
}
|
||||
assert cols["train"] == cols["valid"] == cols["test"], cols
|
||||
print(f"[ok] TACHandler: {len(cols['train'])} features, stoch dropped, segments aligned")
|
||||
|
||||
print("\nALL LAKE PROVIDER CHECKS PASSED")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user