start experiment 42 (exp/42-q10-hmm-regime-overlay-entry-gate-on-sph)
This commit is contained in:
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -56,7 +56,6 @@ import os
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from typing import List, Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from qlib.data.dataset import DatasetH
|
||||
@@ -80,15 +79,11 @@ class RankICEnsembleLGBModel(RankICLGBModel):
|
||||
forwarded.
|
||||
"""
|
||||
|
||||
def __init__(self, seeds: str = "42", parallel: int = 0, weight_mode: str = "equal", **kwargs):
|
||||
def __init__(self, seeds: str = "42", parallel: int = 0, **kwargs):
|
||||
self.seeds = [int(s.strip()) for s in str(seeds).split(",") if s.strip()]
|
||||
if not self.seeds:
|
||||
raise ValueError("seeds must contain at least one integer")
|
||||
self.parallel = int(parallel)
|
||||
if weight_mode not in ("equal", "rolling_ic"):
|
||||
raise ValueError(f"weight_mode must be 'equal' or 'rolling_ic', got {weight_mode!r}")
|
||||
self.weight_mode = weight_mode
|
||||
self.rolling_ic_window = int(kwargs.pop("rolling_ic_window", 21))
|
||||
# drop seed/parallel handling from the base kwargs, keep everything else
|
||||
self._model_kwargs = dict(kwargs)
|
||||
super().__init__(**self._model_kwargs)
|
||||
@@ -184,44 +179,11 @@ class RankICEnsembleLGBModel(RankICLGBModel):
|
||||
|
||||
# -------------------------------------------------------------- predict
|
||||
def predict(self, dataset: DatasetH, segment="test") -> pd.Series:
|
||||
"""Combine per-seed predictions.
|
||||
|
||||
``weight_mode='equal'`` (default): simple average, as before.
|
||||
``weight_mode='rolling_ic'``: weight each seed by its trailing
|
||||
per-day RankIC over the last ``rolling_ic_window`` days of the segment,
|
||||
normalised to sum to 1 — adaptive ensemble blending that up-weights the
|
||||
seed that is currently working (cheap alpha gain; same trained models).
|
||||
"""
|
||||
"""Average the per-seed predictions over the given segment."""
|
||||
if not self._models:
|
||||
raise ValueError("model is not fitted yet!")
|
||||
preds = [m.predict(dataset, segment=segment) for m in self._models]
|
||||
if len(preds) == 1:
|
||||
return preds[0]
|
||||
frame = pd.concat(preds, axis=1)
|
||||
frame.columns = [f"seed{m.params.get('seed', i)}" for i, m in enumerate(self._models)]
|
||||
if self.weight_mode == "equal":
|
||||
return frame.mean(axis=1)
|
||||
|
||||
# rolling-IC blend: weight by per-day Spearman IC of each seed vs the
|
||||
# cross-sectional mean prediction (proxy for the true label) on the last
|
||||
# `rolling_ic_window` days of this segment. No lookahead: only past days
|
||||
# of the segment are used; the final (trading) day is excluded from the
|
||||
# window so the weights are causal.
|
||||
mean_pred = frame.mean(axis=1)
|
||||
dates = sorted(frame.index.get_level_values(0).unique())
|
||||
win = [d for d in dates if d < dates[-1]][-self.rolling_ic_window :]
|
||||
ics = {}
|
||||
for col in frame.columns:
|
||||
if not win:
|
||||
ics[col] = 1.0
|
||||
continue
|
||||
sub = pd.DataFrame({"p": frame[col], "m": mean_pred})
|
||||
vals = []
|
||||
for d in win:
|
||||
s = sub[sub.index.get_level_values(0) == d]
|
||||
if len(s) >= 3 and s["p"].nunique() > 1 and s["m"].nunique() > 1:
|
||||
vals.append(s["p"].rank().corr(s["m"].rank()))
|
||||
ics[col] = float(np.mean(vals)) if vals else 1.0
|
||||
wsum = sum(ics.values()) or len(ics)
|
||||
weights = {c: v / wsum for c, v in ics.items()}
|
||||
return sum(frame[c] * weights[c] for c in frame.columns)
|
||||
return frame.mean(axis=1)
|
||||
|
||||
@@ -53,23 +53,61 @@ from qlib.workflow import R
|
||||
__all__ = ["RankICLGBModel", "rankic_feval"]
|
||||
|
||||
|
||||
def _group_averaged_rank(values: np.ndarray, gid: np.ndarray, offs: np.ndarray) -> np.ndarray:
|
||||
"""Averaged (tie-corrected) rank of ``values`` within each group, vectorized.
|
||||
|
||||
``gid`` maps each row to its group id; ``offs`` holds the cumulative row
|
||||
offsets so that group ``i`` occupies rows ``[offs[i], offs[i+1])``. Returns
|
||||
the same result as ``pandas.Series.rank(method='average')`` applied per
|
||||
group, but in one pass (``np.lexsort`` is the only non-linear step).
|
||||
"""
|
||||
n = len(values)
|
||||
order = np.lexsort((values, gid))
|
||||
ord_rank = np.empty(n, dtype=np.float64)
|
||||
ord_rank[order] = np.arange(n, dtype=np.float64) - offs[gid[order]] + 1.0
|
||||
sg = gid[order]
|
||||
sv = values[order]
|
||||
newblock = np.empty(n, dtype=bool)
|
||||
newblock[0] = True
|
||||
newblock[1:] = (sg[1:] != sg[:-1]) | (sv[1:] != sv[:-1])
|
||||
blockid = np.cumsum(newblock) - 1
|
||||
block_mean = np.bincount(blockid, weights=ord_rank[order]) / np.bincount(blockid)
|
||||
out = np.empty(n)
|
||||
out[order] = block_mean[blockid]
|
||||
return out
|
||||
|
||||
|
||||
def _per_day_spearman(preds: np.ndarray, labels: np.ndarray, group: np.ndarray) -> float:
|
||||
"""Mean per-day Spearman rank correlation of preds vs labels.
|
||||
|
||||
``group`` holds the number of rows of each trading day (query group), in
|
||||
order. Days with <3 valid rows or a constant pred/label are skipped.
|
||||
|
||||
Vectorized: per-day Spearman == Pearson of the per-day rank transforms,
|
||||
and the Pearson moments (``sum``, ``sum`` of products/squares) aggregate
|
||||
over each day with ``np.bincount``. Runs ~10x faster than the per-day
|
||||
``pd.Series.rank()`` loop that preceded it — this feval is invoked on the
|
||||
train and valid panels every boosting round, per seed.
|
||||
"""
|
||||
if group is None or len(group) == 0:
|
||||
return 0.0
|
||||
offs = np.concatenate([[0], np.cumsum(group.astype(int))])
|
||||
vals = []
|
||||
for i in range(len(group)):
|
||||
s = slice(offs[i], offs[i + 1])
|
||||
p, l = preds[s], labels[s]
|
||||
if len(p) < 3 or np.std(p) == 0 or np.std(l) == 0:
|
||||
continue
|
||||
vals.append(np.corrcoef(pd.Series(p).rank(), pd.Series(l).rank())[0, 1])
|
||||
return float(np.mean(vals)) if vals else 0.0
|
||||
gid = np.repeat(np.arange(len(group)), group.astype(int))
|
||||
rp = _group_averaged_rank(preds, gid, offs)
|
||||
rl = _group_averaged_rank(labels, gid, offs)
|
||||
n_g = group.astype(float)
|
||||
s_p = np.bincount(gid, weights=rp)
|
||||
s_l = np.bincount(gid, weights=rl)
|
||||
s_pl = np.bincount(gid, weights=rp * rl)
|
||||
s_pp = np.bincount(gid, weights=rp * rp)
|
||||
s_ll = np.bincount(gid, weights=rl * rl)
|
||||
cov = n_g * s_pl - s_p * s_l
|
||||
var_p = n_g * s_pp - s_p ** 2
|
||||
var_l = n_g * s_ll - s_l ** 2
|
||||
denom = np.sqrt(var_p * var_l)
|
||||
valid = (n_g >= 3) & (denom > 0)
|
||||
corr = np.where(valid, cov / np.where(denom == 0, 1, denom), 0.0)
|
||||
return float(corr[valid].mean()) if valid.any() else 0.0
|
||||
|
||||
|
||||
def rankic_feval(preds, dataset):
|
||||
|
||||
Reference in New Issue
Block a user