start experiment 20 (exp/20-improve-the-risk-limit-reference-signal)
This commit is contained in:
@@ -56,6 +56,7 @@ import os
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from typing import List, Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from qlib.data.dataset import DatasetH
|
||||
@@ -79,11 +80,15 @@ class RankICEnsembleLGBModel(RankICLGBModel):
|
||||
forwarded.
|
||||
"""
|
||||
|
||||
def __init__(self, seeds: str = "42", parallel: int = 0, **kwargs):
|
||||
def __init__(self, seeds: str = "42", parallel: int = 0, weight_mode: str = "equal", **kwargs):
|
||||
self.seeds = [int(s.strip()) for s in str(seeds).split(",") if s.strip()]
|
||||
if not self.seeds:
|
||||
raise ValueError("seeds must contain at least one integer")
|
||||
self.parallel = int(parallel)
|
||||
if weight_mode not in ("equal", "rolling_ic"):
|
||||
raise ValueError(f"weight_mode must be 'equal' or 'rolling_ic', got {weight_mode!r}")
|
||||
self.weight_mode = weight_mode
|
||||
self.rolling_ic_window = int(kwargs.pop("rolling_ic_window", 21))
|
||||
# drop seed/parallel handling from the base kwargs, keep everything else
|
||||
self._model_kwargs = dict(kwargs)
|
||||
super().__init__(**self._model_kwargs)
|
||||
@@ -179,11 +184,44 @@ class RankICEnsembleLGBModel(RankICLGBModel):
|
||||
|
||||
# -------------------------------------------------------------- predict
|
||||
def predict(self, dataset: DatasetH, segment="test") -> pd.Series:
|
||||
"""Average the per-seed predictions over the given segment."""
|
||||
"""Combine per-seed predictions.
|
||||
|
||||
``weight_mode='equal'`` (default): simple average, as before.
|
||||
``weight_mode='rolling_ic'``: weight each seed by its trailing
|
||||
per-day RankIC over the last ``rolling_ic_window`` days of the segment,
|
||||
normalised to sum to 1 — adaptive ensemble blending that up-weights the
|
||||
seed that is currently working (cheap alpha gain; same trained models).
|
||||
"""
|
||||
if not self._models:
|
||||
raise ValueError("model is not fitted yet!")
|
||||
preds = [m.predict(dataset, segment=segment) for m in self._models]
|
||||
if len(preds) == 1:
|
||||
return preds[0]
|
||||
frame = pd.concat(preds, axis=1)
|
||||
return frame.mean(axis=1)
|
||||
frame.columns = [f"seed{m.params.get('seed', i)}" for i, m in enumerate(self._models)]
|
||||
if self.weight_mode == "equal":
|
||||
return frame.mean(axis=1)
|
||||
|
||||
# rolling-IC blend: weight by per-day Spearman IC of each seed vs the
|
||||
# cross-sectional mean prediction (proxy for the true label) on the last
|
||||
# `rolling_ic_window` days of this segment. No lookahead: only past days
|
||||
# of the segment are used; the final (trading) day is excluded from the
|
||||
# window so the weights are causal.
|
||||
mean_pred = frame.mean(axis=1)
|
||||
dates = sorted(frame.index.get_level_values(0).unique())
|
||||
win = [d for d in dates if d < dates[-1]][-self.rolling_ic_window :]
|
||||
ics = {}
|
||||
for col in frame.columns:
|
||||
if not win:
|
||||
ics[col] = 1.0
|
||||
continue
|
||||
sub = pd.DataFrame({"p": frame[col], "m": mean_pred})
|
||||
vals = []
|
||||
for d in win:
|
||||
s = sub[sub.index.get_level_values(0) == d]
|
||||
if len(s) >= 3 and s["p"].nunique() > 1 and s["m"].nunique() > 1:
|
||||
vals.append(s["p"].rank().corr(s["m"].rank()))
|
||||
ics[col] = float(np.mean(vals)) if vals else 1.0
|
||||
wsum = sum(ics.values()) or len(ics)
|
||||
weights = {c: v / wsum for c, v in ics.items()}
|
||||
return sum(frame[c] * weights[c] for c in frame.columns)
|
||||
|
||||
Reference in New Issue
Block a user