79 lines
2.9 KiB
Python
79 lines
2.9 KiB
Python
"""Minimal RankIC early-stopping LightGBM model (the biggest IC/backtest lever).
|
|
|
|
Drop-in replacement for `qlib.contrib.model.gbdt.LGBModel` in a workflow YAML:
|
|
|
|
task.model:
|
|
class: RankICLGBModel
|
|
module_path: tac_qlib.contrib.model.rank_gbdt
|
|
kwargs: { loss: mse, learning_rate: 0.02, num_boost_round: 3000,
|
|
early_stopping_rounds: 200, lambda_l2: 0.5 }
|
|
|
|
What it changes vs stock LGBModel:
|
|
* `_prepare_data` builds `lgb.Dataset` with per-day `group` query groups, so
|
|
metrics are computed per trading day.
|
|
* `fit` injects `feval=rankic_feval` (mean per-day Spearman) into `lgb.train`
|
|
and forces `metric='None'` + `first_metric_only=True` so early stopping
|
|
tracks RankIC — not l2, which keeps improving after RankIC peaks.
|
|
|
|
Why: with ~50 instruments per day, ranking objectives (lambda_rank/xendcg)
|
|
produce near-zero RankIC; MSE objective + RankIC early-stop is what lifts it.
|
|
|
|
Install: copy to tac_qlib/contrib/model/rank_gbdt.py AND the /opt/venv copy.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from typing import Any, Dict
|
|
|
|
import numpy as np
|
|
import pandas as pd
|
|
|
|
from qlib.contrib.model.gbdt import LGBModel
|
|
|
|
|
|
def rankic_feval(preds: np.ndarray, dataset) -> tuple[str, float, bool]:
|
|
"""Mean per-day Spearman rank IC between predictions and the label."""
|
|
label = dataset.get_label()
|
|
group = dataset.get_group() if hasattr(dataset, "get_group") else None
|
|
if group is None:
|
|
return "rankic", _spearman(preds, label), False
|
|
|
|
start = 0
|
|
ics = []
|
|
for g in group:
|
|
sl = slice(start, start + g)
|
|
start += g
|
|
ics.append(_spearman(preds[sl], label[sl]))
|
|
return "rankic", float(np.mean(ics)), False
|
|
|
|
|
|
def _spearman(x: np.ndarray, y: np.ndarray) -> float:
|
|
if len(x) < 2:
|
|
return 0.0
|
|
from scipy.stats import spearmanr
|
|
|
|
rho, _ = spearmanr(x, y)
|
|
return float(rho) if rho == rho else 0.0
|
|
|
|
|
|
class RankICLGBModel(LGBModel):
|
|
"""LGBModel with per-day query groups and RankIC-only early stopping."""
|
|
|
|
def _prepare_data(self, dataset, *args, **kwargs):
|
|
"""Attach per-day group sizes to the train/valid lgb.Dataset."""
|
|
dtrain, dvalid = super()._prepare_data(dataset, *args, **kwargs)
|
|
for d, index in ((dtrain, dataset.get_index_by_segment("train")), (dvalid, dataset.get_index_by_segment("valid"))):
|
|
if d is not None and index is not None:
|
|
# group by calendar day in order
|
|
days = pd.Series([i[0] for i in index])
|
|
group = days.value_counts().sort_index().tolist()
|
|
d.set_group(np.array(group, dtype=np.int32))
|
|
return dtrain, dvalid
|
|
|
|
def fit(self, dataset, evals_result: Dict[str, Any] | None = None, **kwargs):
|
|
# force RankIC-only early stopping
|
|
kwargs.setdefault("feval", rankic_feval)
|
|
kwargs.setdefault("metric", "None")
|
|
kwargs.setdefault("first_metric_only", True)
|
|
return super().fit(dataset, evals_result=evals_result, **kwargs)
|