start experiment 34 (exp/34-q02-seed10-10-seed-rankicensemble-vs-ref)
This commit is contained in:
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -53,23 +53,61 @@ from qlib.workflow import R
|
||||
__all__ = ["RankICLGBModel", "rankic_feval"]
|
||||
|
||||
|
||||
def _group_averaged_rank(values: np.ndarray, gid: np.ndarray, offs: np.ndarray) -> np.ndarray:
|
||||
"""Averaged (tie-corrected) rank of ``values`` within each group, vectorized.
|
||||
|
||||
``gid`` maps each row to its group id; ``offs`` holds the cumulative row
|
||||
offsets so that group ``i`` occupies rows ``[offs[i], offs[i+1])``. Returns
|
||||
the same result as ``pandas.Series.rank(method='average')`` applied per
|
||||
group, but in one pass (``np.lexsort`` is the only non-linear step).
|
||||
"""
|
||||
n = len(values)
|
||||
order = np.lexsort((values, gid))
|
||||
ord_rank = np.empty(n, dtype=np.float64)
|
||||
ord_rank[order] = np.arange(n, dtype=np.float64) - offs[gid[order]] + 1.0
|
||||
sg = gid[order]
|
||||
sv = values[order]
|
||||
newblock = np.empty(n, dtype=bool)
|
||||
newblock[0] = True
|
||||
newblock[1:] = (sg[1:] != sg[:-1]) | (sv[1:] != sv[:-1])
|
||||
blockid = np.cumsum(newblock) - 1
|
||||
block_mean = np.bincount(blockid, weights=ord_rank[order]) / np.bincount(blockid)
|
||||
out = np.empty(n)
|
||||
out[order] = block_mean[blockid]
|
||||
return out
|
||||
|
||||
|
||||
def _per_day_spearman(preds: np.ndarray, labels: np.ndarray, group: np.ndarray) -> float:
|
||||
"""Mean per-day Spearman rank correlation of preds vs labels.
|
||||
|
||||
``group`` holds the number of rows of each trading day (query group), in
|
||||
order. Days with <3 valid rows or a constant pred/label are skipped.
|
||||
|
||||
Vectorized: per-day Spearman == Pearson of the per-day rank transforms,
|
||||
and the Pearson moments (``sum``, ``sum`` of products/squares) aggregate
|
||||
over each day with ``np.bincount``. Runs ~10x faster than the per-day
|
||||
``pd.Series.rank()`` loop that preceded it — this feval is invoked on the
|
||||
train and valid panels every boosting round, per seed.
|
||||
"""
|
||||
if group is None or len(group) == 0:
|
||||
return 0.0
|
||||
offs = np.concatenate([[0], np.cumsum(group.astype(int))])
|
||||
vals = []
|
||||
for i in range(len(group)):
|
||||
s = slice(offs[i], offs[i + 1])
|
||||
p, l = preds[s], labels[s]
|
||||
if len(p) < 3 or np.std(p) == 0 or np.std(l) == 0:
|
||||
continue
|
||||
vals.append(np.corrcoef(pd.Series(p).rank(), pd.Series(l).rank())[0, 1])
|
||||
return float(np.mean(vals)) if vals else 0.0
|
||||
gid = np.repeat(np.arange(len(group)), group.astype(int))
|
||||
rp = _group_averaged_rank(preds, gid, offs)
|
||||
rl = _group_averaged_rank(labels, gid, offs)
|
||||
n_g = group.astype(float)
|
||||
s_p = np.bincount(gid, weights=rp)
|
||||
s_l = np.bincount(gid, weights=rl)
|
||||
s_pl = np.bincount(gid, weights=rp * rl)
|
||||
s_pp = np.bincount(gid, weights=rp * rp)
|
||||
s_ll = np.bincount(gid, weights=rl * rl)
|
||||
cov = n_g * s_pl - s_p * s_l
|
||||
var_p = n_g * s_pp - s_p ** 2
|
||||
var_l = n_g * s_ll - s_l ** 2
|
||||
denom = np.sqrt(var_p * var_l)
|
||||
valid = (n_g >= 3) & (denom > 0)
|
||||
corr = np.where(valid, cov / np.where(denom == 0, 1, denom), 0.0)
|
||||
return float(corr[valid].mean()) if valid.any() else 0.0
|
||||
|
||||
|
||||
def rankic_feval(preds, dataset):
|
||||
|
||||
Reference in New Issue
Block a user