"""Diagnose script-vs-workflow gap v3: replicate workflow execution mechanics exactly. Replicates the WeeklyRebalanceDropoutStrategy execution: 1. Weekly rebalance (first trading day of ISO week only) 2. TopkDropout selection: sell bottom n_drop, buy top fill 3. Cash-after-sells sizing: sell first, then cash * risk_degree / len(buy) 4. Whole-share rounding (floor) 5. Asymmetric costs: open_cost=5bp, close_cost=15bp, min_cost=$5 per order 6. Optional SQ gate (hit-rate threshold) Compares against the idealized script (fractional shares, symmetric cost). """ from __future__ import annotations import json, pathlib import numpy as np import pandas as pd LAKE_ROOT = "/home/data/lake" OUT = pathlib.Path("/app/experiments/book/data/diag_script_vs_wf") WINDOWS = [ {"label": "2026", "start": "2026-01-04", "end": "2026-08-19", "pred": f"{LAKE_ROOT}/mlruns/62/3771f96eb1b74365aeae966af7aec5a3/artifacts/pred.pkl"}, {"label": "2025", "start": "2025-01-02", "end": "2025-12-31", "pred": f"{LAKE_ROOT}/mlruns/62/c57c6a8370cc48619d7cdd2bd109b76a/artifacts/pred.pkl"}, {"label": "2024", "start": "2024-01-02", "end": "2024-12-31", "pred": f"{LAKE_ROOT}/mlruns/62/97cf5f282e6f4e699443e38d9bfb40fd/artifacts/pred.pkl"}, {"label": "2023", "start": "2023-01-03", "end": "2023-12-29", "pred": f"{LAKE_ROOT}/mlruns/62/11b9b65ea4e14b3f8ce50d244da0412e/artifacts/pred.pkl"}, {"label": "2021", "start": "2021-01-04", "end": "2021-12-31", "pred": f"{LAKE_ROOT}/mlruns/62/af3034e5910348a382f2ad1e1741f17c/artifacts/pred.pkl"}, ] SYMS = [ "SPY","QQQ","DIA","IWM","MDY","VTI","VOO","VEA","VWO","VT","EFA","EEM", "TLT","IEF","SHY","AGG","BND","LQD","HYG","JNK","EMB","GLD","SLV", "USO","UNG","DBA","DBC","XLK","XLF","XLE","XLV","XLI","XLY","XLP", "XLU","XLB","XLRE","ARKK","SMH","SOXX","IBB","XBI","ITA","XAR", "ICLN","TAN","FDN","IGV","ESPO","REM", ] OPEN_COST = 0.0005 # 5bp CLOSE_COST = 0.0015 # 15bp MIN_COST = 5.0 # $5 minimum per order def load_pred(path): df = pd.read_pickle(path) s = df["score"] if isinstance(df, pd.DataFrame) and "score" in df.columns else df.iloc[:, 0] if isinstance(df, pd.DataFrame) else df idx = s.index new_dt = pd.to_datetime(idx.get_level_values(0)).normalize() s.index = pd.MultiIndex.from_arrays([new_dt, idx.get_level_values(1)], names=idx.names) return s def load_closes(start, end): from tac_qlib.data.config import LakeConfig, resolve_lake_root cfg = LakeConfig(resolve_lake_root(LAKE_ROOT), "US") closes = {} for sym in SYMS: p = cfg.bar_path("1d", sym) if not p.exists(): continue try: df = pd.read_parquet(p) except: continue if not len(df): continue tcol = df["t"] if "t" in df.columns else df["date"] ts = pd.to_datetime(tcol) df = df.assign(_t=ts).set_index("_t").sort_index() df.index = pd.to_datetime(df.index).normalize() warmup = pd.Timestamp(start) - pd.Timedelta(days=60) df = df.loc[warmup:end] if len(df) >= 22: closes[sym] = df["c"] return pd.DataFrame(closes) def load_opens(start, end): from tac_qlib.data.config import LakeConfig, resolve_lake_root cfg = LakeConfig(resolve_lake_root(LAKE_ROOT), "US") opens = {} for sym in SYMS: p = cfg.bar_path("1d", sym) if not p.exists(): continue try: df = pd.read_parquet(p) except: continue if not len(df): continue tcol = df["t"] if "t" in df.columns else df["date"] ts = pd.to_datetime(tcol) df = df.assign(_t=ts).set_index("_t").sort_index() df.index = pd.to_datetime(df.index).normalize() warmup = pd.Timestamp(start) - pd.Timedelta(days=60) df = df.loc[warmup:end] if len(df) >= 22: opens[sym] = df["o"] return pd.DataFrame(opens) def load_vwap(start, end): from tac_qlib.data.config import LakeConfig, resolve_lake_root cfg = LakeConfig(resolve_lake_root(LAKE_ROOT), "US") vwaps = {} for sym in SYMS: p = cfg.bar_path("1d", sym) if not p.exists(): continue try: df = pd.read_parquet(p) except: continue if not len(df): continue tcol = df["t"] if "t" in df.columns else df["date"] ts = pd.to_datetime(tcol) df = df.assign(_t=ts).set_index("_t").sort_index() warmup = pd.Timestamp(start) - pd.Timedelta(days=60) df = df.loc[warmup:end] if len(df) >= 22: vwaps[sym] = df["vw"] return pd.DataFrame(vwaps) def compute_gate(pred, closes, start, end, gate_topk=10, gate_lookback=5, gate_threshold=0.5): """Compute the SQ gate: rolling average hit-rate of topk predictions.""" ret_df = closes.pct_change() ret_df.index = pd.to_datetime(ret_df.index).normalize() dt_idx = pred.index.get_level_values(0) pred_dates = sorted(dt_idx[(dt_idx >= start) & (dt_idx <= end)].unique()) if len(pred_dates) < 2: return pd.Series(True, index=pd.DatetimeIndex(pred_dates)) hit_rates = {} for i in range(1, len(pred_dates)): day = pred_dates[i] prev_day = pred_dates[i - 1] try: prev_scores = pred.loc[prev_day] except KeyError: continue if isinstance(prev_scores, pd.DataFrame): prev_scores = prev_scores.iloc[:, 0] prev_scores = prev_scores.dropna().sort_values(ascending=False) topk_syms = list(prev_scores.index[:gate_topk]) if day not in ret_df.index: continue today_ret = ret_df.loc[day] topk_rets = today_ret.reindex(topk_syms).dropna() if len(topk_rets) == 0: continue hit_rates[day] = (topk_rets > 0).sum() / len(topk_rets) if not hit_rates: return pd.Series(True, index=pd.DatetimeIndex(pred_dates)) hr_series = pd.Series(hit_rates).sort_index() rolling_hr = hr_series.rolling(gate_lookback, min_periods=1).mean() gate = rolling_hr >= gate_threshold gate.iloc[:gate_lookback] = True return gate def strategy_idealized(pred, closes, start, end, topk=10, risk_degree=1.0, cost_bps=0): """Idealized script: fractional shares, symmetric cost, no gate.""" ret_df = closes.pct_change(fill_method=None) ret_df.index = pd.to_datetime(ret_df.index).normalize() dt_idx = pred.index.get_level_values(0) trade_dates = sorted(dt_idx[(dt_idx >= start) & (dt_idx <= end)].unique()) equity = 1_000_000.0 holdings = [] prev_week = None daily_eq = [] for d in trade_dates: try: day_scores = pred.loc[d] except KeyError: daily_eq.append(equity) continue if isinstance(day_scores, pd.DataFrame): day_scores = day_scores.iloc[:, 0] day_scores = day_scores.dropna().sort_values(ascending=False) cur_week = (d.isocalendar()[0], d.isocalendar()[1]) if cur_week != prev_week: new_holdings = list(day_scores.index[:topk]) if holdings and cost_bps > 0: sold = set(holdings) - set(new_holdings) bought = set(new_holdings) - set(holdings) turnover = (len(sold) + len(bought)) / (2 * max(len(holdings), 1)) equity *= (1 - turnover * cost_bps / 10000) holdings = new_holdings ret_row = ret_df.loc[d] if d in ret_df.index else None if ret_row is not None and holdings: wts = np.array([risk_degree / len(holdings)] * len(holdings)) rets = ret_row.reindex(holdings).fillna(0).values equity *= (1 + (wts * rets).sum()) daily_eq.append(equity) prev_week = cur_week return pd.Series(daily_eq, index=trade_dates) def strategy_workflow_exact(pred, closes, opens, start, end, topk=10, n_drop=1, risk_degree=0.95, use_gate=False, gate_series=None): """Exact replication of WeeklyRebalanceDropoutStrategy execution mechanics. - Sells first (all shares of dropped positions) - Sizes buys as: cash * risk_degree / len(buy) - Rounds to whole shares (floor) - Asymmetric costs: open_cost on buys, close_cost on sells, $5 min per order - Tracks position values for daily equity """ ret_df = closes.pct_change(fill_method=None) ret_df.index = pd.to_datetime(ret_df.index).normalize() open_df = opens.copy() open_df.index = pd.to_datetime(open_df.index).normalize() dt_idx = pred.index.get_level_values(0) trade_dates = sorted(dt_idx[(dt_idx >= start) & (dt_idx <= end)].unique()) cash = 1_000_000.0 positions = {} # {sym: num_shares} prev_week = None daily_eq = [] for d in trade_dates: # Skip non-trading days (pred may include weekends) if d not in closes.index: daily_eq.append(daily_eq[-1] if daily_eq else cash) continue try: day_scores = pred.loc[d] except KeyError: daily_eq.append(daily_eq[-1] if daily_eq else cash) continue if isinstance(day_scores, pd.DataFrame): day_scores = day_scores.iloc[:, 0] day_scores = day_scores.dropna().sort_values(ascending=False) cur_week = (d.isocalendar()[0], d.isocalendar()[1]) if cur_week != prev_week: # === REBALANCE DAY === # Check gate if use_gate and gate_series is not None: known = gate_series[gate_series.index <= d] if len(known) and not bool(known.iloc[-1]): # gate closed: sell everything, go to cash for sym in list(positions.keys()): shares = positions[sym] if shares <= 0: continue sell_price = closes.loc[d, sym] if d in closes.index and sym in closes.columns else None if sell_price is None or pd.isna(sell_price): continue trade_val = shares * sell_price trade_cost = max(trade_val * CLOSE_COST, MIN_COST) if trade_val > 0 else 0 cash += trade_val - trade_cost positions[sym] = 0 positions = {s: v for s, v in positions.items() if v > 0} daily_eq.append(cash) prev_week = cur_week continue # TopkDropout selection (matching WeeklyRebalanceDropoutStrategy exactly) current_syms = [s for s, v in positions.items() if v > 0] last = pred.loc[d].reindex(current_syms).sort_values(ascending=False).index if current_syms else pd.Index([]) # buy candidates: top stocks NOT in current holdings, take n_drop + topk - len(last) buy_cands = day_scores[~day_scores.index.isin(last)].sort_values(ascending=False).index buy_list = list(buy_cands[:n_drop + topk - len(last)]) # comb = union of current holdings + buy candidates (actual strategy line 132) comb = pred.loc[d].reindex(last.union(pd.Index(buy_list))).sort_values(ascending=False).index # sell: items from current holdings that are in the bottom n_drop of comb sell_list = list(last[last.isin(comb[-n_drop:])]) if n_drop > 0 and len(comb) >= n_drop else [] # --- SELL FIRST --- for sym in sell_list: if sym not in positions or positions[sym] <= 0: continue shares = positions[sym] sell_price = closes.loc[d, sym] if d in closes.index and sym in closes.columns else None if sell_price is None or pd.isna(sell_price): continue trade_val = shares * sell_price trade_cost = max(trade_val * CLOSE_COST, MIN_COST) if trade_val > 0 else 0 cash += trade_val - trade_cost positions[sym] = 0 # --- BUY --- n_buy = len(buy_list) if n_buy > 0: buy_budget = cash * risk_degree / n_buy for sym in buy_list: buy_price = closes.loc[d, sym] if d in closes.index and sym in closes.columns else None if buy_price is None or pd.isna(buy_price) or buy_price <= 0: continue shares_to_buy = int(buy_budget / buy_price) # floor to whole shares if shares_to_buy <= 0: continue trade_val = shares_to_buy * buy_price trade_cost = max(trade_val * OPEN_COST, MIN_COST) if trade_val > 0 else 0 total_cost = trade_val + trade_cost if total_cost > cash: shares_to_buy = int((cash - MIN_COST) / buy_price) if shares_to_buy <= 0: continue trade_val = shares_to_buy * buy_price trade_cost = max(trade_val * OPEN_COST, MIN_COST) total_cost = trade_val + trade_cost cash -= total_cost positions[sym] = positions.get(sym, 0) + shares_to_buy positions = {s: v for s, v in positions.items() if v > 0} # === DAILY EQUITY === eq = cash if d in closes.index: for sym, shares in positions.items(): if sym in closes.columns: px = closes.loc[d, sym] if not pd.isna(px): eq += shares * px daily_eq.append(eq) prev_week = cur_week return pd.Series(daily_eq, index=trade_dates) def metrics(eq): if len(eq) < 2: return {"ann_ret": 0, "sharpe": 0, "maxDD": 0} rets = eq.pct_change().dropna() ann_ret = float((eq.iloc[-1] / eq.iloc[0]) ** (252 / max(len(eq), 1)) - 1) vol = float(rets.std() * (252 ** 0.5)) if len(rets) > 1 else 0 sharpe = ann_ret / vol if vol > 0 else 0 peak = eq.cummax() dd = (eq - peak) / peak return {"ann_ret": round(ann_ret, 4), "sharpe": round(sharpe, 4), "maxDD": round(float(dd.min()), 4)} def main(): OUT.mkdir(parents=True, exist_ok=True) results = [] for w in WINDOWS: print(f"\n=== {w['label']} ({w['start']} to {w['end']}) ===") pred = load_pred(w["pred"]) closes = load_closes(w["start"], w["end"]) opens = load_opens(w["start"], w["end"]) print(f" pred: {pred.index.get_level_values(0).min().date()} to {pred.index.get_level_values(0).max().date()}, " f"{pred.index.get_level_values(1).nunique()} syms") print(f" close: {closes.index.min().date()} to {closes.index.max().date()}, {closes.shape[1]} syms") row = {"year": w["label"]} # A. Idealized: fractional shares, 10bp symmetric, no gate (diag v2 baseline) eq = strategy_idealized(pred, closes, w["start"], w["end"], topk=10, risk_degree=1.0, cost_bps=0) m = metrics(eq); row["ideal_100_zc"] = m print(f" A. Ideal 100% zc: ann={m['ann_ret']:+.1%} sharpe={m['sharpe']:.2f} maxDD={m['maxDD']:.1%}") # B. Idealized: 95% invested, 10bp symmetric eq = strategy_idealized(pred, closes, w["start"], w["end"], topk=10, risk_degree=0.95, cost_bps=0) m = metrics(eq); row["ideal_95_zc"] = m print(f" B. Ideal 95% zc: ann={m['ann_ret']:+.1%} sharpe={m['sharpe']:.2f} maxDD={m['maxDD']:.1%}") # C. Idealized: 95%, 10bp cost eq = strategy_idealized(pred, closes, w["start"], w["end"], topk=10, risk_degree=0.95, cost_bps=10) m = metrics(eq); row["ideal_95_10bp"] = m print(f" C. Ideal 95% 10bp: ann={m['ann_ret']:+.1%} sharpe={m['sharpe']:.2f} maxDD={m['maxDD']:.1%}") # D. Workflow-exact: whole shares, 5/15bp, $5 min, no gate eq = strategy_workflow_exact(pred, closes, opens, w["start"], w["end"], topk=10, n_drop=1, risk_degree=0.95, use_gate=False) m = metrics(eq); row["wf_exact_95_nogate"] = m print(f" D. WF exact 95% nogate: ann={m['ann_ret']:+.1%} sharpe={m['sharpe']:.2f} maxDD={m['maxDD']:.1%}") # E. Workflow-exact: whole shares, 5/15bp, $5 min, WITH SQ gate gate = compute_gate(pred, closes, w["start"], w["end"], gate_topk=10, gate_lookback=5, gate_threshold=0.5) gate_open_pct = gate.sum() / len(gate) if len(gate) > 0 else 1.0 eq = strategy_workflow_exact(pred, closes, opens, w["start"], w["end"], topk=10, n_drop=1, risk_degree=0.95, use_gate=True, gate_series=gate) m = metrics(eq); row["wf_exact_95_gate"] = m print(f" E. WF exact 95% gate: ann={m['ann_ret']:+.1%} sharpe={m['sharpe']:.2f} maxDD={m['maxDD']:.1%} gate_open={gate_open_pct:.0%}") # Gap analysis ideal = row["ideal_95_zc"]["ann_ret"] wf_nogate = row["wf_exact_95_nogate"]["ann_ret"] wf_gate = row["wf_exact_95_gate"]["ann_ret"] print(f"\n Gap analysis:") print(f" Ideal (fractional, zc) → WF exact (whole shares, 5/15bp, nogate): {ideal:+.1%} → {wf_nogate:+.1%} (gap: {wf_nogate - ideal:+.1%})") print(f" Ideal (fractional, zc) → WF exact (whole shares, 5/15bp, gate): {ideal:+.1%} → {wf_gate:+.1%} (gap: {wf_gate - ideal:+.1%})") results.append(row) with open(OUT / "diagnosis_v3.json", "w") as f: json.dump(results, f, indent=2, default=str) print(f"\nSaved to {OUT / 'diagnosis_v3.json'}") # Summary table print("\n" + "=" * 80) print("SUMMARY: Ideal vs Workflow-Exact") print("=" * 80) print(f"{'Year':<6} {'Ideal%zc':>10} {'WF nogate':>10} {'WF gate':>10} {'Gap(nogate)':>12} {'Gap(gate)':>12}") for r in results: y = r["year"] i = r["ideal_95_zc"]["ann_ret"] wn = r["wf_exact_95_nogate"]["ann_ret"] wg = r["wf_exact_95_gate"]["ann_ret"] print(f"{y:<6} {i:>+10.1%} {wn:>+10.1%} {wg:>+10.1%} {wn-i:>+12.1%} {wg-i:>+12.1%}") if __name__ == "__main__": main()