# ----------------------------------------------------------------------------- # CANONICAL: LightGBM with RankIC early-stopping on the 50-ETF SP-5d panel. # # Uses the tac-qlib contrib stack so no reinvention is needed: # - model: RankICLGBModel (tac_qlib.contrib.model.rank_gbdt) — early-stops # on per-day cross-sectional RankIC, not l2. The measured lever: # RankIC 0.047 -> 0.075 on the SP-5d signal, and with the tuned # budget the first config that beat SPY net of costs. # - handler: TACHandler (tac_qlib.contrib.data.handler) — lake features # - records: SignalRecord + SigAnaRecord + PortAnaRecord (TopkDropout) # # Feature columns are the 24 sp_* columns computed by the Rust get_lake_sp tool # (7 stochastic-process families: ou,hmm,jump,har,trend,hurst,signature). Any # other column present in the lake features parquet can be listed instead. # # Run: # rd_run_workflow config_path=tac-qlib/workflows/workflow_lgb_sp5d_rankic.yaml \ # experiment_name=tac-rd-rankic # ----------------------------------------------------------------------------- {%- set LAKE = TAC_LAKE_DIR %} {%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %} {%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %} qlib_init: provider_uri: "{{ LAKE }}" region: us expression_cache: null dataset_cache: null calendar_provider: class: tac_qlib.data.providers.LakeCalendarProvider kwargs: lake_root: "{{ LAKE }}" market: US instrument_provider: class: tac_qlib.data.providers.LakeInstrumentProvider kwargs: lake_root: "{{ LAKE }}" market: US markets: {} feature_provider: class: tac_qlib.data.providers.LakeFeatureProvider kwargs: lake_root: "{{ LAKE }}" market: US exp_manager: class: MLflowExpManager module_path: qlib.workflow.expm kwargs: uri: "sqlite:///{{ LAKE }}/mlruns.db" default_exp_name: "tac-rd-rankic" task: model: class: RankICLGBModel module_path: tac_qlib.contrib.model.rank_gbdt kwargs: loss: mse learning_rate: 0.02 num_leaves: 31 n_estimators: 3000 num_boost_round: 3000 early_stopping_rounds: 200 min_data_in_leaf: 20 lambda_l2: 0.5 colsample_bytree: 0.8 subsample: 0.8 subsample_freq: 1 reg_alpha: 0.1 reg_lambda: 1.0 seed: 42 dataset: class: DatasetH module_path: qlib.data.dataset kwargs: handler: class: TACHandler module_path: tac_qlib.contrib.data.handler kwargs: instruments: "{{ UNIVERSE }}" start_time: 2015-01-03 end_time: 2026-08-10 fit_start_time: 2015-01-03 fit_end_time: 2025-09-01 freq: day lake_root: "{{ LAKE }}" market: US label: "Ref($close,-6)/Ref($close,-1)-1" feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}" infer_processors: - class: DropAllNaN kwargs: {} - class: ProcessInf kwargs: {} - class: CSRankNorm kwargs: {} - class: ZScoreNorm kwargs: {} - class: Fillna kwargs: {} segments: train: [2015-01-03, 2025-09-01] valid: [2025-09-03, 2026-01-03] test: [2026-01-04, 2026-08-10] record: - class: SignalRecord module_path: qlib.workflow.record_temp kwargs: {} - class: SigAnaRecord module_path: qlib.workflow.record_temp kwargs: ana_long_short: true ann_scaler: 252 - class: PortAnaRecord module_path: qlib.workflow.record_temp kwargs: config: strategy: class: TopkDropoutStrategy module_path: qlib.contrib.strategy kwargs: signal: "" topk: 10 n_drop: 2 only_tradable: true risk_degree: 0.95 backtest: start_time: 2026-01-04 end_time: 2026-08-10 account: 1000000 benchmark: SPY exchange_kwargs: codes: "{{ UNIVERSE }}" deal_price: $close freq: day open_cost: 0.0005 close_cost: 0.0015 min_cost: 5.0 risk_analysis_freq: 1d