book: evidence boundary (clean-lake watermark) + ch02 cost reality + ch05 clean-lake reset — exp 21-31, chat mining
This commit is contained in:
@@ -0,0 +1,138 @@
|
||||
[user] give a prompt so that I can reproduce same experiment run as https://tradeac-dev.h.lizhao.net/rd/input?expId=16&run=0cea66d9892246519bdf329a0410a277, i.e., the prompt should give the skills/input/data/feature/model/hyperparameters etc
|
||||
|
||||
[assistant] Let me load the relevant skills and look up the experiment details.
|
||||
|
||||
[assistant] Now let me fetch the experiment details.
|
||||
|
||||
[assistant] Now let me also grab the metrics:
|
||||
|
||||
[assistant] Here is the prompt to reproduce experiment 16 / run `0cea66d9892246519bdf329a0410a277`:
|
||||
|
||||
---
|
||||
|
||||
**Reproduce experiment 16 — RankICEnsembleLGBModel (5-seed RankIC early-stop ensemble)**
|
||||
|
||||
Run the following workflow via `rd_run_workflow` into a new experiment. First ensure the lake has 1d bars + stochastic-process features for the 50-ETF universe back to 2015-01-03, then create and run this YAML:
|
||||
|
||||
```yaml
|
||||
{%- set LAKE = TAC_LAKE_DIR %}
|
||||
|
||||
qlib_init:
|
||||
provider_uri: "{{ LAKE }}"
|
||||
region: us
|
||||
expression_cache: null
|
||||
dataset_cache: null
|
||||
calendar_provider:
|
||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
instrument_provider:
|
||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
|
||||
feature_provider:
|
||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
||||
exp_manager:
|
||||
class: MLflowExpManager
|
||||
module_path: qlib.workflow.expm
|
||||
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-rank-ensemble-repro" }
|
||||
|
||||
task:
|
||||
model:
|
||||
class: RankICEnsembleLGBModel
|
||||
module_path: tac_qlib.contrib.model.rank_ensemble
|
||||
kwargs:
|
||||
loss: mse
|
||||
learning_rate: 0.02
|
||||
num_leaves: 31
|
||||
n_estimators: 3000
|
||||
num_boost_round: 3000
|
||||
early_stopping_rounds: 200
|
||||
min_data_in_leaf: 20
|
||||
lambda_l2: 0.5
|
||||
colsample_bytree: 0.8
|
||||
subsample: 0.8
|
||||
subsample_freq: 1
|
||||
reg_alpha: 0.1
|
||||
reg_lambda: 1.0
|
||||
seeds: "42,7,2026,99,123"
|
||||
|
||||
dataset:
|
||||
class: DatasetH
|
||||
module_path: qlib.data.dataset
|
||||
kwargs:
|
||||
handler:
|
||||
class: TACHandler
|
||||
module_path: tac_qlib.contrib.data.handler
|
||||
kwargs:
|
||||
instruments: "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM"
|
||||
start_time: "2015-01-03"
|
||||
end_time: "2026-08-14"
|
||||
fit_start_time: "2016-01-04"
|
||||
fit_end_time: "2025-09-01"
|
||||
freq: day
|
||||
lake_root: "{{ LAKE }}"
|
||||
market: US
|
||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead"
|
||||
infer_processors:
|
||||
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
|
||||
- { class: ProcessInf, kwargs: {} }
|
||||
- { class: CSRankNorm, kwargs: {} }
|
||||
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
|
||||
- { class: Fillna, kwargs: {} }
|
||||
segments:
|
||||
train: ["2016-01-04", "2025-09-01"]
|
||||
valid: ["2025-09-03", "2026-01-03"]
|
||||
test: ["2026-01-04", "2026-08-10"]
|
||||
|
||||
record:
|
||||
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
|
||||
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
|
||||
- class: PortAnaRecord
|
||||
module_path: qlib.workflow.record_temp
|
||||
kwargs:
|
||||
config:
|
||||
strategy:
|
||||
class: TopkDropoutStrategy
|
||||
module_path: qlib.contrib.strategy
|
||||
kwargs: { signal: "<PRED>", topk: 10, n_drop: 2, only_tradable: true, risk_degree: 0.95 }
|
||||
backtest:
|
||||
start_time: "2026-01-04"
|
||||
end_time: "2026-08-10"
|
||||
account: 1000000
|
||||
benchmark: SPY
|
||||
exchange_kwargs:
|
||||
codes: "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM"
|
||||
deal_price: $close
|
||||
freq: day
|
||||
open_cost: 0.0005
|
||||
close_cost: 0.0015
|
||||
min_cost: 5.0
|
||||
risk_analysis_freq: 1d
|
||||
```
|
||||
|
||||
**Prerequisites (lake data backfill):**
|
||||
```
|
||||
# 1. Backfill calendar
|
||||
backfill_lake_calendar symbols=SPY start=2015-01-03 end=2026-08-14
|
||||
|
||||
# 2. Backfill 1d bars for the full universe
|
||||
get_lake_bars symbols="SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" timeframe=1d start=2015-01-03 end=2026-08-14 lazy=true
|
||||
|
||||
# 3. Compute & persist stochastic-process features (one symbol at a time or batch)
|
||||
get_lake_sp symbol=<each> timeframe=1d start=2015-01-03 end=2026-08-14 fit_end=2025-09-01 families="ou,jump,har,trend,hurst,signature,moments"
|
||||
```
|
||||
|
||||
**Run:**
|
||||
```
|
||||
rd_run_workflow config_path=<path-to-yaml> experiment_name=tac-rd-rank-ensemble-repro
|
||||
```
|
||||
|
||||
**Key design choices being reproduced:**
|
||||
- **Model**: `RankICEnsembleLGBModel` — 5 seeds (`42,7,2026,99,123`), each training a `RankICLGBModel` (per-day RankIC feval, `metric='None'` + `first_metric_only`), predictions averaged across seeds
|
||||
- **Label**: 5-day forward return (`Ref($close,-6)/Ref($close,-1)-1`)
|
||||
- **Features**: 26 bar + stochastic-process features (jump intensity, HARRV vol, trend slopes, Hurst exponent, path signatures) — no ta-lib indicators
|
||||
- **Universe**: 50 ETFs (equity factor + bond + commodity + thematic)
|
||||
- **Train window**: 2016-01-04 → 2025-09-01 (~9.7y), valid 2025-09-03 → 2026-01-03 (~4mo), test 2026-01-04 → 2026-08-10 (~7mo)
|
||||
- **Processors**: DropAllNaN → ProcessInf → CSRankNorm → ZScoreNorm → Fillna
|
||||
- **Strategy**: TopkDropout topk=10, n_drop=2, risk_degree=0.95, benchmark=SPY, costs 5/15bps
|
||||
Reference in New Issue
Block a user