Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5153d04983 | ||
|
|
ab919245c2 |
+15
-7
@@ -1,19 +1,27 @@
|
|||||||
# TradeAC custom-qlib-code snapshot (auto-generated)
|
# TradeAC custom-qlib-code snapshot (auto-generated)
|
||||||
# parent repo HEAD : 1075525d6e954dca0bb31daf6675904f8b569f1a
|
# parent repo HEAD : ab919245c2f3d6881cd6e16e77e583bbb6d5b000
|
||||||
# tac-qlib/tac_qlib/contrib
|
# tac-qlib/tac_qlib/contrib
|
||||||
# tac-qlib/tac_qlib/data
|
# tac-qlib/tac_qlib/data
|
||||||
# per-file hashes (git hash-object):
|
# per-file hashes (git hash-object):
|
||||||
1b6298c4a5652f2e863cbdc385a1014a570fcd59 tac-qlib/tac_qlib/contrib/__init__.py
|
1b6298c4a5652f2e863cbdc385a1014a570fcd59 tac-qlib/tac_qlib/contrib/__init__.py
|
||||||
|
bb903ca28c70ae5802d673d5b85481679f58c286 tac-qlib/tac_qlib/contrib/__pycache__/__init__.cpython-312.pyc
|
||||||
c76a9f17f680e74eea766eff27f7624359749ed6 tac-qlib/tac_qlib/contrib/data/__init__.py
|
c76a9f17f680e74eea766eff27f7624359749ed6 tac-qlib/tac_qlib/contrib/data/__init__.py
|
||||||
0dd25ef161c6e0f15eafc84886e7e1381deb38c3 tac-qlib/tac_qlib/contrib/data/handler.py
|
9749bb730880371ea7bf9bf0c5ffd78fcf6b5a91 tac-qlib/tac_qlib/contrib/data/__pycache__/__init__.cpython-312.pyc
|
||||||
|
9eb94cfac5d41ae20acb12612c219f210d463ab4 tac-qlib/tac_qlib/contrib/data/__pycache__/handler.cpython-312.pyc
|
||||||
|
871ff1e163c29261f140c3f53d42a41e6504c779 tac-qlib/tac_qlib/contrib/data/handler.py
|
||||||
b151d139a0dcde87d74b21e7c4b729176ba5c39b tac-qlib/tac_qlib/contrib/model/__init__.py
|
b151d139a0dcde87d74b21e7c4b729176ba5c39b tac-qlib/tac_qlib/contrib/model/__init__.py
|
||||||
|
d209c3e3e8c2a8683cd337ff3b10c04015f16dc6 tac-qlib/tac_qlib/contrib/model/__pycache__/__init__.cpython-312.pyc
|
||||||
|
532527ebe81b269f723c806286122fb5483d0379 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_ensemble.cpython-312.pyc
|
||||||
|
1681c9bc021188c0f134e33e6402f521a9d46e9c tac-qlib/tac_qlib/contrib/model/__pycache__/rank_gbdt.cpython-312.pyc
|
||||||
d3f051f3a8650c42fedc7b367b966f7c74fb5789 tac-qlib/tac_qlib/contrib/model/rank_ensemble.py
|
d3f051f3a8650c42fedc7b367b966f7c74fb5789 tac-qlib/tac_qlib/contrib/model/rank_ensemble.py
|
||||||
ccfe7d554989aa7f3e5a2128ae663e51b2207149 tac-qlib/tac_qlib/contrib/model/rank_gbdt.py
|
ccfe7d554989aa7f3e5a2128ae663e51b2207149 tac-qlib/tac_qlib/contrib/model/rank_gbdt.py
|
||||||
4afcf9058231111c412925f4c4b84e81d656db87 tac-qlib/tac_qlib/contrib/strategy/__init__.py
|
4afcf9058231111c412925f4c4b84e81d656db87 tac-qlib/tac_qlib/contrib/strategy/__init__.py
|
||||||
|
a6fb3d6c2d111b7ad423939582df67aacb36fead tac-qlib/tac_qlib/contrib/strategy/__pycache__/__init__.cpython-312.pyc
|
||||||
|
44ed28758151eb7fa4d388646cb1c6f04b450d8c tac-qlib/tac_qlib/contrib/strategy/__pycache__/optimal_stop.cpython-312.pyc
|
||||||
79aaad9e39fcc740a773f4f63c512ce1086cfde0 tac-qlib/tac_qlib/contrib/strategy/optimal_stop.py
|
79aaad9e39fcc740a773f4f63c512ce1086cfde0 tac-qlib/tac_qlib/contrib/strategy/optimal_stop.py
|
||||||
92e6e90eb0cd0a25142034560f27adb6b705b1a8 tac-qlib/tac_qlib/data/__init__.py
|
92e6e90eb0cd0a25142034560f27adb6b705b1a8 tac-qlib/tac_qlib/data/__init__.py
|
||||||
b6dc9ced54f4044f5954b60ddd199acae9eef456 tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc
|
319e3f728b093d6c483eb29de42617a45ee88830 tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc
|
||||||
5cede5184d17910b0232d343f8ecca460098ea11 tac-qlib/tac_qlib/data/__pycache__/config.cpython-312.pyc
|
08a7dcbcf3f34bdb784d3b16c04daf265d04ae5f tac-qlib/tac_qlib/data/__pycache__/config.cpython-312.pyc
|
||||||
3a5accd6d239f342354cf1fa9410a6d2fb921de0 tac-qlib/tac_qlib/data/__pycache__/providers.cpython-312.pyc
|
47337bd1e54b6e333f26a9088d645ed48fbc44f6 tac-qlib/tac_qlib/data/__pycache__/providers.cpython-312.pyc
|
||||||
53c9007a928841fd3c3b08450f9a6520ce1ac091 tac-qlib/tac_qlib/data/config.py
|
686d36f6d101c547491ca866aa143aa542e17518 tac-qlib/tac_qlib/data/config.py
|
||||||
8d0644f6f0d1efb94798ed444cc73e63b643459b tac-qlib/tac_qlib/data/providers.py
|
d9f839be30026f337754a3f015425a8efdbe8e2a tac-qlib/tac_qlib/data/providers.py
|
||||||
|
|||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -64,13 +64,9 @@ def check_transform_proc(proc_l, fit_start_time, fit_end_time):
|
|||||||
|
|
||||||
|
|
||||||
def get_common_feature_fields(lake_root=None, market="US", timeframe="1d") -> List[str]:
|
def get_common_feature_fields(lake_root=None, market="US", timeframe="1d") -> List[str]:
|
||||||
"""Discover feature columns present in *every* feature file of the lake.
|
"""Discover ta-lib columns present in *every* features parquet file of the lake.
|
||||||
|
|
||||||
Walks the `family=ta|sp` partition layout (plus any legacy flat files).
|
Returns sorted field names (without the ``$`` prefix). Empty if no features are persisted.
|
||||||
TA and SP columns are disjoint by construction, so the common set is
|
|
||||||
computed per family (columns shared by all symbol files of that family),
|
|
||||||
then the per-family results are unioned. Returns sorted field names
|
|
||||||
(without the ``$`` prefix). Empty if no features are persisted.
|
|
||||||
"""
|
"""
|
||||||
cfg = LakeConfig(lake_root, market)
|
cfg = LakeConfig(lake_root, market)
|
||||||
feat_dir = cfg.features_dir(timeframe)
|
feat_dir = cfg.features_dir(timeframe)
|
||||||
@@ -78,30 +74,16 @@ def get_common_feature_fields(lake_root=None, market="US", timeframe="1d") -> Li
|
|||||||
return []
|
return []
|
||||||
import pyarrow.parquet as pq
|
import pyarrow.parquet as pq
|
||||||
|
|
||||||
def _family_common(fam_dir: Path) -> set:
|
common = None
|
||||||
common = None
|
for p in sorted(feat_dir.glob("symbol=*.parquet")):
|
||||||
for p in sorted(fam_dir.glob("symbol=*.parquet")):
|
try:
|
||||||
try:
|
cols = set(pq.read_schema(p).names) - set(NON_FEATURE_COLUMNS)
|
||||||
cols = set(pq.read_schema(p).names) - set(NON_FEATURE_COLUMNS)
|
except Exception: # pragma: no cover - skip unreadable files
|
||||||
except Exception: # pragma: no cover - skip unreadable files
|
continue
|
||||||
continue
|
common = cols if common is None else (common & cols)
|
||||||
common = cols if common is None else (common & cols)
|
if not common:
|
||||||
if not common:
|
break
|
||||||
break
|
return sorted(common) if common else []
|
||||||
return common or set()
|
|
||||||
|
|
||||||
common: set = set()
|
|
||||||
# family tier: features/market=*/timeframe=*/family=*/symbol=*.parquet
|
|
||||||
for fam in ("ta", "sp"):
|
|
||||||
fam_dir = feat_dir / f"family={fam}"
|
|
||||||
if fam_dir.is_dir():
|
|
||||||
common |= _family_common(fam_dir)
|
|
||||||
# legacy flat: features/market=*/timeframe=*/symbol=*.parquet
|
|
||||||
if (feat_dir / "family=ta").exists() or (feat_dir / "family=sp").exists():
|
|
||||||
pass # family layout already covered
|
|
||||||
else:
|
|
||||||
common |= _family_common(feat_dir)
|
|
||||||
return sorted(common)
|
|
||||||
|
|
||||||
|
|
||||||
class DropAllNaN(processor_module.Processor):
|
class DropAllNaN(processor_module.Processor):
|
||||||
|
|||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -6,11 +6,10 @@ The lake is a hive-partitioned parquet store (see ``tac-engine/skills/tradeac-la
|
|||||||
├── market=US/
|
├── market=US/
|
||||||
│ └── timeframe=1d/
|
│ └── timeframe=1d/
|
||||||
│ └── symbol=AAPL.parquet # OHLCV bars: t, date, o, h, l, c, v, n, vw
|
│ └── symbol=AAPL.parquet # OHLCV bars: t, date, o, h, l, c, v, n, vw
|
||||||
├── features/ # indicators, wide format, family tier
|
├── features/ # ta-lib indicators, wide format
|
||||||
│ └── market=US/
|
│ └── market=US/
|
||||||
│ └── timeframe=1d/
|
│ └── timeframe=1d/
|
||||||
│ ├── family=ta/symbol=AAPL.parquet # t, sma_5, sma_20, rsi_14, ...
|
│ └── symbol=AAPL.parquet # t, sma_5, sma_20, rsi_14, ...
|
||||||
│ └── family=sp/symbol=AAPL.parquet # t, sp_ou_*, sp_hmm_*, ...
|
|
||||||
├── calendar.parquet # trading days per market
|
├── calendar.parquet # trading days per market
|
||||||
├── coverage.parquet # per (market,timeframe,symbol) loaded windows
|
├── coverage.parquet # per (market,timeframe,symbol) loaded windows
|
||||||
└── symbols.parquet # asset master
|
└── symbols.parquet # asset master
|
||||||
@@ -108,34 +107,8 @@ class LakeConfig:
|
|||||||
return self.lake_root / "features" / f"market={self.market}" / f"timeframe={timeframe}"
|
return self.lake_root / "features" / f"market={self.market}" / f"timeframe={timeframe}"
|
||||||
|
|
||||||
def features_path(self, timeframe: str, symbol: str) -> Path:
|
def features_path(self, timeframe: str, symbol: str) -> Path:
|
||||||
# Legacy flat path (no family tier). Prefer `load_features` which
|
|
||||||
# resolves the family=ta|sp partition layout.
|
|
||||||
return self.features_dir(timeframe) / f"symbol={str(symbol).upper()}.parquet"
|
return self.features_dir(timeframe) / f"symbol={str(symbol).upper()}.parquet"
|
||||||
|
|
||||||
def load_features(self, timeframe: str, symbol: str) -> pd.DataFrame:
|
|
||||||
"""All feature columns for a symbol, merging the `family=ta` and
|
|
||||||
`family=sp` partitions by timestamp. Returns an empty frame when no
|
|
||||||
feature files exist (legacy flat layout falls back transparently)."""
|
|
||||||
sym = str(symbol).upper()
|
|
||||||
frames = []
|
|
||||||
for family in ("ta", "sp"):
|
|
||||||
p = self.features_dir(timeframe) / f"family={family}" / f"symbol={sym}.parquet"
|
|
||||||
if p.exists():
|
|
||||||
frames.append(pd.read_parquet(p))
|
|
||||||
if not frames:
|
|
||||||
flat = self.features_dir(timeframe) / f"symbol={sym}.parquet"
|
|
||||||
if flat.exists():
|
|
||||||
return pd.read_parquet(flat)
|
|
||||||
return pd.DataFrame()
|
|
||||||
if len(frames) == 1:
|
|
||||||
return frames[0]
|
|
||||||
merged = frames[0]
|
|
||||||
for extra in frames[1:]:
|
|
||||||
merged = merged.merge(extra, on="t", how="outer", suffixes=("", "_dup"))
|
|
||||||
for c in [c for c in merged.columns if c.endswith("_dup")]:
|
|
||||||
merged = merged.drop(columns=c)
|
|
||||||
return merged
|
|
||||||
|
|
||||||
def calendar_path(self) -> Path:
|
def calendar_path(self) -> Path:
|
||||||
return self.lake_root / "calendar.parquet"
|
return self.lake_root / "calendar.parquet"
|
||||||
|
|
||||||
|
|||||||
@@ -173,7 +173,8 @@ class LakeFeatureProvider(FeatureProvider):
|
|||||||
def _load_feature_df(self, instrument: str, timeframe: str) -> pd.DataFrame:
|
def _load_feature_df(self, instrument: str, timeframe: str) -> pd.DataFrame:
|
||||||
key = (instrument, timeframe)
|
key = (instrument, timeframe)
|
||||||
if key not in self._feature_cache:
|
if key not in self._feature_cache:
|
||||||
self._feature_cache[key] = self.cfg.load_features(timeframe, instrument)
|
p = self.cfg.features_path(timeframe, instrument)
|
||||||
|
self._feature_cache[key] = pd.read_parquet(p) if p.exists() else pd.DataFrame()
|
||||||
return self._feature_cache[key]
|
return self._feature_cache[key]
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
|
|||||||
@@ -1,97 +0,0 @@
|
|||||||
# Re-run of experiment 16 with validated family=ta and family=sp lake features.
|
|
||||||
{%- set LAKE = TAC_LAKE_DIR %}
|
|
||||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
|
||||||
{%- set FEATURES = "$open,$high,$low,$close,$vwap,$volume,sma_5,sma_20,ema_12,ema_26,rsi_14,macd,macd_signal,macd_hist,bb_upper,bb_middle,bb_lower,atr_14,adx_14,sp_ret,sp_ou_half_life,sp_ou_revert,sp_ou_zscore,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_down,sp_max_move,sp_max_up,sp_rv1,sp_rv5,sp_rv22,sp_rv_ac1,sp_rv_cv_22,sp_vol_ratio_1_22,sp_vol_ratio_5_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_rskew_5,sp_rskew_22,sp_rkurt_5,sp_rkurt_22,sp_dsv_1,sp_dsv_5,sp_dsv_22,sp_dsv_ratio_1,sp_dsv_ratio_5,sp_dsv_ratio_22,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead,sp_sig_level2_lead_lag_5,sp_sig_level2_lag_lead_5" %}
|
|
||||||
|
|
||||||
qlib_init:
|
|
||||||
provider_uri: "{{ LAKE }}"
|
|
||||||
region: us
|
|
||||||
expression_cache: null
|
|
||||||
dataset_cache: null
|
|
||||||
calendar_provider:
|
|
||||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
|
||||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
|
||||||
instrument_provider:
|
|
||||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
|
||||||
kwargs: { lake_root: "{{ LAKE }}", market: US, markets: {} }
|
|
||||||
feature_provider:
|
|
||||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
|
||||||
kwargs: { lake_root: "{{ LAKE }}", market: US }
|
|
||||||
exp_manager:
|
|
||||||
class: MLflowExpManager
|
|
||||||
module_path: qlib.workflow.expm
|
|
||||||
kwargs: { uri: "sqlite:///mlruns.db", default_exp_name: "tac-rd-exp16-db-ta-sp" }
|
|
||||||
|
|
||||||
task:
|
|
||||||
model:
|
|
||||||
class: RankICEnsembleLGBModel
|
|
||||||
module_path: tac_qlib.contrib.model.rank_ensemble
|
|
||||||
kwargs:
|
|
||||||
loss: mse
|
|
||||||
learning_rate: 0.02
|
|
||||||
num_leaves: 31
|
|
||||||
n_estimators: 3000
|
|
||||||
num_boost_round: 3000
|
|
||||||
early_stopping_rounds: 200
|
|
||||||
min_data_in_leaf: 20
|
|
||||||
lambda_l2: 0.5
|
|
||||||
colsample_bytree: 0.8
|
|
||||||
subsample: 0.8
|
|
||||||
subsample_freq: 1
|
|
||||||
reg_alpha: 0.1
|
|
||||||
reg_lambda: 1.0
|
|
||||||
seeds: "42,7,2026,99,123"
|
|
||||||
|
|
||||||
dataset:
|
|
||||||
class: DatasetH
|
|
||||||
module_path: qlib.data.dataset
|
|
||||||
kwargs:
|
|
||||||
handler:
|
|
||||||
class: TACHandler
|
|
||||||
module_path: tac_qlib.contrib.data.handler
|
|
||||||
kwargs:
|
|
||||||
instruments: "{{ UNIVERSE }}"
|
|
||||||
start_time: 2015-01-03
|
|
||||||
end_time: 2026-08-10
|
|
||||||
fit_start_time: 2016-01-04
|
|
||||||
fit_end_time: 2025-09-01
|
|
||||||
freq: day
|
|
||||||
lake_root: "{{ LAKE }}"
|
|
||||||
market: US
|
|
||||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
|
||||||
feature_fields: "{{ FEATURES }}"
|
|
||||||
infer_processors:
|
|
||||||
- { class: DropAllNaN, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
|
|
||||||
- { class: ProcessInf, kwargs: {} }
|
|
||||||
- { class: CSRankNorm, kwargs: {} }
|
|
||||||
- { class: ZScoreNorm, kwargs: { fit_start_time: "2016-01-04", fit_end_time: "2025-09-01" } }
|
|
||||||
- { class: Fillna, kwargs: {} }
|
|
||||||
segments:
|
|
||||||
train: [2016-01-04, 2025-09-01]
|
|
||||||
valid: [2025-09-03, 2026-01-03]
|
|
||||||
test: [2026-01-04, 2026-08-10]
|
|
||||||
|
|
||||||
record:
|
|
||||||
- { class: SignalRecord, module_path: qlib.workflow.record_temp, kwargs: {} }
|
|
||||||
- { class: SigAnaRecord, module_path: qlib.workflow.record_temp, kwargs: { ana_long_short: true, ann_scaler: 252 } }
|
|
||||||
- class: PortAnaRecord
|
|
||||||
module_path: qlib.workflow.record_temp
|
|
||||||
kwargs:
|
|
||||||
config:
|
|
||||||
strategy:
|
|
||||||
class: TopkDropoutStrategy
|
|
||||||
module_path: qlib.contrib.strategy
|
|
||||||
kwargs: { signal: "<PRED>", topk: 10, n_drop: 2, only_tradable: true, risk_degree: 0.95 }
|
|
||||||
backtest:
|
|
||||||
start_time: 2026-01-04
|
|
||||||
end_time: 2026-08-10
|
|
||||||
account: 1000000
|
|
||||||
benchmark: SPY
|
|
||||||
exchange_kwargs:
|
|
||||||
codes: "{{ UNIVERSE }}"
|
|
||||||
deal_price: $close
|
|
||||||
freq: day
|
|
||||||
open_cost: 0.0005
|
|
||||||
close_cost: 0.0015
|
|
||||||
min_cost: 5.0
|
|
||||||
risk_analysis_freq: 1d
|
|
||||||
Reference in New Issue
Block a user