Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d03ad94745 |
@@ -1,27 +0,0 @@
|
|||||||
# TradeAC custom-qlib-code snapshot (auto-generated)
|
|
||||||
# parent repo HEAD : f9d1fe66f6e4f8ace0d3d774e23f1c79def9bae0
|
|
||||||
# tac-qlib/tac_qlib/contrib
|
|
||||||
# tac-qlib/tac_qlib/data
|
|
||||||
# per-file hashes (git hash-object):
|
|
||||||
1b6298c4a5652f2e863cbdc385a1014a570fcd59 tac-qlib/tac_qlib/contrib/__init__.py
|
|
||||||
b8112569f9b2537c45b6535e1a505a207878d322 tac-qlib/tac_qlib/contrib/__pycache__/__init__.cpython-312.pyc
|
|
||||||
c76a9f17f680e74eea766eff27f7624359749ed6 tac-qlib/tac_qlib/contrib/data/__init__.py
|
|
||||||
8d5333ebd2b44165c50cba639ca2d4ac3fc7cfec tac-qlib/tac_qlib/contrib/data/__pycache__/__init__.cpython-312.pyc
|
|
||||||
18cb37c0354184c49fa2e598396d7df0634cce0f tac-qlib/tac_qlib/contrib/data/__pycache__/handler.cpython-312.pyc
|
|
||||||
871ff1e163c29261f140c3f53d42a41e6504c779 tac-qlib/tac_qlib/contrib/data/handler.py
|
|
||||||
b151d139a0dcde87d74b21e7c4b729176ba5c39b tac-qlib/tac_qlib/contrib/model/__init__.py
|
|
||||||
ab958203f33a99d12c7d923b6efb435189231666 tac-qlib/tac_qlib/contrib/model/__pycache__/__init__.cpython-312.pyc
|
|
||||||
9dc36de7e343073b7d511349ee5aede086c38f94 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_ensemble.cpython-312.pyc
|
|
||||||
9f9014ddd9bce37490061312d51e8e6fe540fec4 tac-qlib/tac_qlib/contrib/model/__pycache__/rank_gbdt.cpython-312.pyc
|
|
||||||
d3f051f3a8650c42fedc7b367b966f7c74fb5789 tac-qlib/tac_qlib/contrib/model/rank_ensemble.py
|
|
||||||
ccfe7d554989aa7f3e5a2128ae663e51b2207149 tac-qlib/tac_qlib/contrib/model/rank_gbdt.py
|
|
||||||
4afcf9058231111c412925f4c4b84e81d656db87 tac-qlib/tac_qlib/contrib/strategy/__init__.py
|
|
||||||
74e5ecbbbb20bb71fd5cd083383de4ce88476712 tac-qlib/tac_qlib/contrib/strategy/__pycache__/__init__.cpython-312.pyc
|
|
||||||
afaf562aeaa12cebc8529cd916153252e7e3c38a tac-qlib/tac_qlib/contrib/strategy/__pycache__/optimal_stop.cpython-312.pyc
|
|
||||||
79aaad9e39fcc740a773f4f63c512ce1086cfde0 tac-qlib/tac_qlib/contrib/strategy/optimal_stop.py
|
|
||||||
92e6e90eb0cd0a25142034560f27adb6b705b1a8 tac-qlib/tac_qlib/data/__init__.py
|
|
||||||
0ed1ead6c1314a3f25784d453e54a15a8a04baaa tac-qlib/tac_qlib/data/__pycache__/__init__.cpython-312.pyc
|
|
||||||
9609782800944c45b78bb58eaa7b51ba1b7f8f43 tac-qlib/tac_qlib/data/__pycache__/config.cpython-312.pyc
|
|
||||||
a85628d71d12cfe5b18b1c884c5d829c89594579 tac-qlib/tac_qlib/data/__pycache__/providers.cpython-312.pyc
|
|
||||||
686d36f6d101c547491ca866aa143aa542e17518 tac-qlib/tac_qlib/data/config.py
|
|
||||||
d9f839be30026f337754a3f015425a8efdbe8e2a tac-qlib/tac_qlib/data/providers.py
|
|
||||||
@@ -1,11 +0,0 @@
|
|||||||
from . import data # noqa: F401 (registers tac_qlib.contrib.data)
|
|
||||||
from . import model, strategy # noqa: F401
|
|
||||||
from .data import TACHandler # noqa: F401
|
|
||||||
from .model import RankICLGBModel # noqa: F401
|
|
||||||
from .strategy import OptimalStopControl # noqa: F401
|
|
||||||
|
|
||||||
__all__ = [
|
|
||||||
"TACHandler",
|
|
||||||
"RankICLGBModel",
|
|
||||||
"OptimalStopControl",
|
|
||||||
]
|
|
||||||
Binary file not shown.
@@ -1,3 +0,0 @@
|
|||||||
from .handler import TACHandler
|
|
||||||
|
|
||||||
__all__ = ["TACHandler"]
|
|
||||||
Binary file not shown.
Binary file not shown.
@@ -1,236 +0,0 @@
|
|||||||
"""TACHandler: a qlib DataHandlerLP that builds datasets from the TradeAC lake.
|
|
||||||
|
|
||||||
This is the "custom DataHandler" entry point (Option B): the handler is referenced from the
|
|
||||||
workflow yaml's ``dataset.handler`` and reads OHLCV + pre-computed ta-lib features straight
|
|
||||||
from the lake parquet files through ``QLibDataLoader`` + the tac_qlib feature provider.
|
|
||||||
|
|
||||||
The standard qlib processor pipeline (``infer_processors`` / ``learn_processors``) still runs
|
|
||||||
on top, so existing recipes such as ``DropnaLabel``, ``CSZScoreNorm`` or ``RobustZScoreNorm``
|
|
||||||
keep working unchanged.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import os
|
|
||||||
from inspect import getfullargspec
|
|
||||||
from typing import List, Optional, Tuple, Union
|
|
||||||
|
|
||||||
from qlib.data.dataset import processor as processor_module
|
|
||||||
from qlib.data.dataset.handler import DataHandlerLP
|
|
||||||
from qlib.utils import get_callable_kwargs
|
|
||||||
|
|
||||||
from ...data.config import (
|
|
||||||
LakeConfig,
|
|
||||||
timeframe_for_freq,
|
|
||||||
NON_FEATURE_COLUMNS,
|
|
||||||
)
|
|
||||||
|
|
||||||
DEFAULT_INFER_PROCESSORS = [
|
|
||||||
{"class": "DropAllNaN", "kwargs": {}},
|
|
||||||
{"class": "ProcessInf", "kwargs": {}},
|
|
||||||
{"class": "ZScoreNorm", "kwargs": {}},
|
|
||||||
{"class": "Fillna", "kwargs": {}},
|
|
||||||
]
|
|
||||||
DEFAULT_LEARN_PROCESSORS = [
|
|
||||||
{"class": "DropnaLabel"},
|
|
||||||
{"class": "CSZScoreNorm", "kwargs": {"fields_group": "label"}},
|
|
||||||
]
|
|
||||||
|
|
||||||
#: always include raw OHLCV; ta-lib columns are discovered from the lake and appended.
|
|
||||||
RAW_FEATURE_FIELDS = ("$open", "$high", "$low", "$close", "$vwap", "$volume")
|
|
||||||
|
|
||||||
DEFAULT_LABEL = "Ref($close,-2)/Ref($close,-1)-1"
|
|
||||||
|
|
||||||
|
|
||||||
def check_transform_proc(proc_l, fit_start_time, fit_end_time):
|
|
||||||
"""Port of ``qlib.contrib.data.handler.check_transform_proc`` (inject fit window into procs)."""
|
|
||||||
new_l = []
|
|
||||||
for p in proc_l:
|
|
||||||
if not isinstance(p, processor_module.Processor):
|
|
||||||
klass, pkwargs = get_callable_kwargs(p, processor_module)
|
|
||||||
args = getfullargspec(klass).args
|
|
||||||
if "fit_start_time" in args and "fit_end_time" in args:
|
|
||||||
assert fit_start_time is not None and fit_end_time is not None, (
|
|
||||||
"Make sure `fit_start_time` and `fit_end_time` are not None."
|
|
||||||
)
|
|
||||||
pkwargs.update({"fit_start_time": fit_start_time, "fit_end_time": fit_end_time})
|
|
||||||
proc_config = {"class": klass.__name__, "kwargs": pkwargs}
|
|
||||||
if isinstance(p, dict) and "module_path" in p:
|
|
||||||
proc_config["module_path"] = p["module_path"]
|
|
||||||
new_l.append(proc_config)
|
|
||||||
else:
|
|
||||||
new_l.append(p)
|
|
||||||
return new_l
|
|
||||||
|
|
||||||
|
|
||||||
def get_common_feature_fields(lake_root=None, market="US", timeframe="1d") -> List[str]:
|
|
||||||
"""Discover ta-lib columns present in *every* features parquet file of the lake.
|
|
||||||
|
|
||||||
Returns sorted field names (without the ``$`` prefix). Empty if no features are persisted.
|
|
||||||
"""
|
|
||||||
cfg = LakeConfig(lake_root, market)
|
|
||||||
feat_dir = cfg.features_dir(timeframe)
|
|
||||||
if not feat_dir.exists():
|
|
||||||
return []
|
|
||||||
import pyarrow.parquet as pq
|
|
||||||
|
|
||||||
common = None
|
|
||||||
for p in sorted(feat_dir.glob("symbol=*.parquet")):
|
|
||||||
try:
|
|
||||||
cols = set(pq.read_schema(p).names) - set(NON_FEATURE_COLUMNS)
|
|
||||||
except Exception: # pragma: no cover - skip unreadable files
|
|
||||||
continue
|
|
||||||
common = cols if common is None else (common & cols)
|
|
||||||
if not common:
|
|
||||||
break
|
|
||||||
return sorted(common) if common else []
|
|
||||||
|
|
||||||
|
|
||||||
class DropAllNaN(processor_module.Processor):
|
|
||||||
"""Drop feature columns that are all-NaN over the fit window.
|
|
||||||
|
|
||||||
The lake can hold fully-empty indicator columns (e.g. a ta-lib output that was NaN
|
|
||||||
from the start). Such columns carry no learnable signal and make ``ZScoreNorm.fit``
|
|
||||||
warn on empty slices, so we drop them before any other processor runs. The drop set
|
|
||||||
is fixed on the fit window once (during ``fit``), then applied consistently to every
|
|
||||||
segment so train/valid/test keep identical feature columns.
|
|
||||||
"""
|
|
||||||
|
|
||||||
def __init__(self, fit_start_time=None, fit_end_time=None):
|
|
||||||
self.fit_start_time = fit_start_time
|
|
||||||
self.fit_end_time = fit_end_time
|
|
||||||
self.cols_to_drop = []
|
|
||||||
|
|
||||||
def fit(self, df=None):
|
|
||||||
if df is None or len(df) == 0:
|
|
||||||
return self
|
|
||||||
window = df
|
|
||||||
if self.fit_start_time is not None and self.fit_end_time is not None:
|
|
||||||
try:
|
|
||||||
from qlib.data.dataset.utils import fetch_df_by_index
|
|
||||||
|
|
||||||
window = fetch_df_by_index(
|
|
||||||
df, slice(self.fit_start_time, self.fit_end_time), level="datetime"
|
|
||||||
)
|
|
||||||
except Exception: # pragma: no cover - defensive
|
|
||||||
window = df
|
|
||||||
if len(window) == 0:
|
|
||||||
return self
|
|
||||||
self.cols_to_drop = [c for c in window.columns if window[c].isna().all()]
|
|
||||||
return self
|
|
||||||
|
|
||||||
def __call__(self, df):
|
|
||||||
if self.cols_to_drop:
|
|
||||||
return df.drop(columns=self.cols_to_drop, errors="ignore")
|
|
||||||
return df
|
|
||||||
|
|
||||||
|
|
||||||
class TACHandler(DataHandlerLP):
|
|
||||||
"""DataHandlerLP backed by the TradeAC parquet lake.
|
|
||||||
|
|
||||||
Parameters mirror ``Alpha158``: ``instruments``/``start_time``/``end_time``/``freq`` define
|
|
||||||
the queried window; ``feature_fields`` selects the features (default: raw OHLCV + all common
|
|
||||||
ta-lib columns found in the lake); ``label`` is a qlib expression for the target.
|
|
||||||
"""
|
|
||||||
|
|
||||||
def __init__(
|
|
||||||
self,
|
|
||||||
instruments="all",
|
|
||||||
start_time=None,
|
|
||||||
end_time=None,
|
|
||||||
freq="day",
|
|
||||||
infer_processors=DEFAULT_INFER_PROCESSORS,
|
|
||||||
learn_processors=DEFAULT_LEARN_PROCESSORS,
|
|
||||||
fit_start_time=None,
|
|
||||||
fit_end_time=None,
|
|
||||||
process_type=DataHandlerLP.PTYPE_A,
|
|
||||||
filter_pipe=None,
|
|
||||||
feature_fields=None,
|
|
||||||
label=DEFAULT_LABEL,
|
|
||||||
lake_root=None,
|
|
||||||
market="US",
|
|
||||||
**kwargs,
|
|
||||||
):
|
|
||||||
# default the processor fit window to the queried window (like Alpha158 without a split)
|
|
||||||
if fit_start_time is None:
|
|
||||||
fit_start_time = start_time
|
|
||||||
if fit_end_time is None:
|
|
||||||
fit_end_time = end_time
|
|
||||||
|
|
||||||
infer_processors = check_transform_proc(infer_processors, fit_start_time, fit_end_time)
|
|
||||||
learn_processors = check_transform_proc(learn_processors, fit_start_time, fit_end_time)
|
|
||||||
|
|
||||||
feature_fields = self._normalize_feature_fields(feature_fields, freq, lake_root, market)
|
|
||||||
if not feature_fields:
|
|
||||||
raise ValueError(
|
|
||||||
"no feature fields available for the lake; set `feature_fields` explicitly "
|
|
||||||
"(e.g. ['$close', '$rsi_14', '$sma_20'])"
|
|
||||||
)
|
|
||||||
|
|
||||||
label_expr, label_names = self._normalize_label(label)
|
|
||||||
|
|
||||||
data_loader = {
|
|
||||||
"class": "QlibDataLoader",
|
|
||||||
"kwargs": {
|
|
||||||
"config": {
|
|
||||||
"feature": (feature_fields, feature_fields),
|
|
||||||
"label": (label_expr, label_names),
|
|
||||||
},
|
|
||||||
"filter_pipe": filter_pipe,
|
|
||||||
"freq": freq,
|
|
||||||
},
|
|
||||||
}
|
|
||||||
super().__init__(
|
|
||||||
instruments=instruments,
|
|
||||||
start_time=start_time,
|
|
||||||
end_time=end_time,
|
|
||||||
data_loader=data_loader,
|
|
||||||
infer_processors=infer_processors,
|
|
||||||
learn_processors=learn_processors,
|
|
||||||
process_type=process_type,
|
|
||||||
**kwargs,
|
|
||||||
)
|
|
||||||
|
|
||||||
# ------------------------------------------------------------------ config
|
|
||||||
@staticmethod
|
|
||||||
def _normalize_feature_fields(feature_fields, freq, lake_root, market) -> List[str]:
|
|
||||||
if feature_fields is None:
|
|
||||||
common = get_common_feature_fields(lake_root, market, timeframe_for_freq(freq))
|
|
||||||
feature_fields = list(RAW_FEATURE_FIELDS) + ["$" + f for f in common if "$" + f not in RAW_FEATURE_FIELDS]
|
|
||||||
elif isinstance(feature_fields, str):
|
|
||||||
feature_fields = [f.strip() for f in feature_fields.split(",") if f.strip()]
|
|
||||||
fields = [f if f.startswith("$") else "$" + f for f in feature_fields]
|
|
||||||
# de-dup while preserving order
|
|
||||||
seen, out = set(), []
|
|
||||||
for f in fields:
|
|
||||||
if f not in seen:
|
|
||||||
seen.add(f)
|
|
||||||
out.append(f)
|
|
||||||
return out
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def _normalize_label(label) -> Tuple[List[str], List[str]]:
|
|
||||||
if isinstance(label, str):
|
|
||||||
return [label], ["LABEL0"]
|
|
||||||
if isinstance(label, (list, tuple)):
|
|
||||||
if len(label) == 2 and isinstance(label[0], str):
|
|
||||||
return [label[0]], list(label[1]) if isinstance(label[1], (list, tuple)) else [label[1]]
|
|
||||||
return list(label), ["LABEL%d" % i for i in range(len(label))]
|
|
||||||
raise TypeError(f"unsupported label config: {label!r}")
|
|
||||||
|
|
||||||
# ------------------------------------------------------------------ utils
|
|
||||||
def get_label_config(self):
|
|
||||||
return DEFAULT_LABEL
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def discover_feature_fields(lake_root=None, market="US", freq="day") -> List[str]:
|
|
||||||
return get_common_feature_fields(lake_root, market, timeframe_for_freq(freq))
|
|
||||||
|
|
||||||
|
|
||||||
__all__ = ["TACHandler", "DropAllNaN", "get_common_feature_fields"]
|
|
||||||
|
|
||||||
|
|
||||||
# Make `DropAllNaN` resolvable by bare name from processor configs (e.g. the default
|
|
||||||
# ``infer_processors`` and workflow yamls that reference it without a ``module_path``),
|
|
||||||
# mirroring how qlib registers its own processors in ``qlib.data.dataset.processor``.
|
|
||||||
processor_module.DropAllNaN = DropAllNaN
|
|
||||||
@@ -1,4 +0,0 @@
|
|||||||
from .rank_ensemble import RankICEnsembleLGBModel # noqa: F401
|
|
||||||
from .rank_gbdt import RankICLGBModel, rankic_feval # noqa: F401
|
|
||||||
|
|
||||||
__all__ = ["RankICLGBModel", "rankic_feval", "RankICEnsembleLGBModel"]
|
|
||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -1,189 +0,0 @@
|
|||||||
"""Seed-ensembled LightGBM that early-stops on cross-sectional RankIC.
|
|
||||||
|
|
||||||
``RankICEnsembleLGBModel`` wraps ``RankICLGBModel`` (per-day RankIC feval +
|
|
||||||
``metric='None'`` + ``first_metric_only`` early stopping) over a seed ensemble:
|
|
||||||
one sub-model is trained per seed with identical hyper-parameters, and
|
|
||||||
predictions are averaged across seeds. This is the model class the
|
|
||||||
``tac-rd-rank-ensemble-isolated`` reference run wires into its workflow
|
|
||||||
(``module_path: tac_qlib.contrib.model.rank_ensemble``).
|
|
||||||
|
|
||||||
The ensemble inherits the RankIC early-stopping behaviour of the single-seed
|
|
||||||
model (valid RankIC drives the stopping iteration) while the seed averaging
|
|
||||||
stabilizes the prediction against any single seed's early-stopping path.
|
|
||||||
|
|
||||||
Training is parallelized: the seed sub-models train in a thread pool —
|
|
||||||
``lgb.train`` is C++ and releases the GIL, so concurrent seeds do not block on
|
|
||||||
the GIL (5 seeds ~40min/5 on this box). Measured on a 6-physical-core / 12 SMT
|
|
||||||
host: the seeds scale ~2x, not linearly — the runs are memory-bandwidth bound
|
|
||||||
and each Booster caps its threads at ``cores // workers`` so 5 concurrent
|
|
||||||
boosters don't oversubscribe; larger-core hosts scale better. The qlib data
|
|
||||||
pipeline is warmed once on the calling thread (fills the handler cache), and
|
|
||||||
each worker then prepares its **own** ``lgb.Dataset`` (independent handle, so
|
|
||||||
no concurrent ``construct()`` on a shared handle — LightGBM's ``Dataset`` is
|
|
||||||
not thread-safe to build). qlib's ``R`` recorder is also not thread-safe, so
|
|
||||||
the per-seed evaluation curves are logged on the calling thread after the pool
|
|
||||||
finishes.
|
|
||||||
|
|
||||||
Wired into a workflow yaml like:
|
|
||||||
|
|
||||||
model:
|
|
||||||
class: RankICEnsembleLGBModel
|
|
||||||
module_path: tac_qlib.contrib.model.rank_ensemble
|
|
||||||
kwargs:
|
|
||||||
loss: mse
|
|
||||||
learning_rate: 0.02
|
|
||||||
num_leaves: 31
|
|
||||||
n_estimators: 3000
|
|
||||||
num_boost_round: 3000
|
|
||||||
early_stopping_rounds: 200
|
|
||||||
min_data_in_leaf: 20
|
|
||||||
lambda_l2: 0.5
|
|
||||||
colsample_bytree: 0.8
|
|
||||||
subsample: 0.8
|
|
||||||
subsample_freq: 1
|
|
||||||
reg_alpha: 0.1
|
|
||||||
reg_lambda: 1.0
|
|
||||||
seeds: "42,7,2026,99,123"
|
|
||||||
parallel: 5
|
|
||||||
|
|
||||||
Any ``**kwargs`` other than ``seeds``/``parallel`` are forwarded unchanged to
|
|
||||||
every ``RankICLGBModel`` sub-model (same params, different ``seed``).
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import os
|
|
||||||
from concurrent.futures import ThreadPoolExecutor
|
|
||||||
from typing import List, Optional
|
|
||||||
|
|
||||||
import pandas as pd
|
|
||||||
|
|
||||||
from qlib.data.dataset import DatasetH
|
|
||||||
from qlib.data.dataset.handler import DataHandlerLP
|
|
||||||
|
|
||||||
from tac_qlib.contrib.model.rank_gbdt import RankICLGBModel
|
|
||||||
|
|
||||||
__all__ = ["RankICEnsembleLGBModel"]
|
|
||||||
|
|
||||||
|
|
||||||
class RankICEnsembleLGBModel(RankICLGBModel):
|
|
||||||
"""Seed ensemble of RankIC-early-stopping LightGBM models.
|
|
||||||
|
|
||||||
Parameters
|
|
||||||
----------
|
|
||||||
seeds : comma-separated integers, one sub-model per seed.
|
|
||||||
parallel : number of seeds to train concurrently. ``0`` (default) = auto
|
|
||||||
(all seeds, bounded by the available cores); ``1`` = sequential.
|
|
||||||
**kwargs : forwarded to every ``RankICLGBModel`` sub-model (model
|
|
||||||
hyper-parameters). ``seeds``/``parallel`` are consumed here and not
|
|
||||||
forwarded.
|
|
||||||
"""
|
|
||||||
|
|
||||||
def __init__(self, seeds: str = "42", parallel: int = 0, **kwargs):
|
|
||||||
self.seeds = [int(s.strip()) for s in str(seeds).split(",") if s.strip()]
|
|
||||||
if not self.seeds:
|
|
||||||
raise ValueError("seeds must contain at least one integer")
|
|
||||||
self.parallel = int(parallel)
|
|
||||||
# drop seed/parallel handling from the base kwargs, keep everything else
|
|
||||||
self._model_kwargs = dict(kwargs)
|
|
||||||
super().__init__(**self._model_kwargs)
|
|
||||||
self._models: List[RankICLGBModel] = []
|
|
||||||
|
|
||||||
# --------------------------------------------------------------- helpers
|
|
||||||
@staticmethod
|
|
||||||
def _cores() -> int:
|
|
||||||
try:
|
|
||||||
return max(1, len(os.sched_getaffinity(0)))
|
|
||||||
except AttributeError:
|
|
||||||
return max(1, os.cpu_count() or 1)
|
|
||||||
|
|
||||||
def _worker_count(self) -> int:
|
|
||||||
if self.parallel > 0:
|
|
||||||
return min(len(self.seeds), self.parallel)
|
|
||||||
return min(len(self.seeds), self._cores())
|
|
||||||
|
|
||||||
# ------------------------------------------------------------------ fit
|
|
||||||
def fit(
|
|
||||||
self,
|
|
||||||
dataset: DatasetH,
|
|
||||||
num_boost_round: Optional[int] = None,
|
|
||||||
early_stopping_rounds: Optional[int] = None,
|
|
||||||
verbose_eval: int = 20,
|
|
||||||
evals_result=None,
|
|
||||||
reweighter=None,
|
|
||||||
**kwargs,
|
|
||||||
):
|
|
||||||
"""Train one RankICLGBModel per seed and keep them for prediction.
|
|
||||||
|
|
||||||
The qlib data pipeline is warmed once on this thread (handler cache),
|
|
||||||
then each seed sub-model trains in a parallel worker thread on its own
|
|
||||||
``lgb.Dataset`` (LightGBM releases the GIL in ``lgb.train``). Evals
|
|
||||||
are logged on this thread after the pool (qlib's ``R`` is not
|
|
||||||
thread-safe).
|
|
||||||
"""
|
|
||||||
n_round = num_boost_round or self.num_boost_round
|
|
||||||
n_es = early_stopping_rounds or self.early_stopping_rounds
|
|
||||||
|
|
||||||
if len(self.seeds) == 1:
|
|
||||||
m = RankICLGBModel(seed=self.seeds[0], **self._model_kwargs)
|
|
||||||
m.fit(
|
|
||||||
dataset,
|
|
||||||
num_boost_round=n_round,
|
|
||||||
early_stopping_rounds=n_es,
|
|
||||||
verbose_eval=verbose_eval,
|
|
||||||
evals_result=evals_result,
|
|
||||||
reweighter=reweighter,
|
|
||||||
**kwargs,
|
|
||||||
)
|
|
||||||
self._models = [m]
|
|
||||||
return
|
|
||||||
|
|
||||||
# Warm the qlib handler cache once on this thread so the workers'
|
|
||||||
# concurrent prepare() calls only hit cached frames (no first-write race).
|
|
||||||
proto = RankICLGBModel(seed=self.seeds[0], **self._model_kwargs)
|
|
||||||
proto._prepare_data(dataset, reweighter)
|
|
||||||
|
|
||||||
workers = self._worker_count()
|
|
||||||
# Cap per-Booster threads so concurrent seeds don't oversubscribe
|
|
||||||
# (LightGBM's num_threads=0 uses ALL cores per Booster).
|
|
||||||
per_booster = max(1, self._cores() // workers)
|
|
||||||
|
|
||||||
def fit_seed(seed):
|
|
||||||
m = RankICLGBModel(seed=seed, **self._model_kwargs)
|
|
||||||
if workers > 1 and "num_threads" not in m.params:
|
|
||||||
m.params["num_threads"] = per_booster
|
|
||||||
ds_l = m._prepare_data(dataset, reweighter)
|
|
||||||
booster, evals, names = m._train_from_datasets(
|
|
||||||
ds_l,
|
|
||||||
num_boost_round=n_round,
|
|
||||||
early_stopping_rounds=n_es,
|
|
||||||
verbose_eval=verbose_eval,
|
|
||||||
**kwargs,
|
|
||||||
)
|
|
||||||
m.model = booster
|
|
||||||
return m, evals, names
|
|
||||||
|
|
||||||
with ThreadPoolExecutor(max_workers=workers) as ex:
|
|
||||||
results = list(ex.map(fit_seed, self.seeds))
|
|
||||||
|
|
||||||
self._models = [m for m, _, _ in results]
|
|
||||||
|
|
||||||
# Merge + log evals on the main thread (qlib's R is not thread-safe).
|
|
||||||
if evals_result is not None:
|
|
||||||
for m, evals, names in results:
|
|
||||||
for k in names:
|
|
||||||
for key, val in evals.get(k, {}).items():
|
|
||||||
evals_result.setdefault(f"{k}.seed{m.params['seed']}", {})[key] = val
|
|
||||||
for m, evals, names in results:
|
|
||||||
self._log_evals(evals, names, prefix=f"seed{m.params['seed']}.")
|
|
||||||
|
|
||||||
# -------------------------------------------------------------- predict
|
|
||||||
def predict(self, dataset: DatasetH, segment="test") -> pd.Series:
|
|
||||||
"""Average the per-seed predictions over the given segment."""
|
|
||||||
if not self._models:
|
|
||||||
raise ValueError("model is not fitted yet!")
|
|
||||||
preds = [m.predict(dataset, segment=segment) for m in self._models]
|
|
||||||
if len(preds) == 1:
|
|
||||||
return preds[0]
|
|
||||||
frame = pd.concat(preds, axis=1)
|
|
||||||
return frame.mean(axis=1)
|
|
||||||
@@ -1,200 +0,0 @@
|
|||||||
"""LGBModel variant that early-stops on cross-sectional RankIC instead of l2.
|
|
||||||
|
|
||||||
Standard qlib ``LGBModel`` early-stops on the regression loss (mse). For
|
|
||||||
cross-sectional alpha signals the quantity we actually care about is the per-day
|
|
||||||
rank correlation (Rank IC), which mse early-stopping does not optimize for.
|
|
||||||
Experiments on the 50-ETF lake (SP-5d 55-feature panel) show that early-stopping
|
|
||||||
on a custom RankIC feval lifts RankIC 0.047 -> 0.075 vs. the mse-stopped model.
|
|
||||||
|
|
||||||
This class reuses ``LGBModel``'s data preparation but:
|
|
||||||
|
|
||||||
- tags each ``lgb.Dataset`` with per-day query ``group`` sizes so a ranking
|
|
||||||
metric can be computed per trading day;
|
|
||||||
- injects a custom ``feval`` (mean per-day Spearman of pred vs label) into
|
|
||||||
``lgb.train``; early stopping then selects the iteration that maximizes
|
|
||||||
RankIC on the valid set;
|
|
||||||
- forces ``metric='None'`` + ``first_metric_only=True`` so early-stopping
|
|
||||||
tracks RankIC only (not the regression loss).
|
|
||||||
|
|
||||||
Wired into a workflow yaml like:
|
|
||||||
|
|
||||||
model:
|
|
||||||
class: RankICLGBModel
|
|
||||||
module_path: tac_qlib.contrib.model.rank_gbdt
|
|
||||||
kwargs:
|
|
||||||
loss: mse
|
|
||||||
learning_rate: 0.03
|
|
||||||
num_leaves: 31
|
|
||||||
n_estimators: 500
|
|
||||||
...
|
|
||||||
|
|
||||||
The rank feval is used for early-stopping selection only; the objective stays
|
|
||||||
the configured loss (default mse). Set ``rank_eval=False`` to fall back to the
|
|
||||||
plain LGBModel behaviour (early-stop on the loss).
|
|
||||||
|
|
||||||
Generic: works for any cross-sectional panel whose qlib dataset index has a
|
|
||||||
``datetime`` level (each level value = one query group). The per-day groups are
|
|
||||||
derived automatically, so no universe-specific configuration is needed.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
from typing import List, Optional, Tuple
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import pandas as pd
|
|
||||||
import lightgbm as lgb
|
|
||||||
|
|
||||||
from qlib.data.dataset import DatasetH
|
|
||||||
from qlib.data.dataset.handler import DataHandlerLP
|
|
||||||
from qlib.contrib.model.gbdt import LGBModel
|
|
||||||
from qlib.workflow import R
|
|
||||||
|
|
||||||
__all__ = ["RankICLGBModel", "rankic_feval"]
|
|
||||||
|
|
||||||
|
|
||||||
def _per_day_spearman(preds: np.ndarray, labels: np.ndarray, group: np.ndarray) -> float:
|
|
||||||
"""Mean per-day Spearman rank correlation of preds vs labels.
|
|
||||||
|
|
||||||
``group`` holds the number of rows of each trading day (query group), in
|
|
||||||
order. Days with <3 valid rows or a constant pred/label are skipped.
|
|
||||||
"""
|
|
||||||
if group is None or len(group) == 0:
|
|
||||||
return 0.0
|
|
||||||
offs = np.concatenate([[0], np.cumsum(group.astype(int))])
|
|
||||||
vals = []
|
|
||||||
for i in range(len(group)):
|
|
||||||
s = slice(offs[i], offs[i + 1])
|
|
||||||
p, l = preds[s], labels[s]
|
|
||||||
if len(p) < 3 or np.std(p) == 0 or np.std(l) == 0:
|
|
||||||
continue
|
|
||||||
vals.append(np.corrcoef(pd.Series(p).rank(), pd.Series(l).rank())[0, 1])
|
|
||||||
return float(np.mean(vals)) if vals else 0.0
|
|
||||||
|
|
||||||
|
|
||||||
def rankic_feval(preds, dataset):
|
|
||||||
"""LightGBM feval: mean RankIC (higher is better in lgb convention)."""
|
|
||||||
labels = dataset.get_label()
|
|
||||||
group = dataset.get_group()
|
|
||||||
ric = _per_day_spearman(preds, labels, group)
|
|
||||||
return "rankic", ric, True # (name, value, higher_is_better)
|
|
||||||
|
|
||||||
|
|
||||||
class RankICLGBModel(LGBModel):
|
|
||||||
"""LGBModel that early-stops on per-day RankIC via a custom feval."""
|
|
||||||
|
|
||||||
def __init__(self, rank_eval: bool = True, **kwargs):
|
|
||||||
super().__init__(**kwargs)
|
|
||||||
self.rank_eval = rank_eval
|
|
||||||
|
|
||||||
def _prepare_data(self, dataset: DatasetH, reweighter=None) -> List[Tuple[lgb.Dataset, str]]:
|
|
||||||
ds_l = []
|
|
||||||
assert "train" in dataset.segments
|
|
||||||
for key in ["train", "valid"]:
|
|
||||||
if key in dataset.segments:
|
|
||||||
df = dataset.prepare(key, col_set=["feature", "label"], data_key=DataHandlerLP.DK_L)
|
|
||||||
if df.empty:
|
|
||||||
raise ValueError("Empty data from dataset, please check your dataset config.")
|
|
||||||
x, y = df["feature"], df["label"]
|
|
||||||
if y.values.ndim == 2 and y.values.shape[1] == 1:
|
|
||||||
y = np.squeeze(y.values)
|
|
||||||
else:
|
|
||||||
raise ValueError("LightGBM doesn't support multi-label training")
|
|
||||||
|
|
||||||
if reweighter is None:
|
|
||||||
w = None
|
|
||||||
elif hasattr(reweighter, "reweight"):
|
|
||||||
w = reweighter.reweight(df)
|
|
||||||
else:
|
|
||||||
raise ValueError("Unsupported reweighter type.")
|
|
||||||
|
|
||||||
# per-day query groups: each trading day is one group
|
|
||||||
if self.rank_eval and isinstance(df.index, pd.MultiIndex) and "datetime" in df.index.names:
|
|
||||||
group = df.groupby(level="datetime").size().to_numpy(dtype=np.int32)
|
|
||||||
else:
|
|
||||||
group = None
|
|
||||||
|
|
||||||
d = lgb.Dataset(x.values, label=y, weight=w, group=group, free_raw_data=False)
|
|
||||||
ds_l.append((d, key))
|
|
||||||
return ds_l
|
|
||||||
|
|
||||||
def _train_from_datasets(
|
|
||||||
self,
|
|
||||||
ds_l: List[Tuple[lgb.Dataset, str]],
|
|
||||||
num_boost_round: Optional[int] = None,
|
|
||||||
early_stopping_rounds: Optional[int] = None,
|
|
||||||
verbose_eval: int = 20,
|
|
||||||
evals_result=None,
|
|
||||||
**kwargs,
|
|
||||||
) -> Tuple[lgb.Booster, dict, List[str]]:
|
|
||||||
"""Train a Booster from already-prepared ``lgb.Dataset`` objects.
|
|
||||||
|
|
||||||
Pure training — no ``R.log_metrics`` — so it can be called from worker
|
|
||||||
threads (qlib's ``R`` recorder is not thread-safe; the caller decides
|
|
||||||
when/where to log). Returns ``(booster, evals_result, segment_names)``.
|
|
||||||
"""
|
|
||||||
if evals_result is None:
|
|
||||||
evals_result = {}
|
|
||||||
ds, names = list(zip(*ds_l))
|
|
||||||
|
|
||||||
callbacks = [
|
|
||||||
lgb.early_stopping(
|
|
||||||
self.early_stopping_rounds if early_stopping_rounds is None else early_stopping_rounds
|
|
||||||
),
|
|
||||||
lgb.log_evaluation(period=verbose_eval),
|
|
||||||
lgb.record_evaluation(evals_result),
|
|
||||||
]
|
|
||||||
if self.rank_eval:
|
|
||||||
# early-stopping must be driven ONLY by the RankIC feval, not l2.
|
|
||||||
# metric='None' suppresses the default l2 metric; first_metric_only
|
|
||||||
# makes early_stopping track the single remaining (rankic) metric.
|
|
||||||
self.params["metric"] = "None"
|
|
||||||
self.params["first_metric_only"] = True
|
|
||||||
feval = rankic_feval
|
|
||||||
else:
|
|
||||||
self.params.pop("metric", None)
|
|
||||||
self.params.pop("first_metric_only", None)
|
|
||||||
feval = None
|
|
||||||
|
|
||||||
booster = lgb.train(
|
|
||||||
self.params,
|
|
||||||
ds[0],
|
|
||||||
num_boost_round=self.num_boost_round if num_boost_round is None else num_boost_round,
|
|
||||||
valid_sets=ds,
|
|
||||||
valid_names=names,
|
|
||||||
feval=feval,
|
|
||||||
callbacks=callbacks,
|
|
||||||
**kwargs,
|
|
||||||
)
|
|
||||||
return booster, evals_result, list(names)
|
|
||||||
|
|
||||||
def _log_evals(self, evals_result, names: List[str], prefix: str = "") -> None:
|
|
||||||
"""Log recorded evaluation curves to qlib's active recorder."""
|
|
||||||
for k in names:
|
|
||||||
for key, val in evals_result.get(k, {}).items():
|
|
||||||
name = f"{prefix}{key}.{k}"
|
|
||||||
for epoch, m in enumerate(val):
|
|
||||||
R.log_metrics(**{name.replace("@", "_"): m}, step=epoch)
|
|
||||||
|
|
||||||
def fit(
|
|
||||||
self,
|
|
||||||
dataset: DatasetH,
|
|
||||||
num_boost_round: Optional[int] = None,
|
|
||||||
early_stopping_rounds: Optional[int] = None,
|
|
||||||
verbose_eval: int = 20,
|
|
||||||
evals_result=None,
|
|
||||||
reweighter=None,
|
|
||||||
**kwargs,
|
|
||||||
):
|
|
||||||
if evals_result is None:
|
|
||||||
evals_result = {}
|
|
||||||
ds_l = self._prepare_data(dataset, reweighter)
|
|
||||||
self.model, evals_result, names = self._train_from_datasets(
|
|
||||||
ds_l,
|
|
||||||
num_boost_round=num_boost_round,
|
|
||||||
early_stopping_rounds=early_stopping_rounds,
|
|
||||||
verbose_eval=verbose_eval,
|
|
||||||
evals_result=evals_result,
|
|
||||||
**kwargs,
|
|
||||||
)
|
|
||||||
self._log_evals(evals_result, names)
|
|
||||||
@@ -1,3 +0,0 @@
|
|||||||
from .optimal_stop import OptimalStopControl # noqa: F401
|
|
||||||
|
|
||||||
__all__ = ["OptimalStopControl"]
|
|
||||||
Binary file not shown.
Binary file not shown.
@@ -1,217 +0,0 @@
|
|||||||
"""Optimal-stopping / stochastic-control strategy for cross-sectional signals.
|
|
||||||
|
|
||||||
Entry is a control policy: a symbol opens a position only when its cross-sectional
|
|
||||||
signal percentile is at or above ``entry_pct`` (i.e. it is one of the top-ranked
|
|
||||||
names) and the portfolio has fewer than ``topk`` open positions.
|
|
||||||
|
|
||||||
Exit is an optimal-stopping rule: a held position is stopped (closed) when its
|
|
||||||
signal percentile falls below ``exit_pct`` (the continuation value of holding is
|
|
||||||
no longer worth the risk), OR after ``max_hold_days`` (time stop / finite
|
|
||||||
horizon), OR when the position P&L breaches ``sl`` (loss control) and the
|
|
||||||
position has been held at least ``min_hold_days``.
|
|
||||||
|
|
||||||
Sizing is fixed ``notional`` per position (equal-weight control), unlike the
|
|
||||||
TopkDropout cash-allocation heuristic.
|
|
||||||
|
|
||||||
Wired into qrun workflows like any ``BaseStrategy`` (see ``PortAnaRecord``
|
|
||||||
config). Mirrors the API usage of qlib's ``TopkDropoutStrategy``: ``Order``/
|
|
||||||
``OrderDir`` from ``qlib.backtest.decision``, ``trade_calendar`` /
|
|
||||||
``trade_exchange`` / ``trade_position`` injected by the backtest executor.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
from typing import List
|
|
||||||
|
|
||||||
import pandas as pd
|
|
||||||
|
|
||||||
from qlib.backtest import Order
|
|
||||||
from qlib.backtest.decision import OrderDir, TradeDecisionWO
|
|
||||||
from qlib.contrib.strategy.signal_strategy import BaseSignalStrategy
|
|
||||||
|
|
||||||
__all__ = ["OptimalStopControl"]
|
|
||||||
|
|
||||||
DEFAULT_NOTIONAL = 20_000.0
|
|
||||||
DEFAULT_ENTRY_PCT = 0.80
|
|
||||||
DEFAULT_EXIT_PCT = 0.50
|
|
||||||
DEFAULT_MAX_HOLD_DAYS = 10
|
|
||||||
DEFAULT_MIN_HOLD_DAYS = 2
|
|
||||||
DEFAULT_SL = -0.06
|
|
||||||
|
|
||||||
|
|
||||||
class OptimalStopControl(BaseSignalStrategy):
|
|
||||||
"""Optimal-stopping long-only strategy over a cross-sectional signal.
|
|
||||||
|
|
||||||
Parameters
|
|
||||||
----------
|
|
||||||
topk : max number of concurrent positions.
|
|
||||||
entry_pct : min cross-sectional score percentile required to OPEN (0..1).
|
|
||||||
exit_pct : held positions are stopped when score percentile < exit_pct.
|
|
||||||
max_hold_days : hard time stop (finite-horizon close).
|
|
||||||
min_hold_days : minimum holding days before stop-loss is evaluated.
|
|
||||||
notional : $ per position (equal-weight control).
|
|
||||||
sl : stop-loss threshold as fraction of entry price (<= 0), disabled if 0.
|
|
||||||
"""
|
|
||||||
|
|
||||||
def __init__(
|
|
||||||
self,
|
|
||||||
*,
|
|
||||||
signal=None,
|
|
||||||
topk: int = 10,
|
|
||||||
entry_pct: float = DEFAULT_ENTRY_PCT,
|
|
||||||
exit_pct: float = DEFAULT_EXIT_PCT,
|
|
||||||
max_hold_days: int = DEFAULT_MAX_HOLD_DAYS,
|
|
||||||
min_hold_days: int = DEFAULT_MIN_HOLD_DAYS,
|
|
||||||
notional: float = DEFAULT_NOTIONAL,
|
|
||||||
sl: float = DEFAULT_SL,
|
|
||||||
risk_degree: float = 0.95,
|
|
||||||
trade_exchange=None,
|
|
||||||
level_infra=None,
|
|
||||||
common_infra=None,
|
|
||||||
**kwargs,
|
|
||||||
):
|
|
||||||
super().__init__(
|
|
||||||
signal=signal,
|
|
||||||
trade_exchange=trade_exchange,
|
|
||||||
level_infra=level_infra,
|
|
||||||
common_infra=common_infra,
|
|
||||||
**kwargs,
|
|
||||||
)
|
|
||||||
self.topk = topk
|
|
||||||
self.entry_pct = entry_pct
|
|
||||||
self.exit_pct = exit_pct
|
|
||||||
self.max_hold_days = max_hold_days
|
|
||||||
self.min_hold_days = min_hold_days
|
|
||||||
self.notional = notional
|
|
||||||
self.sl = sl
|
|
||||||
|
|
||||||
# ------------------------------------------------------------------ utils
|
|
||||||
@staticmethod
|
|
||||||
def _pct_rank(score: pd.Series) -> pd.Series:
|
|
||||||
return score.rank(pct=True)
|
|
||||||
|
|
||||||
def _entry_price(self, pos) -> float:
|
|
||||||
# Position stores avg entry price under key "price" (see Position.position)
|
|
||||||
price = pos.position.get("price")
|
|
||||||
if price is None:
|
|
||||||
price = pos.get_stock_amount("price")
|
|
||||||
return float(price)
|
|
||||||
|
|
||||||
def _pnl_pct(self, pos, mark: float) -> float:
|
|
||||||
entry = self._entry_price(pos)
|
|
||||||
if not entry or entry != entry:
|
|
||||||
return 0.0
|
|
||||||
return mark / entry - 1.0
|
|
||||||
|
|
||||||
def _is_tradable(self, code, start, end, direction) -> bool:
|
|
||||||
try:
|
|
||||||
return self.trade_exchange.is_stock_tradable(
|
|
||||||
stock_id=code, start_time=start, end_time=end, direction=direction
|
|
||||||
)
|
|
||||||
except TypeError: # some exchanges take no direction kwarg
|
|
||||||
return self.trade_exchange.is_stock_tradable(stock_id=code, start_time=start, end_time=end)
|
|
||||||
|
|
||||||
# ------------------------------------------------------------ decision
|
|
||||||
def generate_trade_decision(self, execute_result=None):
|
|
||||||
trade_step = self.trade_calendar.get_trade_step()
|
|
||||||
trade_start, trade_end = self.trade_calendar.get_step_time(trade_step)
|
|
||||||
pred_start, pred_end = self.trade_calendar.get_step_time(trade_step, shift=1)
|
|
||||||
pred_score = self.signal.get_signal(start_time=pred_start, end_time=pred_end)
|
|
||||||
if isinstance(pred_score, pd.DataFrame):
|
|
||||||
pred_score = pred_score.iloc[:, 0]
|
|
||||||
if pred_score is None or len(pred_score) == 0:
|
|
||||||
return TradeDecisionWO([], self)
|
|
||||||
|
|
||||||
pct = self._pct_rank(pred_score)
|
|
||||||
time_per_step = self.trade_calendar.get_freq()
|
|
||||||
current_temp = __import__("copy").deepcopy(self.trade_position)
|
|
||||||
|
|
||||||
holdings = {}
|
|
||||||
for code in current_temp.get_stock_list():
|
|
||||||
if abs(current_temp.get_stock_amount(code)) > 1e-6:
|
|
||||||
holdings[code] = current_temp
|
|
||||||
|
|
||||||
# ---- optimal stopping: close held positions -----------------------
|
|
||||||
sell_orders: List[Order] = []
|
|
||||||
closed_today = set()
|
|
||||||
kept = {}
|
|
||||||
for code, pos in holdings.items():
|
|
||||||
held = current_temp.get_stock_count(code, bar=time_per_step)
|
|
||||||
mark = self.trade_exchange.get_deal_price(
|
|
||||||
stock_id=code, start_time=trade_start, end_time=trade_end, direction=Order.SELL
|
|
||||||
)
|
|
||||||
if mark is None or mark != mark:
|
|
||||||
continue
|
|
||||||
rank = pct.get(code, 0.0)
|
|
||||||
stop_pnl = held >= self.min_hold_days and self.sl < 0 and self._pnl_pct(pos, mark) <= self.sl
|
|
||||||
if held >= self.max_hold_days or rank < self.exit_pct or stop_pnl:
|
|
||||||
amt = abs(current_temp.get_stock_amount(code))
|
|
||||||
o = Order(stock_id=code, amount=amt, start_time=trade_start,
|
|
||||||
end_time=trade_end, direction=Order.SELL)
|
|
||||||
if self.trade_exchange.check_order(o):
|
|
||||||
sell_orders.append(o)
|
|
||||||
self.trade_exchange.deal_order(o, position=current_temp)
|
|
||||||
closed_today.add(code)
|
|
||||||
else:
|
|
||||||
kept[code] = mark
|
|
||||||
|
|
||||||
# ---- equal-weight control: target notional per name -----------------
|
|
||||||
# candidate opens: top-ranked names whose signal pct >= entry_pct
|
|
||||||
rank_desc = pred_score.sort_values(ascending=False)
|
|
||||||
held_codes = set(kept)
|
|
||||||
opens = []
|
|
||||||
for sym in rank_desc.index:
|
|
||||||
if len(opens) >= self.topk:
|
|
||||||
break
|
|
||||||
if sym in held_codes:
|
|
||||||
continue
|
|
||||||
if pct.get(sym, 0.0) < self.entry_pct:
|
|
||||||
continue
|
|
||||||
if not self._is_tradable(sym, trade_start, trade_end, OrderDir.BUY):
|
|
||||||
continue
|
|
||||||
opens.append(sym)
|
|
||||||
|
|
||||||
targets = held_codes | set(opens)
|
|
||||||
if not targets:
|
|
||||||
return TradeDecisionWO(sell_orders, self)
|
|
||||||
|
|
||||||
# total value (cash + marked positions) -> per-target notional
|
|
||||||
total_value = current_temp.get_cash()
|
|
||||||
for code, mark in kept.items():
|
|
||||||
total_value += abs(current_temp.get_stock_amount(code)) * mark
|
|
||||||
|
|
||||||
target_notional = total_value * self.risk_degree / max(1, len(targets))
|
|
||||||
|
|
||||||
# ---- rebalance kept positions toward target weight ------------------
|
|
||||||
buy_orders: List[Order] = []
|
|
||||||
for code, mark in kept.items():
|
|
||||||
cur = abs(current_temp.get_stock_amount(code)) * mark
|
|
||||||
diff_notional = target_notional - cur
|
|
||||||
if abs(diff_notional) / target_notional < 0.02:
|
|
||||||
continue # skip tiny rebalances
|
|
||||||
amount_delta = diff_notional / mark
|
|
||||||
direction = Order.BUY if amount_delta > 0 else Order.SELL
|
|
||||||
o = Order(stock_id=code, amount=abs(amount_delta), start_time=trade_start,
|
|
||||||
end_time=trade_end, direction=direction)
|
|
||||||
if self.trade_exchange.check_order(o):
|
|
||||||
(buy_orders if direction == Order.BUY else sell_orders).append(o)
|
|
||||||
self.trade_exchange.deal_order(o, position=current_temp)
|
|
||||||
|
|
||||||
# ---- open new positions at target weight ----------------------------
|
|
||||||
for sym in opens:
|
|
||||||
px = self.trade_exchange.get_deal_price(
|
|
||||||
stock_id=sym, start_time=trade_start, end_time=trade_end, direction=OrderDir.BUY
|
|
||||||
)
|
|
||||||
if px is None or px != px or px <= 0:
|
|
||||||
continue
|
|
||||||
amount = target_notional / px
|
|
||||||
factor = self.trade_exchange.get_factor(
|
|
||||||
stock_id=sym, start_time=trade_start, end_time=trade_end
|
|
||||||
)
|
|
||||||
amount = self.trade_exchange.round_amount_by_trade_unit(amount, factor)
|
|
||||||
o = Order(stock_id=sym, amount=amount, start_time=trade_start,
|
|
||||||
end_time=trade_end, direction=Order.BUY)
|
|
||||||
if self.trade_exchange.check_order(o):
|
|
||||||
buy_orders.append(o)
|
|
||||||
|
|
||||||
return TradeDecisionWO(sell_orders + buy_orders, self)
|
|
||||||
@@ -1,25 +0,0 @@
|
|||||||
from .config import (
|
|
||||||
LakeConfig,
|
|
||||||
BAR_FIELD_MAP,
|
|
||||||
FREQ_TO_TIMEFRAME,
|
|
||||||
UNKNOWN_FIELD_NAMES,
|
|
||||||
timeframe_for_freq,
|
|
||||||
resolve_lake_root,
|
|
||||||
)
|
|
||||||
from .providers import (
|
|
||||||
LakeCalendarProvider,
|
|
||||||
LakeInstrumentProvider,
|
|
||||||
LakeFeatureProvider,
|
|
||||||
)
|
|
||||||
|
|
||||||
__all__ = [
|
|
||||||
"LakeConfig",
|
|
||||||
"BAR_FIELD_MAP",
|
|
||||||
"FREQ_TO_TIMEFRAME",
|
|
||||||
"UNKNOWN_FIELD_NAMES",
|
|
||||||
"timeframe_for_freq",
|
|
||||||
"resolve_lake_root",
|
|
||||||
"LakeCalendarProvider",
|
|
||||||
"LakeInstrumentProvider",
|
|
||||||
"LakeFeatureProvider",
|
|
||||||
]
|
|
||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -1,175 +0,0 @@
|
|||||||
"""TradeAC lake configuration helpers.
|
|
||||||
|
|
||||||
The lake is a hive-partitioned parquet store (see ``tac-engine/skills/tradeac-lake``):
|
|
||||||
|
|
||||||
$TAC_LAKE_DIR/
|
|
||||||
├── market=US/
|
|
||||||
│ └── timeframe=1d/
|
|
||||||
│ └── symbol=AAPL.parquet # OHLCV bars: t, date, o, h, l, c, v, n, vw
|
|
||||||
├── features/ # ta-lib indicators, wide format
|
|
||||||
│ └── market=US/
|
|
||||||
│ └── timeframe=1d/
|
|
||||||
│ └── symbol=AAPL.parquet # t, sma_5, sma_20, rsi_14, ...
|
|
||||||
├── calendar.parquet # trading days per market
|
|
||||||
├── coverage.parquet # per (market,timeframe,symbol) loaded windows
|
|
||||||
└── symbols.parquet # asset master
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import os
|
|
||||||
from pathlib import Path
|
|
||||||
from typing import Dict, List, Optional
|
|
||||||
|
|
||||||
import pandas as pd
|
|
||||||
|
|
||||||
#: qlib freq string (Freq.__str__) -> lake timeframe partition name
|
|
||||||
FREQ_TO_TIMEFRAME: Dict[str, str] = {
|
|
||||||
"day": "1d",
|
|
||||||
"1d": "1d",
|
|
||||||
"min": "1m",
|
|
||||||
"1min": "1m",
|
|
||||||
"5min": "5m",
|
|
||||||
"10min": "10m",
|
|
||||||
"15min": "15m",
|
|
||||||
"30min": "30m",
|
|
||||||
"hour": "1h",
|
|
||||||
"1hour": "1h",
|
|
||||||
"2hour": "2h",
|
|
||||||
"4hour": "4h",
|
|
||||||
"week": "1w",
|
|
||||||
"1week": "1w",
|
|
||||||
"month": "1M",
|
|
||||||
"1month": "1M",
|
|
||||||
}
|
|
||||||
|
|
||||||
#: bar-field map: qlib field name (without the leading ``$``) -> lake bar column
|
|
||||||
BAR_FIELD_MAP: Dict[str, str] = {
|
|
||||||
"open": "o",
|
|
||||||
"high": "h",
|
|
||||||
"low": "l",
|
|
||||||
"close": "c",
|
|
||||||
"volume": "v",
|
|
||||||
"vwap": "vw",
|
|
||||||
"avg_amount": "vw", # amount / volume
|
|
||||||
}
|
|
||||||
|
|
||||||
#: fields that qlib core/backtest queries but the lake does not store -> all-NaN
|
|
||||||
UNKNOWN_FIELD_NAMES = ("factor", "change", "trade_unit", "suspend_flag")
|
|
||||||
|
|
||||||
#: columns in the parquet files that are not features
|
|
||||||
NON_FEATURE_COLUMNS = ("t", "date", "market", "timeframe", "symbol")
|
|
||||||
|
|
||||||
|
|
||||||
def timeframe_for_freq(freq: str) -> str:
|
|
||||||
"""Map a qlib frequency (e.g. ``day``, ``1min``) to a lake timeframe (e.g. ``1d``)."""
|
|
||||||
f = str(freq).lower()
|
|
||||||
if f not in FREQ_TO_TIMEFRAME:
|
|
||||||
raise ValueError(
|
|
||||||
f"unsupported qlib freq {freq!r}; supported freqs: {sorted(set(FREQ_TO_TIMEFRAME))}"
|
|
||||||
)
|
|
||||||
return FREQ_TO_TIMEFRAME[f]
|
|
||||||
|
|
||||||
|
|
||||||
def resolve_lake_root(lake_root: Optional[str] = None) -> Path:
|
|
||||||
"""Resolve the lake root: explicit arg > ``TAC_LAKE_DIR`` (no fallback).
|
|
||||||
|
|
||||||
``TAC_LAKE_DIR`` is **mandatory** — there is deliberately no default
|
|
||||||
A missing/empty value raises so a
|
|
||||||
misconfigured environment never silently points at a wrong directory.
|
|
||||||
"""
|
|
||||||
if lake_root is None:
|
|
||||||
lake_root = os.environ.get("TAC_LAKE_DIR")
|
|
||||||
if not lake_root:
|
|
||||||
raise RuntimeError(
|
|
||||||
"TAC_LAKE_DIR is not set. Point it at the TradeAC lake root, e.g. "
|
|
||||||
"export TAC_LAKE_DIR=/home/data/lake (docker) or set an absolute "
|
|
||||||
"path in your local .env."
|
|
||||||
)
|
|
||||||
return Path(str(lake_root)).expanduser().resolve()
|
|
||||||
|
|
||||||
|
|
||||||
class LakeConfig:
|
|
||||||
"""Path helpers + cached readers for a (lake_root, market) combination."""
|
|
||||||
|
|
||||||
def __init__(self, lake_root: Optional[str] = None, market: str = "US"):
|
|
||||||
self.lake_root: Path = resolve_lake_root(lake_root)
|
|
||||||
self.market: str = (market or "US").upper()
|
|
||||||
|
|
||||||
# ---- paths --------------------------------------------------------------
|
|
||||||
def bar_dir(self, timeframe: str) -> Path:
|
|
||||||
return self.lake_root / f"market={self.market}" / f"timeframe={timeframe}"
|
|
||||||
|
|
||||||
def bar_path(self, timeframe: str, symbol: str) -> Path:
|
|
||||||
return self.bar_dir(timeframe) / f"symbol={str(symbol).upper()}.parquet"
|
|
||||||
|
|
||||||
def features_dir(self, timeframe: str) -> Path:
|
|
||||||
return self.lake_root / "features" / f"market={self.market}" / f"timeframe={timeframe}"
|
|
||||||
|
|
||||||
def features_path(self, timeframe: str, symbol: str) -> Path:
|
|
||||||
return self.features_dir(timeframe) / f"symbol={str(symbol).upper()}.parquet"
|
|
||||||
|
|
||||||
def calendar_path(self) -> Path:
|
|
||||||
return self.lake_root / "calendar.parquet"
|
|
||||||
|
|
||||||
def symbols_path(self) -> Path:
|
|
||||||
return self.lake_root / "symbols.parquet"
|
|
||||||
|
|
||||||
def coverage_path(self) -> Path:
|
|
||||||
return self.lake_root / "coverage.parquet"
|
|
||||||
|
|
||||||
# ---- metadata readers ----------------------------------------------------
|
|
||||||
def load_symbols(self) -> List[str]:
|
|
||||||
"""All symbols known to the lake (from ``symbols.parquet``)."""
|
|
||||||
p = self.symbols_path()
|
|
||||||
if not p.exists():
|
|
||||||
return []
|
|
||||||
df = pd.read_parquet(p)
|
|
||||||
if "symbol" not in df.columns:
|
|
||||||
return []
|
|
||||||
return sorted(df["symbol"].astype(str).str.upper().tolist())
|
|
||||||
|
|
||||||
def symbol_spans(self, symbol: str, timeframe: str) -> List[tuple]:
|
|
||||||
"""Listing span(s) ``[(start_iso, end_iso)]`` for a symbol from coverage.parquet."""
|
|
||||||
p = self.coverage_path()
|
|
||||||
if p.exists():
|
|
||||||
try:
|
|
||||||
df = pd.read_parquet(p)
|
|
||||||
except Exception: # pragma: no cover - defensive
|
|
||||||
df = pd.DataFrame()
|
|
||||||
if len(df):
|
|
||||||
df = df[
|
|
||||||
(df.get("market") == self.market)
|
|
||||||
& (df.get("timeframe") == timeframe)
|
|
||||||
& (df.get("symbol") == str(symbol).upper())
|
|
||||||
]
|
|
||||||
if len(df):
|
|
||||||
row = df.iloc[0]
|
|
||||||
first = pd.Timestamp(row["first_t"]).date()
|
|
||||||
last = pd.Timestamp(row["last_t"]).date()
|
|
||||||
return [(first.isoformat(), last.isoformat())]
|
|
||||||
# fallback: derive from the bar file itself
|
|
||||||
p = self.bar_path(timeframe, symbol)
|
|
||||||
if p.exists():
|
|
||||||
import pyarrow.parquet as pq
|
|
||||||
|
|
||||||
tbl = pq.read_table(p, columns=["t"])
|
|
||||||
first = pd.Timestamp(tbl.column("t")[0].as_py()).date()
|
|
||||||
last = pd.Timestamp(tbl.column("t")[-1].as_py()).date()
|
|
||||||
return [(first.isoformat(), last.isoformat())]
|
|
||||||
return [("1970-01-01", "2099-12-31")]
|
|
||||||
|
|
||||||
def load_calendar_dates(self) -> List[pd.Timestamp]:
|
|
||||||
"""Trading days (midnight timestamps) for the market, from ``calendar.parquet``."""
|
|
||||||
p = self.calendar_path()
|
|
||||||
if p.exists():
|
|
||||||
df = pd.read_parquet(p)
|
|
||||||
if "date" in df.columns:
|
|
||||||
if "market" in df.columns:
|
|
||||||
df = df[df["market"] == self.market]
|
|
||||||
dates = pd.to_datetime(df["date"]).dt.normalize().sort_values().unique()
|
|
||||||
return [pd.Timestamp(x) for x in dates]
|
|
||||||
return []
|
|
||||||
|
|
||||||
def __repr__(self) -> str: # pragma: no cover
|
|
||||||
return f"LakeConfig(lake_root={self.lake_root}, market={self.market})"
|
|
||||||
@@ -1,231 +0,0 @@
|
|||||||
"""qlib data providers backed by the TradeAC parquet lake.
|
|
||||||
|
|
||||||
These providers plug into the standard qlib mechanism: ``qlib.init(calendar_provider=...,
|
|
||||||
instrument_provider=..., feature_provider=...)`` instantiates them and binds them to the
|
|
||||||
``Cal`` / ``Inst`` / ``FeatureD`` wrappers (see ``qlib.data.data.register_all_wrappers``).
|
|
||||||
The rest of qlib (``LocalDatasetProvider`` expression engine, backtest ``Exchange``) keeps
|
|
||||||
working unchanged because the interface contract is identical to the file-based providers:
|
|
||||||
|
|
||||||
- ``feature()`` returns a ``pd.Series`` indexed by the **calendar position** range
|
|
||||||
``[start_index, end_index]`` (matching ``FileFeatureStorage.__getitem__`` semantics).
|
|
||||||
- ``list_instruments()`` returns ``{symbol: [(start, end), ...]}``.
|
|
||||||
- ``load_calendar()`` returns a list of ``pd.Timestamp`` trading days.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import bisect
|
|
||||||
from typing import Dict, List, Optional, Union
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import pandas as pd
|
|
||||||
|
|
||||||
from qlib.data.data import CalendarProvider, FeatureProvider, InstrumentProvider
|
|
||||||
from qlib.log import get_module_logger
|
|
||||||
|
|
||||||
from .config import (
|
|
||||||
BAR_FIELD_MAP,
|
|
||||||
LakeConfig,
|
|
||||||
UNKNOWN_FIELD_NAMES,
|
|
||||||
timeframe_for_freq,
|
|
||||||
)
|
|
||||||
|
|
||||||
logger = get_module_logger("tac_qlib.data.providers")
|
|
||||||
|
|
||||||
|
|
||||||
def _day_freq(freq: str) -> bool:
|
|
||||||
return str(freq).lower() in ("day", "1d")
|
|
||||||
|
|
||||||
|
|
||||||
def _calendar_keys(cal: List[pd.Timestamp], freq: str) -> pd.Index:
|
|
||||||
"""Convert calendar timestamps into the same key space as the lake parquet."""
|
|
||||||
if _day_freq(freq):
|
|
||||||
return pd.Index([pd.Timestamp(x).date() for x in cal])
|
|
||||||
return pd.Index([pd.Timestamp(x) for x in cal])
|
|
||||||
|
|
||||||
|
|
||||||
class LakeCalendarProvider(CalendarProvider):
|
|
||||||
"""Trading calendar read from ``<lake>/calendar.parquet`` (fallback: derived from bars)."""
|
|
||||||
|
|
||||||
def __init__(self, lake_root: Optional[str] = None, market: str = "US"):
|
|
||||||
super().__init__()
|
|
||||||
self.cfg = LakeConfig(lake_root, market)
|
|
||||||
|
|
||||||
def load_calendar(self, freq, future):
|
|
||||||
timeframe = timeframe_for_freq(freq)
|
|
||||||
if not _day_freq(freq):
|
|
||||||
raise NotImplementedError(
|
|
||||||
f"freq={freq!r} (timeframe={timeframe}) is not supported yet: the lake calendar "
|
|
||||||
f"only covers daily sessions; add a minute-level calendar to `calendar.parquet`"
|
|
||||||
)
|
|
||||||
|
|
||||||
dates = self.cfg.load_calendar_dates()
|
|
||||||
if not dates:
|
|
||||||
# Fallback: derive the trading-day set from the persisted bar files.
|
|
||||||
bar_dir = self.cfg.bar_dir(timeframe)
|
|
||||||
if bar_dir.exists():
|
|
||||||
import pyarrow.parquet as pq
|
|
||||||
|
|
||||||
cal: Dict[pd.Timestamp, None] = {}
|
|
||||||
for p in sorted(bar_dir.glob("symbol=*.parquet")):
|
|
||||||
tbl = pq.read_table(p, columns=["t"])
|
|
||||||
for v in tbl.column("t"):
|
|
||||||
cal[pd.Timestamp(v.as_py()).normalize()] = None
|
|
||||||
dates = sorted(cal.keys())
|
|
||||||
if not dates:
|
|
||||||
return []
|
|
||||||
|
|
||||||
if future:
|
|
||||||
# append the next calendar day so that "today" is a valid trade date
|
|
||||||
last = dates[-1]
|
|
||||||
dates = dates + [pd.Timestamp(last) + pd.Timedelta(days=1)]
|
|
||||||
return dates
|
|
||||||
|
|
||||||
|
|
||||||
class LakeInstrumentProvider(InstrumentProvider):
|
|
||||||
"""Instruments from ``<lake>/symbols.parquet`` with listing spans from ``coverage.parquet``."""
|
|
||||||
|
|
||||||
def __init__(
|
|
||||||
self,
|
|
||||||
lake_root: Optional[str] = None,
|
|
||||||
market: str = "US",
|
|
||||||
markets: Optional[Dict[str, list]] = None,
|
|
||||||
):
|
|
||||||
super().__init__()
|
|
||||||
self.cfg = LakeConfig(lake_root, market)
|
|
||||||
#: optional named pools, e.g. ``{"sp500": ["AAPL", "MSFT"], "etf": ["SPY"]}``.
|
|
||||||
#: ``all`` / any unregistered name resolves to every symbol in the lake.
|
|
||||||
self.markets: Dict[str, list] = markets or {}
|
|
||||||
|
|
||||||
def _resolve_symbols(self, market: Union[str, list]) -> List[str]:
|
|
||||||
if isinstance(market, (list, tuple, pd.Index, np.ndarray)):
|
|
||||||
return [str(s).upper() for s in market]
|
|
||||||
if isinstance(market, str) and "," in market:
|
|
||||||
return [s.strip().upper() for s in market.split(",") if s.strip()]
|
|
||||||
if market in self.markets:
|
|
||||||
return [str(s).upper() for s in self.markets[market]]
|
|
||||||
return self.cfg.load_symbols()
|
|
||||||
|
|
||||||
def list_instruments(self, instruments, start_time=None, end_time=None, freq="day", as_list=False):
|
|
||||||
market = instruments["market"]
|
|
||||||
timeframe = timeframe_for_freq(freq)
|
|
||||||
|
|
||||||
symbols = self._resolve_symbols(market)
|
|
||||||
if not symbols:
|
|
||||||
if as_list:
|
|
||||||
return []
|
|
||||||
return {}
|
|
||||||
|
|
||||||
# clip listing spans to the queried window (mirror of LocalInstrumentProvider)
|
|
||||||
from qlib.data.data import Cal # pylint: disable=C0415
|
|
||||||
|
|
||||||
cal = Cal.calendar(freq=freq)
|
|
||||||
start_time = pd.Timestamp(start_time or cal[0])
|
|
||||||
end_time = pd.Timestamp(end_time or cal[-1])
|
|
||||||
|
|
||||||
out: Dict[str, list] = {}
|
|
||||||
for symbol in symbols:
|
|
||||||
spans = []
|
|
||||||
for begin, end in self.cfg.symbol_spans(symbol, timeframe):
|
|
||||||
lo = max(start_time, pd.Timestamp(begin))
|
|
||||||
hi = min(end_time, pd.Timestamp(end))
|
|
||||||
if lo <= hi:
|
|
||||||
spans.append((lo, hi))
|
|
||||||
if spans:
|
|
||||||
out[symbol] = spans
|
|
||||||
|
|
||||||
filter_pipe = instruments.get("filter_pipe") or []
|
|
||||||
for filter_config in filter_pipe:
|
|
||||||
from qlib.data import filter as F # pylint: disable=C0415
|
|
||||||
|
|
||||||
filter_t = getattr(F, filter_config["filter_type"]).from_config(filter_config)
|
|
||||||
out = filter_t(out, start_time, end_time, freq)
|
|
||||||
|
|
||||||
if as_list:
|
|
||||||
return list(out)
|
|
||||||
return out
|
|
||||||
|
|
||||||
|
|
||||||
class LakeFeatureProvider(FeatureProvider):
|
|
||||||
"""Feature data from the lake parquet (OHLCV bars + pre-computed ta-lib features).
|
|
||||||
|
|
||||||
Field routing:
|
|
||||||
- ``$open/$high/$low/$close/$volume/$vwap`` -> bar parquet columns
|
|
||||||
- ``$amount`` (= v*vw), ``$avg_amount`` (= vw) -> derived from bar parquet
|
|
||||||
- ``$factor/$change/...`` -> all-NaN (not stored)
|
|
||||||
- anything else -> a ta-lib column in the features parquet
|
|
||||||
"""
|
|
||||||
|
|
||||||
def __init__(self, lake_root: Optional[str] = None, market: str = "US"):
|
|
||||||
super().__init__()
|
|
||||||
self.cfg = LakeConfig(lake_root, market)
|
|
||||||
self._bar_cache: Dict[tuple, pd.DataFrame] = {}
|
|
||||||
self._feature_cache: Dict[tuple, pd.DataFrame] = {}
|
|
||||||
|
|
||||||
# ------------------------------------------------------------------ caches
|
|
||||||
def _load_bar_df(self, instrument: str, timeframe: str) -> pd.DataFrame:
|
|
||||||
key = (instrument, timeframe)
|
|
||||||
if key not in self._bar_cache:
|
|
||||||
p = self.cfg.bar_path(timeframe, instrument)
|
|
||||||
self._bar_cache[key] = pd.read_parquet(p) if p.exists() else pd.DataFrame()
|
|
||||||
return self._bar_cache[key]
|
|
||||||
|
|
||||||
def _load_feature_df(self, instrument: str, timeframe: str) -> pd.DataFrame:
|
|
||||||
key = (instrument, timeframe)
|
|
||||||
if key not in self._feature_cache:
|
|
||||||
p = self.cfg.features_path(timeframe, instrument)
|
|
||||||
self._feature_cache[key] = pd.read_parquet(p) if p.exists() else pd.DataFrame()
|
|
||||||
return self._feature_cache[key]
|
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def _keys(df: pd.DataFrame, freq: str) -> pd.Index:
|
|
||||||
ts = pd.to_datetime(df["t"])
|
|
||||||
return ts.dt.date if _day_freq(freq) else ts
|
|
||||||
|
|
||||||
# ------------------------------------------------------------------ fields
|
|
||||||
def _extract(self, instrument: str, field: str, timeframe: str, freq: str) -> Optional[pd.Series]:
|
|
||||||
"""Return the field as a Series keyed by date/timestamp (None if not present in the lake)."""
|
|
||||||
bar = self._load_bar_df(instrument, timeframe)
|
|
||||||
|
|
||||||
if field in BAR_FIELD_MAP:
|
|
||||||
col = BAR_FIELD_MAP[field]
|
|
||||||
if col in bar.columns:
|
|
||||||
return bar[col].astype(float).set_axis(self._keys(bar, freq))
|
|
||||||
return None
|
|
||||||
if field == "amount":
|
|
||||||
if "v" in bar.columns and "vw" in bar.columns:
|
|
||||||
return (bar["v"] * bar["vw"]).astype(float).set_axis(self._keys(bar, freq))
|
|
||||||
return None
|
|
||||||
if field in UNKNOWN_FIELD_NAMES:
|
|
||||||
return None
|
|
||||||
|
|
||||||
feat = self._load_feature_df(instrument, timeframe)
|
|
||||||
if field in feat.columns:
|
|
||||||
return feat[field].astype(float).set_axis(self._keys(feat, freq))
|
|
||||||
return None
|
|
||||||
|
|
||||||
# ------------------------------------------------------------------ api
|
|
||||||
def _get_calendar(self, freq: str) -> List[pd.Timestamp]:
|
|
||||||
from qlib.data.data import Cal # pylint: disable=C0415
|
|
||||||
|
|
||||||
cal = Cal.calendar(freq=freq)
|
|
||||||
return list(cal)
|
|
||||||
|
|
||||||
def feature(self, instrument, field, start_index, end_index, freq):
|
|
||||||
field = str(field)[1:]
|
|
||||||
timeframe = timeframe_for_freq(freq)
|
|
||||||
|
|
||||||
cal = self._get_calendar(freq)
|
|
||||||
n = len(cal)
|
|
||||||
lo = max(0, int(start_index))
|
|
||||||
hi = min(n - 1, int(end_index))
|
|
||||||
if lo > hi:
|
|
||||||
return pd.Series(dtype=np.float32)
|
|
||||||
|
|
||||||
keys = _calendar_keys(cal[lo : hi + 1], freq)
|
|
||||||
ser = self._extract(str(instrument).upper(), field, timeframe, freq)
|
|
||||||
if ser is None:
|
|
||||||
vals = np.full(len(keys), np.nan, dtype=np.float64)
|
|
||||||
else:
|
|
||||||
vals = ser.reindex(keys).to_numpy(dtype=np.float64)
|
|
||||||
return pd.Series(vals, index=pd.RangeIndex(lo, hi + 1))
|
|
||||||
@@ -0,0 +1,108 @@
|
|||||||
|
# Exp 7 (baseline) — LightGBM on the 60-ETF lake, typical algo-trading hyperparams.
|
||||||
|
# Universe: all 60 lake ETFs. Features: OHLCV + 28 ta-lib (auto-discovered).
|
||||||
|
# train 2022-01-01..2025-06-30 / valid 2025-07-01..2025-12-31 / test 2026-01-01..2026-08-12
|
||||||
|
{%- set LAKE = TAC_LAKE_DIR %}
|
||||||
|
|
||||||
|
qlib_init:
|
||||||
|
provider_uri: "{{ LAKE }}"
|
||||||
|
region: us
|
||||||
|
expression_cache: null
|
||||||
|
dataset_cache: null
|
||||||
|
|
||||||
|
calendar_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeCalendarProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
instrument_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeInstrumentProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
markets: {}
|
||||||
|
feature_provider:
|
||||||
|
class: tac_qlib.data.providers.LakeFeatureProvider
|
||||||
|
kwargs:
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
|
||||||
|
exp_manager:
|
||||||
|
class: MLflowExpManager
|
||||||
|
module_path: qlib.workflow.expm
|
||||||
|
kwargs:
|
||||||
|
uri: "sqlite:///mlruns.db"
|
||||||
|
default_exp_name: "tac-lake-lgb-baseline"
|
||||||
|
|
||||||
|
task:
|
||||||
|
model:
|
||||||
|
class: LGBModel
|
||||||
|
module_path: qlib.contrib.model.gbdt
|
||||||
|
kwargs:
|
||||||
|
loss: mse
|
||||||
|
learning_rate: 0.05
|
||||||
|
num_leaves: 64
|
||||||
|
n_estimators: 200
|
||||||
|
colsample_bytree: 0.8
|
||||||
|
subsample: 0.8
|
||||||
|
subsample_freq: 1
|
||||||
|
reg_alpha: 0.01
|
||||||
|
reg_lambda: 0.01
|
||||||
|
|
||||||
|
dataset:
|
||||||
|
class: DatasetH
|
||||||
|
module_path: qlib.data.dataset
|
||||||
|
kwargs:
|
||||||
|
handler:
|
||||||
|
class: TACHandler
|
||||||
|
module_path: tac_qlib.contrib.data.handler
|
||||||
|
kwargs:
|
||||||
|
instruments: all
|
||||||
|
start_time: 2021-10-01
|
||||||
|
end_time: 2026-08-12
|
||||||
|
fit_start_time: 2022-01-01
|
||||||
|
fit_end_time: 2025-06-30
|
||||||
|
freq: day
|
||||||
|
lake_root: "{{ LAKE }}"
|
||||||
|
market: US
|
||||||
|
segments:
|
||||||
|
train: [2022-01-01, 2025-06-30]
|
||||||
|
valid: [2025-07-01, 2025-12-31]
|
||||||
|
test: [2026-01-01, 2026-08-12]
|
||||||
|
|
||||||
|
record:
|
||||||
|
- class: SignalRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs: {}
|
||||||
|
|
||||||
|
- class: SigAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
ana_long_short: true
|
||||||
|
ann_scaler: 252
|
||||||
|
|
||||||
|
- class: PortAnaRecord
|
||||||
|
module_path: qlib.workflow.record_temp
|
||||||
|
kwargs:
|
||||||
|
config:
|
||||||
|
strategy:
|
||||||
|
class: TopkDropoutStrategy
|
||||||
|
module_path: qlib.contrib.strategy
|
||||||
|
kwargs:
|
||||||
|
signal: "<PRED>"
|
||||||
|
topk: 2
|
||||||
|
n_drop: 1
|
||||||
|
only_tradable: true
|
||||||
|
risk_degree: 0.95
|
||||||
|
backtest:
|
||||||
|
start_time: 2026-01-01
|
||||||
|
end_time: 2026-08-12
|
||||||
|
account: 1000000
|
||||||
|
benchmark: SPY
|
||||||
|
exchange_kwargs:
|
||||||
|
codes: all
|
||||||
|
deal_price: $close
|
||||||
|
freq: day
|
||||||
|
open_cost: 0.0005
|
||||||
|
close_cost: 0.0015
|
||||||
|
min_cost: 5.0
|
||||||
|
risk_analysis_freq: 1d
|
||||||
@@ -1,20 +0,0 @@
|
|||||||
# exp/10 sp5d-moment-features
|
|
||||||
|
|
||||||
Variant C: generic-only 19 + 16 new moment/volatility families (skew, kurt,
|
|
||||||
DSV+ratios, max_up/down, rv_ac1, rv_cv_22, sig lag-5). 35 sp_* fields, ou/hmm excluded.
|
|
||||||
|
|
||||||
Run a3f7d1d40c3d4b839314fcf5b40f9b08 (tac-rd-moments / exp 12) — FINISHED.
|
|
||||||
|
|
||||||
## Result: NEGATIVE (regression vs generic-only baseline)
|
|
||||||
|
|
||||||
| Metric | generic-only 19 (run 7b1e797) | +moments 35 (run a3f7d1d) |
|
|
||||||
|---|---|---|
|
|
||||||
| Rank IC | 0.0635 | 0.0466 |
|
|
||||||
| Rank ICIR | 0.276 | 0.183 |
|
|
||||||
| L-S Sharpe | 2.55 | 1.44 |
|
|
||||||
| net excess (cost) | +3.1% IR 0.28 | -16.2% IR -1.57 |
|
|
||||||
| MDD | -7.3% | -11.1% |
|
|
||||||
|
|
||||||
Same failure mode as ou/hmm in exp 9: adding cross-sectional moment features
|
|
||||||
to the 50-name panel degrades the rank signal. Generic-only 19 remains the
|
|
||||||
best configuration. No further moment-family variants planned.
|
|
||||||
@@ -1,133 +0,0 @@
|
|||||||
# -----------------------------------------------------------------------------
|
|
||||||
# ABLATION A (baseline): LightGBM with RankIC early-stopping on the 50-ETF SP-5d
|
|
||||||
# panel, using ALL 24 sp_* feature columns (ou,hmm,jump,har,trend,hurst,
|
|
||||||
# signature). Copy of the canonical workflow_lgb_sp5d_rankic.yaml with a
|
|
||||||
# distinct experiment name so the ablation runs are isolated.
|
|
||||||
#
|
|
||||||
# Run:
|
|
||||||
# rd_run_workflow config_path=tac-qlib/workflows/ablate_baseline_all_sp_fields.yaml \
|
|
||||||
# experiment_name=tac-rd-rank-ablate
|
|
||||||
# -----------------------------------------------------------------------------
|
|
||||||
{%- set LAKE = TAC_LAKE_DIR %}
|
|
||||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
|
||||||
{%- set SP_FIELDS = "sp_ret,sp_ou_zscore,sp_ou_half_life,sp_ou_revert,sp_hmm_p_regime1,sp_hmm_state,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
|
||||||
|
|
||||||
qlib_init:
|
|
||||||
provider_uri: "{{ LAKE }}"
|
|
||||||
region: us
|
|
||||||
expression_cache: null
|
|
||||||
dataset_cache: null
|
|
||||||
|
|
||||||
calendar_provider:
|
|
||||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
|
||||||
kwargs:
|
|
||||||
lake_root: "{{ LAKE }}"
|
|
||||||
market: US
|
|
||||||
instrument_provider:
|
|
||||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
|
||||||
kwargs:
|
|
||||||
lake_root: "{{ LAKE }}"
|
|
||||||
market: US
|
|
||||||
markets: {}
|
|
||||||
feature_provider:
|
|
||||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
|
||||||
kwargs:
|
|
||||||
lake_root: "{{ LAKE }}"
|
|
||||||
market: US
|
|
||||||
|
|
||||||
exp_manager:
|
|
||||||
class: MLflowExpManager
|
|
||||||
module_path: qlib.workflow.expm
|
|
||||||
kwargs:
|
|
||||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
|
||||||
default_exp_name: "tac-rd-rank-ablate"
|
|
||||||
|
|
||||||
task:
|
|
||||||
model:
|
|
||||||
class: RankICLGBModel
|
|
||||||
module_path: tac_qlib.contrib.model.rank_gbdt
|
|
||||||
kwargs:
|
|
||||||
loss: mse
|
|
||||||
learning_rate: 0.02
|
|
||||||
num_leaves: 31
|
|
||||||
n_estimators: 3000
|
|
||||||
num_boost_round: 3000
|
|
||||||
early_stopping_rounds: 200
|
|
||||||
min_data_in_leaf: 20
|
|
||||||
lambda_l2: 0.5
|
|
||||||
colsample_bytree: 0.8
|
|
||||||
subsample: 0.8
|
|
||||||
subsample_freq: 1
|
|
||||||
reg_alpha: 0.1
|
|
||||||
reg_lambda: 1.0
|
|
||||||
seed: 42
|
|
||||||
|
|
||||||
dataset:
|
|
||||||
class: DatasetH
|
|
||||||
module_path: qlib.data.dataset
|
|
||||||
kwargs:
|
|
||||||
handler:
|
|
||||||
class: TACHandler
|
|
||||||
module_path: tac_qlib.contrib.data.handler
|
|
||||||
kwargs:
|
|
||||||
instruments: "{{ UNIVERSE }}"
|
|
||||||
start_time: 2015-01-03
|
|
||||||
end_time: 2026-08-10
|
|
||||||
fit_start_time: 2015-01-03
|
|
||||||
fit_end_time: 2025-09-01
|
|
||||||
freq: day
|
|
||||||
lake_root: "{{ LAKE }}"
|
|
||||||
market: US
|
|
||||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
|
||||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
|
||||||
infer_processors:
|
|
||||||
- class: DropAllNaN
|
|
||||||
kwargs: {}
|
|
||||||
- class: ProcessInf
|
|
||||||
kwargs: {}
|
|
||||||
- class: CSRankNorm
|
|
||||||
kwargs: {}
|
|
||||||
- class: ZScoreNorm
|
|
||||||
kwargs: {}
|
|
||||||
- class: Fillna
|
|
||||||
kwargs: {}
|
|
||||||
segments:
|
|
||||||
train: [2015-01-03, 2025-09-01]
|
|
||||||
valid: [2025-09-03, 2026-01-03]
|
|
||||||
test: [2026-01-04, 2026-08-10]
|
|
||||||
|
|
||||||
record:
|
|
||||||
- class: SignalRecord
|
|
||||||
module_path: qlib.workflow.record_temp
|
|
||||||
kwargs: {}
|
|
||||||
- class: SigAnaRecord
|
|
||||||
module_path: qlib.workflow.record_temp
|
|
||||||
kwargs:
|
|
||||||
ana_long_short: true
|
|
||||||
ann_scaler: 252
|
|
||||||
- class: PortAnaRecord
|
|
||||||
module_path: qlib.workflow.record_temp
|
|
||||||
kwargs:
|
|
||||||
config:
|
|
||||||
strategy:
|
|
||||||
class: TopkDropoutStrategy
|
|
||||||
module_path: qlib.contrib.strategy
|
|
||||||
kwargs:
|
|
||||||
signal: "<PRED>"
|
|
||||||
topk: 10
|
|
||||||
n_drop: 2
|
|
||||||
only_tradable: true
|
|
||||||
risk_degree: 0.95
|
|
||||||
backtest:
|
|
||||||
start_time: 2026-01-04
|
|
||||||
end_time: 2026-08-10
|
|
||||||
account: 1000000
|
|
||||||
benchmark: SPY
|
|
||||||
exchange_kwargs:
|
|
||||||
codes: "{{ UNIVERSE }}"
|
|
||||||
deal_price: $close
|
|
||||||
freq: day
|
|
||||||
open_cost: 0.0005
|
|
||||||
close_cost: 0.0015
|
|
||||||
min_cost: 5.0
|
|
||||||
risk_analysis_freq: 1d
|
|
||||||
@@ -1,139 +0,0 @@
|
|||||||
# -----------------------------------------------------------------------------
|
|
||||||
# VARIANT C (generic + moments): keeps the winning generic-only 19-field set
|
|
||||||
# (jump,har,trend,hurst,signature) and adds the NEW generic moment families the
|
|
||||||
# engine now exposes:
|
|
||||||
# - realized skewness / kurtosis (sp_rskew_5, sp_rskew_22, sp_rkurt_5, sp_rkurt_22)
|
|
||||||
# - downside semi-variance + ratios (sp_dsv_1/5/22, sp_dsv_ratio_1/5/22)
|
|
||||||
# - signed max moves (sp_max_up, sp_max_down)
|
|
||||||
# - RV autocorr / vol-of-vol (sp_rv_ac1, sp_rv_cv_22)
|
|
||||||
# - longer-lag signature terms (sp_sig_level2_*_5)
|
|
||||||
# Drops the model-specific ou/hmm families (they scored high in importance but
|
|
||||||
# hurt the rank dimension in the all-24 run). Same panel/model as baseline.
|
|
||||||
#
|
|
||||||
# Run:
|
|
||||||
# rd_run_workflow config_path=experiments/workflows/ablate_generic_moments.yaml \
|
|
||||||
# experiment_name=tac-rd-moments
|
|
||||||
# -----------------------------------------------------------------------------
|
|
||||||
{%- set LAKE = TAC_LAKE_DIR %}
|
|
||||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
|
||||||
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_max_up,sp_max_down,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_rv_ac1,sp_rv_cv_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead,sp_sig_level2_lead_lag_5,sp_sig_level2_lag_lead_5,sp_rskew_5,sp_rskew_22,sp_rkurt_5,sp_rkurt_22,sp_dsv_1,sp_dsv_5,sp_dsv_22,sp_dsv_ratio_1,sp_dsv_ratio_5,sp_dsv_ratio_22" %}
|
|
||||||
|
|
||||||
qlib_init:
|
|
||||||
provider_uri: "{{ LAKE }}"
|
|
||||||
region: us
|
|
||||||
expression_cache: null
|
|
||||||
dataset_cache: null
|
|
||||||
|
|
||||||
calendar_provider:
|
|
||||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
|
||||||
kwargs:
|
|
||||||
lake_root: "{{ LAKE }}"
|
|
||||||
market: US
|
|
||||||
instrument_provider:
|
|
||||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
|
||||||
kwargs:
|
|
||||||
lake_root: "{{ LAKE }}"
|
|
||||||
market: US
|
|
||||||
markets: {}
|
|
||||||
feature_provider:
|
|
||||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
|
||||||
kwargs:
|
|
||||||
lake_root: "{{ LAKE }}"
|
|
||||||
market: US
|
|
||||||
|
|
||||||
exp_manager:
|
|
||||||
class: MLflowExpManager
|
|
||||||
module_path: qlib.workflow.expm
|
|
||||||
kwargs:
|
|
||||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
|
||||||
default_exp_name: "tac-rd-moments"
|
|
||||||
|
|
||||||
task:
|
|
||||||
model:
|
|
||||||
class: RankICLGBModel
|
|
||||||
module_path: tac_qlib.contrib.model.rank_gbdt
|
|
||||||
kwargs:
|
|
||||||
loss: mse
|
|
||||||
learning_rate: 0.02
|
|
||||||
num_leaves: 31
|
|
||||||
n_estimators: 3000
|
|
||||||
num_boost_round: 3000
|
|
||||||
early_stopping_rounds: 200
|
|
||||||
min_data_in_leaf: 20
|
|
||||||
lambda_l2: 0.5
|
|
||||||
colsample_bytree: 0.8
|
|
||||||
subsample: 0.8
|
|
||||||
subsample_freq: 1
|
|
||||||
reg_alpha: 0.1
|
|
||||||
reg_lambda: 1.0
|
|
||||||
seed: 42
|
|
||||||
|
|
||||||
dataset:
|
|
||||||
class: DatasetH
|
|
||||||
module_path: qlib.data.dataset
|
|
||||||
kwargs:
|
|
||||||
handler:
|
|
||||||
class: TACHandler
|
|
||||||
module_path: tac_qlib.contrib.data.handler
|
|
||||||
kwargs:
|
|
||||||
instruments: "{{ UNIVERSE }}"
|
|
||||||
start_time: 2015-01-03
|
|
||||||
end_time: 2026-08-10
|
|
||||||
fit_start_time: 2015-01-03
|
|
||||||
fit_end_time: 2025-09-01
|
|
||||||
freq: day
|
|
||||||
lake_root: "{{ LAKE }}"
|
|
||||||
market: US
|
|
||||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
|
||||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
|
||||||
infer_processors:
|
|
||||||
- class: DropAllNaN
|
|
||||||
kwargs: {}
|
|
||||||
- class: ProcessInf
|
|
||||||
kwargs: {}
|
|
||||||
- class: CSRankNorm
|
|
||||||
kwargs: {}
|
|
||||||
- class: ZScoreNorm
|
|
||||||
kwargs: {}
|
|
||||||
- class: Fillna
|
|
||||||
kwargs: {}
|
|
||||||
segments:
|
|
||||||
train: [2015-01-03, 2025-09-01]
|
|
||||||
valid: [2025-09-03, 2026-01-03]
|
|
||||||
test: [2026-01-04, 2026-08-10]
|
|
||||||
|
|
||||||
record:
|
|
||||||
- class: SignalRecord
|
|
||||||
module_path: qlib.workflow.record_temp
|
|
||||||
kwargs: {}
|
|
||||||
- class: SigAnaRecord
|
|
||||||
module_path: qlib.workflow.record_temp
|
|
||||||
kwargs:
|
|
||||||
ana_long_short: true
|
|
||||||
ann_scaler: 252
|
|
||||||
- class: PortAnaRecord
|
|
||||||
module_path: qlib.workflow.record_temp
|
|
||||||
kwargs:
|
|
||||||
config:
|
|
||||||
strategy:
|
|
||||||
class: TopkDropoutStrategy
|
|
||||||
module_path: qlib.contrib.strategy
|
|
||||||
kwargs:
|
|
||||||
signal: "<PRED>"
|
|
||||||
topk: 10
|
|
||||||
n_drop: 2
|
|
||||||
only_tradable: true
|
|
||||||
risk_degree: 0.95
|
|
||||||
backtest:
|
|
||||||
start_time: 2026-01-04
|
|
||||||
end_time: 2026-08-10
|
|
||||||
account: 1000000
|
|
||||||
benchmark: SPY
|
|
||||||
exchange_kwargs:
|
|
||||||
codes: "{{ UNIVERSE }}"
|
|
||||||
deal_price: $close
|
|
||||||
freq: day
|
|
||||||
open_cost: 0.0005
|
|
||||||
close_cost: 0.0015
|
|
||||||
min_cost: 5.0
|
|
||||||
risk_analysis_freq: 1d
|
|
||||||
@@ -1,134 +0,0 @@
|
|||||||
# -----------------------------------------------------------------------------
|
|
||||||
# ABLATION B (generic-only): same panel/model as the baseline, but feature
|
|
||||||
# fields restricted to the model-free / generic stochastic-process families
|
|
||||||
# (jump,har,trend,hurst,signature). Drops the model-specific ou (OU/AR-1
|
|
||||||
# half-life) and hmm (2-state regime) families to test whether the generic
|
|
||||||
# families alone dominate the rank dimension.
|
|
||||||
#
|
|
||||||
# Run:
|
|
||||||
# rd_run_workflow config_path=tac-qlib/workflows/ablate_generic_only_sp_fields.yaml \
|
|
||||||
# experiment_name=tac-rd-rank-ablate
|
|
||||||
# -----------------------------------------------------------------------------
|
|
||||||
{%- set LAKE = TAC_LAKE_DIR %}
|
|
||||||
{%- set UNIVERSE = "SPY,QQQ,DIA,IWM,MDY,VTI,VOO,VEA,VWO,VT,EFA,EEM,TLT,IEF,SHY,AGG,BND,LQD,HYG,JNK,EMB,GLD,SLV,USO,UNG,DBA,DBC,XLK,XLF,XLE,XLV,XLI,XLY,XLP,XLU,XLB,XLRE,ARKK,SMH,SOXX,IBB,XBI,ITA,XAR,ICLN,TAN,FDN,IGV,ESPO,REM" %}
|
|
||||||
{%- set SP_FIELDS = "sp_ret,sp_jump_ratio,sp_jump_flag,sp_jump_tail,sp_max_move,sp_rv1,sp_rv5,sp_rv22,sp_vol_ratio_5_22,sp_vol_ratio_1_22,sp_trend_slope_5,sp_trend_slope_20,sp_trend_slope_60,sp_logp,sp_hurst_exponent,sp_sig_level1_lead,sp_sig_level1_lag,sp_sig_level2_lead_lag,sp_sig_level2_lag_lead" %}
|
|
||||||
|
|
||||||
qlib_init:
|
|
||||||
provider_uri: "{{ LAKE }}"
|
|
||||||
region: us
|
|
||||||
expression_cache: null
|
|
||||||
dataset_cache: null
|
|
||||||
|
|
||||||
calendar_provider:
|
|
||||||
class: tac_qlib.data.providers.LakeCalendarProvider
|
|
||||||
kwargs:
|
|
||||||
lake_root: "{{ LAKE }}"
|
|
||||||
market: US
|
|
||||||
instrument_provider:
|
|
||||||
class: tac_qlib.data.providers.LakeInstrumentProvider
|
|
||||||
kwargs:
|
|
||||||
lake_root: "{{ LAKE }}"
|
|
||||||
market: US
|
|
||||||
markets: {}
|
|
||||||
feature_provider:
|
|
||||||
class: tac_qlib.data.providers.LakeFeatureProvider
|
|
||||||
kwargs:
|
|
||||||
lake_root: "{{ LAKE }}"
|
|
||||||
market: US
|
|
||||||
|
|
||||||
exp_manager:
|
|
||||||
class: MLflowExpManager
|
|
||||||
module_path: qlib.workflow.expm
|
|
||||||
kwargs:
|
|
||||||
uri: "sqlite:///{{ LAKE }}/mlruns.db"
|
|
||||||
default_exp_name: "tac-rd-rank-ablate"
|
|
||||||
|
|
||||||
task:
|
|
||||||
model:
|
|
||||||
class: RankICLGBModel
|
|
||||||
module_path: tac_qlib.contrib.model.rank_gbdt
|
|
||||||
kwargs:
|
|
||||||
loss: mse
|
|
||||||
learning_rate: 0.02
|
|
||||||
num_leaves: 31
|
|
||||||
n_estimators: 3000
|
|
||||||
num_boost_round: 3000
|
|
||||||
early_stopping_rounds: 200
|
|
||||||
min_data_in_leaf: 20
|
|
||||||
lambda_l2: 0.5
|
|
||||||
colsample_bytree: 0.8
|
|
||||||
subsample: 0.8
|
|
||||||
subsample_freq: 1
|
|
||||||
reg_alpha: 0.1
|
|
||||||
reg_lambda: 1.0
|
|
||||||
seed: 42
|
|
||||||
|
|
||||||
dataset:
|
|
||||||
class: DatasetH
|
|
||||||
module_path: qlib.data.dataset
|
|
||||||
kwargs:
|
|
||||||
handler:
|
|
||||||
class: TACHandler
|
|
||||||
module_path: tac_qlib.contrib.data.handler
|
|
||||||
kwargs:
|
|
||||||
instruments: "{{ UNIVERSE }}"
|
|
||||||
start_time: 2015-01-03
|
|
||||||
end_time: 2026-08-10
|
|
||||||
fit_start_time: 2015-01-03
|
|
||||||
fit_end_time: 2025-09-01
|
|
||||||
freq: day
|
|
||||||
lake_root: "{{ LAKE }}"
|
|
||||||
market: US
|
|
||||||
label: "Ref($close,-6)/Ref($close,-1)-1"
|
|
||||||
feature_fields: "$open,$high,$low,$close,$vwap,$volume,{{ SP_FIELDS }}"
|
|
||||||
infer_processors:
|
|
||||||
- class: DropAllNaN
|
|
||||||
kwargs: {}
|
|
||||||
- class: ProcessInf
|
|
||||||
kwargs: {}
|
|
||||||
- class: CSRankNorm
|
|
||||||
kwargs: {}
|
|
||||||
- class: ZScoreNorm
|
|
||||||
kwargs: {}
|
|
||||||
- class: Fillna
|
|
||||||
kwargs: {}
|
|
||||||
segments:
|
|
||||||
train: [2015-01-03, 2025-09-01]
|
|
||||||
valid: [2025-09-03, 2026-01-03]
|
|
||||||
test: [2026-01-04, 2026-08-10]
|
|
||||||
|
|
||||||
record:
|
|
||||||
- class: SignalRecord
|
|
||||||
module_path: qlib.workflow.record_temp
|
|
||||||
kwargs: {}
|
|
||||||
- class: SigAnaRecord
|
|
||||||
module_path: qlib.workflow.record_temp
|
|
||||||
kwargs:
|
|
||||||
ana_long_short: true
|
|
||||||
ann_scaler: 252
|
|
||||||
- class: PortAnaRecord
|
|
||||||
module_path: qlib.workflow.record_temp
|
|
||||||
kwargs:
|
|
||||||
config:
|
|
||||||
strategy:
|
|
||||||
class: TopkDropoutStrategy
|
|
||||||
module_path: qlib.contrib.strategy
|
|
||||||
kwargs:
|
|
||||||
signal: "<PRED>"
|
|
||||||
topk: 10
|
|
||||||
n_drop: 2
|
|
||||||
only_tradable: true
|
|
||||||
risk_degree: 0.95
|
|
||||||
backtest:
|
|
||||||
start_time: 2026-01-04
|
|
||||||
end_time: 2026-08-10
|
|
||||||
account: 1000000
|
|
||||||
benchmark: SPY
|
|
||||||
exchange_kwargs:
|
|
||||||
codes: "{{ UNIVERSE }}"
|
|
||||||
deal_price: $close
|
|
||||||
freq: day
|
|
||||||
open_cost: 0.0005
|
|
||||||
close_cost: 0.0015
|
|
||||||
min_cost: 5.0
|
|
||||||
risk_analysis_freq: 1d
|
|
||||||
Reference in New Issue
Block a user