검증 데이터 누수 없이 선택된 홀드아웃 예측 내보내기
코드 Machine Learning for Trading
요약
이 유틸리티는 열 이름과 날짜 형식이 서로 다른 예측 아티팩트를 공통 타임스탬프, 자산, 예측 구조로 정규화해 라이브 데모 노트북에 사용할 수 있게 준비합니다. 타임스탬프와 자산이 중복되는 관측값을 묶어 점수의 평균을 계산합니다. 로더는 먼저 저장소에 커밋된 모델 출력을 검색하고, 해당 파일이 없으면 레지스트리를 조회합니다.
레지스트리 대체 경로는 기록된 신호, 배분, 리스크 오버레이 백테스트 가운데 검증 샤프가 가장 높은 모델 실행을 우선해 해당 실행의 홀드아웃 예측을 내보냅니다. 이렇게 내보낸 예측을 설정 선택에 사용된 검증 데이터와 분리해, 선택한 모델을 그 선택 표본에서 평가할 위험을 줄입니다. 적격 백테스트 우승 기록이 없으면 요청한 기간에 대한 평균 정보계수로 홀드아웃 예측 집합을 선택합니다. 이 대체 경로는 실용적인 복구 방법이지만 다른 선택 기준을 사용합니다. 이 코드는 아티팩트를 불러오는 로직이며, 선택된 모델이 표본 외에서 잘 작동한다는 증거를 제공하지 않습니다.
핵심 아이디어
- 아티팩트를 후속 노트북에 전달하기 전에 예측 열과 타임스탬프를 정규화합니다.
- 검증 데이터가 모델 선택에 사용됐다면 외부 평가에는 홀드아웃 예측을 우선 사용합니다.
- 주요 레지스트리 경로는 지정된 백테스트 단계 전반의 검증 샤프를 기준으로 학습 실행을 선택합니다.
- 적격 백테스트 단계가 기록되지 않은 경우 정보계수 순위로 대체 예측을 선택합니다.
- 커밋된 파일도 없고 적격 레지스트리 예측도 없으면 로더는 아티팩트를 반환하지 않습니다.
태그
전문
# demo_artifacts.py
```py
from __future__ import annotations
import polars as pl
from utils.paths import get_case_study_source_dir
def normalize_demo_predictions(df: pl.DataFrame, asset_column: str) -> pl.DataFrame:
"""Normalize prediction artifacts to ``timestamp``, ``asset_column``, ``prediction``."""
rename_map = {}
# Normalize entity column: accept "asset" or "symbol" as source
if asset_column not in df.columns:
if "asset" in df.columns:
rename_map["asset"] = asset_column
elif "symbol" in df.columns:
rename_map["symbol"] = asset_column
if "prediction" not in df.columns and "y_score" in df.columns:
rename_map["y_score"] = "prediction"
# Normalize legacy "date" column to canonical "timestamp"
if "date" in df.columns and "timestamp" not in df.columns:
rename_map["date"] = "timestamp"
if rename_map:
df = df.rename(rename_map)
required = {"timestamp", asset_column, "prediction"}
if required - set(df.columns):
raise ValueError(
f"Prediction artifact missing required columns: {required - set(df.columns)}"
)
ts = df["timestamp"]
if ts.dtype == pl.Utf8:
df = df.with_columns(pl.col("timestamp").str.to_date())
elif ts.dtype in (pl.Datetime, pl.Date):
df = df.with_columns(pl.col("timestamp").dt.date())
return (
df.select(["timestamp", asset_column, "prediction"])
.group_by(["timestamp", asset_column])
.agg(pl.col("prediction").mean().alias("prediction"))
.sort([asset_column, "timestamp"])
)
def load_demo_predictions(strategy_id: str, horizon: int, asset_column: str) -> pl.DataFrame | None:
"""Load prediction artifacts for live-demo notebooks.
Searches committed model directories first, then falls back to the
content-addressed registry (run_log/predictions/) which is the
primary output of the model training pipeline. The registry fallback
returns the sealed **holdout** split — the once-touched out-of-sample
set — never validation, so a deployment export never ships predictions
from the data used to select the model.
"""
import os
import sqlite3
from pathlib import Path
source_dir = get_case_study_source_dir(strategy_id)
models_dir = source_dir / "models"
candidates = [
models_dir / "gbm" / f"fwd_ret_{horizon}d" / "predictions.parquet",
models_dir / "linear" / f"fwd_ret_{horizon}d" / "predictions.parquet",
models_dir / "deep_learning" / f"fwd_ret_{horizon}d" / "predictions.parquet",
models_dir / "tabular_dl" / f"fwd_ret_{horizon}d" / "predictions.parquet",
]
# Also check seeded predictions in ML4T_OUTPUT_DIR (CI / test mode)
output_dir = os.environ.get("ML4T_OUTPUT_DIR", "")
if output_dir:
seeded_dir = Path(output_dir) / strategy_id / "models"
candidates.append(seeded_dir / f"predictions_reg_{horizon}d.parquet")
for path in candidates:
if not path.exists():
continue
df = pl.read_parquet(path)
has_time = "date" in df.columns or "timestamp" in df.columns
has_entity = "asset" in df.columns or "symbol" in df.columns
has_score = "y_score" in df.columns or "prediction" in df.columns
if not (has_time and has_entity and has_score):
continue
return normalize_demo_predictions(df, asset_column)
# Fall back to registry — export the sealed HOLDOUT prediction set, never
# validation. A deployment bridge that exported validation predictions would
# be backtesting on the data used to select the model (leakage). The holdout
# is the once-touched out-of-sample set, so it is the only honest thing to
# ship to an external backtester.
#
# Pick that holdout by the *selected winner*, not by raw IC: the selected configuration is
# the cross-stage validation-Sharpe rank-1 config, pooling the selection
# stages (signal/allocation/risk_overlay). cost_sensitivity is a
# perturbation, not a selection axis, and the holdout stage is the sealed
# set itself, so both are excluded. The winner's training run owns exactly
# one holdout prediction set (the "one holdout per case study" rule), which
# is what we export. This also derives the label/horizon from the winner
# rather than trusting the caller's `horizon`. The IC-sorted query is kept
# only as a fallback for case studies that have predictions but no backtest
# stages recorded.
registry_db = source_dir / "run_log" / "registry.db"
if registry_db.exists():
conn = sqlite3.connect(str(registry_db))
pred_hash = None
winner = conn.execute(
"""SELECT ps.training_hash
FROM backtest_runs br
JOIN backtest_metrics bm ON br.backtest_hash = bm.backtest_hash
JOIN prediction_sets ps ON br.prediction_hash = ps.prediction_hash
WHERE br.stage IN ('signal', 'allocation', 'risk_overlay')
ORDER BY bm.sharpe DESC LIMIT 1""",
).fetchone()
if winner:
row = conn.execute(
"""SELECT prediction_hash FROM prediction_sets
WHERE training_hash = ? AND split = 'holdout'
ORDER BY prediction_hash LIMIT 1""",
(winner[0],),
).fetchone()
if row:
pred_hash = row[0]
if pred_hash is None:
label = f"fwd_ret_{horizon}d"
row = conn.execute(
"""SELECT ps.prediction_hash
FROM prediction_sets ps
JOIN training_runs tr ON ps.training_hash = tr.training_hash
JOIN prediction_metrics pm ON ps.prediction_hash = pm.prediction_hash
WHERE tr.label = ? AND ps.split = 'holdout'
ORDER BY pm.ic_mean DESC LIMIT 1""",
(label,),
).fetchone()
if row:
pred_hash = row[0]
conn.close()
if pred_hash:
pred_path = source_dir / "run_log" / "predictions" / pred_hash / "predictions.parquet"
if pred_path.exists():
df = pl.read_parquet(pred_path)
return normalize_demo_predictions(df, asset_column)
return None
```출처의 라이선스에 따라 출처를 표시하고 전문을 공개합니다. 라이선스: MIT
이 요약은 원문을 바탕으로 Stratmill의 리서치 에이전트가 작성했으며, 원문을 복사한 것이 아닙니다.