为 ATM 跨式期权分析筛选并保留合约
代码 《交易机器学习》
总结
本文概述一种两阶段方法,用于提取 S&P 500 跨式期权研究所需的源观测值。第一阶段根据距到期日天数、绝对 Delta、隐含波动率收敛、最低报价要求和相对买卖价差,筛选出接近平值的候选看涨与看跌期权合约。所选合约以代码、行权价和到期日表示。第二阶段保留这些合约整个生命周期内的每个日度观测,以支持分析同一合约的退出价格和每日 Delta 对冲。
候选筛选会应用于多个日历年,再先进行全局合并,之后筛选生命周期观测。全局并集旨在保留符合条件的时段和到期日跨越年界的合约。既定设计目标是保留足够的原始数据,以便使用匹配的筛选参数重新生成衍生的日度跨式期权数据集。这是数据准备流程,并非交易结果,也不代表筛选能够找到盈利机会。最终样本受指定年份、报价质量阈值、Delta 区间和可用源记录限制;从中得出的结论未必适用于其他时期或筛选规则。
核心观点
- 根据到期时间、绝对 Delta、隐含波动率收敛、最低报价数量和相对价差标准筛选候选合约。
- 选出任何时点满足条件的合约,再保留其直至到期的观测,以分析完整生命周期。
- 在筛选日度数据前合并各年的候选合约,以保留跨越日历年边界的合约。
- 原始数据提取与后续物化采用一致的筛选规则,有助于生成可复现的衍生数据。
- 经过筛选的期权数据集可用于分析,但本身并不能说明跨式期权能够盈利。
标签
全文
# build_options_straddles_raw.py
```py
"""Build ``sp500/options_straddles_raw/`` — source chains for the
sp500_options case study.
Two-pass extraction:
1. Identify every ``(symbol, strike, expiration)`` that passes the 30D ATM
candidate filter (DTE ∈ [25, 35], |delta| ∈ [0.35, 0.65], Converged IV,
bid ≥ 0.01, relative spread ≤ 0.30) at any point in 2017-2021. Filter
parameters match ``compute_straddles()`` in ``materialize_options.py`` so
the derived ``options_straddles_daily.parquet`` is byte-identical when
regenerated from this slice.
2. Emit all daily observations for every candidate contract (both legs,
from first listing through expiration) — the full lifecycle needed for
same-contract exit prices and daily delta hedging.
Output: ``options_straddles_raw/year=YYYY.parquet`` (hive-partitioned).
Run from repo root:
uv run python data/equities/market/sp500/build_options_straddles_raw.py
"""
from __future__ import annotations
import argparse
import gc
import time
from pathlib import Path
import polars as pl
from utils.downloading import resolve_data_dir
def sp500_data_dir(data_path: Path | None = None) -> Path:
"""Where the loaders read this dataset from.
Not ``Path(__file__).parent``: the converter writes under ``$ML4T_DATA_PATH``,
which a reader may point outside the repository, and a build script anchored
to its own directory would then look in the wrong place and leave its output
somewhere the loaders never read.
"""
return resolve_data_dir(data_path) / "equities" / "market" / "sp500"
YEARS = [2017, 2018, 2019, 2020, 2021]
# Must match compute_straddles() in materialize_options.py so the derived
# options_straddles_daily.parquet is byte-identical when regenerated from
# this slim set.
STRADDLE_DTE_WINDOW = (25, 35)
STRADDLE_TARGET_DELTA = 0.50
STRADDLE_DELTA_TOL = 0.15
STRADDLE_MIN_BID = 0.01
STRADDLE_MAX_REL_SPREAD = 0.30
def identify_candidate_contracts(df: pl.DataFrame) -> pl.DataFrame:
"""Return the unique `(symbol, strike, expiration)` triples that pass
the ATM straddle candidate filter at any point in the year.
"""
rel_spread = (pl.col("ask") - pl.col("bid")) / pl.col("mid_price").clip(lower_bound=0.01)
abs_delta = pl.col("delta").abs()
candidates = (
df.filter(
pl.col("days_to_maturity").is_between(*STRADDLE_DTE_WINDOW)
& (pl.col("bid") >= STRADDLE_MIN_BID)
& (pl.col("ask") >= STRADDLE_MIN_BID)
& (rel_spread <= STRADDLE_MAX_REL_SPREAD)
& (pl.col("iv_convergence") == "Converged")
& abs_delta.is_between(
STRADDLE_TARGET_DELTA - STRADDLE_DELTA_TOL,
STRADDLE_TARGET_DELTA + STRADDLE_DELTA_TOL,
)
)
.select(["symbol", "strike", "expiration"])
.unique()
)
return candidates
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
parser.add_argument(
"--data-path",
type=Path,
default=None,
help="Data storage location (default: $ML4T_DATA_PATH or repo/data)",
)
args = parser.parse_args()
base = sp500_data_dir(args.data_path)
RAW_DIR = base / "options"
OUT_DIR = base / "options_straddles_raw"
if not RAW_DIR.exists():
msg = f"Raw options directory not found: {RAW_DIR}"
raise FileNotFoundError(msg)
OUT_DIR.mkdir(parents=True, exist_ok=True)
print("=" * 60)
print("Building SP500 options straddles raw slice (ATM-band, lifecycle-preserving)")
print(f" Source: {RAW_DIR}")
print(f" Output: {OUT_DIR}")
print(f" Years: {YEARS}")
print(
f" Filter: DTE ∈ [{STRADDLE_DTE_WINDOW[0]}, {STRADDLE_DTE_WINDOW[1]}], "
f"|delta| ∈ [{STRADDLE_TARGET_DELTA - STRADDLE_DELTA_TOL:.2f}, "
f"{STRADDLE_TARGET_DELTA + STRADDLE_DELTA_TOL:.2f}], Converged IV"
)
print("=" * 60)
total_raw = 0
total_kept = 0
total_size_mb = 0.0
overall_start = time.time()
for year in YEARS:
t0 = time.time()
pattern = f"year={year}/*.parquet"
files = list(RAW_DIR.glob(pattern))
if not files:
print(f"\n[{year}] No files — skipping")
continue
print(f"\n[{year}] Loading {len(files)} partitions …")
df = pl.read_parquet(RAW_DIR / pattern, hive_partitioning=True)
n_raw = len(df)
total_raw += n_raw
print(f" Raw: {n_raw:,} rows, {df['symbol'].n_unique()} symbols")
candidates = identify_candidate_contracts(df)
n_candidates = len(candidates)
print(f" Candidate contracts (year-local): {n_candidates:,}")
# Also include candidates identified in adjacent years whose expirations
# fall within this year — a contract can enter the 25-35 DTE window in
# one year and be held across a year boundary. Safer to process all
# candidates globally in one final join, so do that below.
candidates.write_parquet(OUT_DIR / f"_candidates_{year}.parquet")
del df, candidates
gc.collect()
print(f" Year {year} candidate scan: {time.time() - t0:.0f}s")
# Global candidate union (so cross-year lifecycles are preserved)
print("\nBuilding global candidate union …")
candidate_files = [OUT_DIR / f"_candidates_{y}.parquet" for y in YEARS]
candidate_files = [p for p in candidate_files if p.exists()]
all_candidates = (
pl.concat([pl.read_parquet(p) for p in candidate_files])
.unique()
.sort(["symbol", "expiration", "strike"])
)
print(f" Total unique candidate contracts: {len(all_candidates):,}")
# Pass 2: second sweep, keeping every daily observation of any candidate
# contract (both legs, entry through expiration).
for year in YEARS:
pattern = f"year={year}/*.parquet"
files = list(RAW_DIR.glob(pattern))
if not files:
continue
t0 = time.time()
print(f"\n[{year}] Filtering to lifecycle observations …")
df = pl.read_parquet(RAW_DIR / pattern, hive_partitioning=True)
kept = df.join(all_candidates, on=["symbol", "strike", "expiration"], how="semi").sort(
["symbol", "date", "expiration", "call_put", "strike"]
)
out_path = OUT_DIR / f"year={year}.parquet"
kept.write_parquet(out_path, compression="zstd", compression_level=22, statistics=True)
size_mb = out_path.stat().st_size / 1024 / 1024
total_kept += len(kept)
total_size_mb += size_mb
print(
f" {year}: {len(kept):,} rows kept ({len(kept) / len(df):.1%}), "
f"{size_mb:.1f} MB, {time.time() - t0:.0f}s"
)
del df, kept
gc.collect()
# Clean up interim candidate files
for p in candidate_files:
p.unlink()
print()
print("=" * 60)
print(f"Total raw rows: {total_raw:,}")
print(f"Total kept rows: {total_kept:,} ({total_kept / total_raw:.1%})")
print(f"Total output size: {total_size_mb:.1f} MB")
print(f"Overall elapsed: {time.time() - overall_start:.0f}s")
print(f"Output: {OUT_DIR}")
if __name__ == "__main__":
main()
```在遵守原作品许可的前提下,附作者信息全文展示。 许可协议: MIT
此摘要由 Stratmill 研究智能体根据原文撰写,并非原文副本。