跳至正文
返回文库全部文档

为 ATM 跨式期权分析筛选并保留合约

代码 《交易机器学习》

总结

本文概述一种两阶段方法,用于提取 S&P 500 跨式期权研究所需的源观测值。第一阶段根据距到期日天数、绝对 Delta、隐含波动率收敛、最低报价要求和相对买卖价差,筛选出接近平值的候选看涨与看跌期权合约。所选合约以代码、行权价和到期日表示。第二阶段保留这些合约整个生命周期内的每个日度观测,以支持分析同一合约的退出价格和每日 Delta 对冲。

候选筛选会应用于多个日历年,再先进行全局合并,之后筛选生命周期观测。全局并集旨在保留符合条件的时段和到期日跨越年界的合约。既定设计目标是保留足够的原始数据,以便使用匹配的筛选参数重新生成衍生的日度跨式期权数据集。这是数据准备流程,并非交易结果,也不代表筛选能够找到盈利机会。最终样本受指定年份、报价质量阈值、Delta 区间和可用源记录限制;从中得出的结论未必适用于其他时期或筛选规则。

核心观点

  • 根据到期时间、绝对 Delta、隐含波动率收敛、最低报价数量和相对价差标准筛选候选合约。
  • 选出任何时点满足条件的合约,再保留其直至到期的观测,以分析完整生命周期。
  • 在筛选日度数据前合并各年的候选合约,以保留跨越日历年边界的合约。
  • 原始数据提取与后续物化采用一致的筛选规则,有助于生成可复现的衍生数据。
  • 经过筛选的期权数据集可用于分析,但本身并不能说明跨式期权能够盈利。

标签

全文
# build_options_straddles_raw.py


```py
"""Build ``sp500/options_straddles_raw/`` — source chains for the
sp500_options case study.

Two-pass extraction:

1. Identify every ``(symbol, strike, expiration)`` that passes the 30D ATM
   candidate filter (DTE ∈ [25, 35], |delta| ∈ [0.35, 0.65], Converged IV,
   bid ≥ 0.01, relative spread ≤ 0.30) at any point in 2017-2021. Filter
   parameters match ``compute_straddles()`` in ``materialize_options.py`` so
   the derived ``options_straddles_daily.parquet`` is byte-identical when
   regenerated from this slice.
2. Emit all daily observations for every candidate contract (both legs,
   from first listing through expiration) — the full lifecycle needed for
   same-contract exit prices and daily delta hedging.

Output: ``options_straddles_raw/year=YYYY.parquet`` (hive-partitioned).

Run from repo root:

    uv run python data/equities/market/sp500/build_options_straddles_raw.py
"""

from __future__ import annotations

import argparse
import gc
import time
from pathlib import Path

import polars as pl

from utils.downloading import resolve_data_dir


def sp500_data_dir(data_path: Path | None = None) -> Path:
    """Where the loaders read this dataset from.

    Not ``Path(__file__).parent``: the converter writes under ``$ML4T_DATA_PATH``,
    which a reader may point outside the repository, and a build script anchored
    to its own directory would then look in the wrong place and leave its output
    somewhere the loaders never read.
    """
    return resolve_data_dir(data_path) / "equities" / "market" / "sp500"


YEARS = [2017, 2018, 2019, 2020, 2021]

# Must match compute_straddles() in materialize_options.py so the derived
# options_straddles_daily.parquet is byte-identical when regenerated from
# this slim set.
STRADDLE_DTE_WINDOW = (25, 35)
STRADDLE_TARGET_DELTA = 0.50
STRADDLE_DELTA_TOL = 0.15
STRADDLE_MIN_BID = 0.01
STRADDLE_MAX_REL_SPREAD = 0.30


def identify_candidate_contracts(df: pl.DataFrame) -> pl.DataFrame:
    """Return the unique `(symbol, strike, expiration)` triples that pass
    the ATM straddle candidate filter at any point in the year.
    """
    rel_spread = (pl.col("ask") - pl.col("bid")) / pl.col("mid_price").clip(lower_bound=0.01)
    abs_delta = pl.col("delta").abs()

    candidates = (
        df.filter(
            pl.col("days_to_maturity").is_between(*STRADDLE_DTE_WINDOW)
            & (pl.col("bid") >= STRADDLE_MIN_BID)
            & (pl.col("ask") >= STRADDLE_MIN_BID)
            & (rel_spread <= STRADDLE_MAX_REL_SPREAD)
            & (pl.col("iv_convergence") == "Converged")
            & abs_delta.is_between(
                STRADDLE_TARGET_DELTA - STRADDLE_DELTA_TOL,
                STRADDLE_TARGET_DELTA + STRADDLE_DELTA_TOL,
            )
        )
        .select(["symbol", "strike", "expiration"])
        .unique()
    )
    return candidates


def main() -> None:
    parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
    parser.add_argument(
        "--data-path",
        type=Path,
        default=None,
        help="Data storage location (default: $ML4T_DATA_PATH or repo/data)",
    )
    args = parser.parse_args()

    base = sp500_data_dir(args.data_path)
    RAW_DIR = base / "options"
    OUT_DIR = base / "options_straddles_raw"

    if not RAW_DIR.exists():
        msg = f"Raw options directory not found: {RAW_DIR}"
        raise FileNotFoundError(msg)

    OUT_DIR.mkdir(parents=True, exist_ok=True)

    print("=" * 60)
    print("Building SP500 options straddles raw slice (ATM-band, lifecycle-preserving)")
    print(f"  Source: {RAW_DIR}")
    print(f"  Output: {OUT_DIR}")
    print(f"  Years: {YEARS}")
    print(
        f"  Filter: DTE ∈ [{STRADDLE_DTE_WINDOW[0]}, {STRADDLE_DTE_WINDOW[1]}], "
        f"|delta| ∈ [{STRADDLE_TARGET_DELTA - STRADDLE_DELTA_TOL:.2f}, "
        f"{STRADDLE_TARGET_DELTA + STRADDLE_DELTA_TOL:.2f}], Converged IV"
    )
    print("=" * 60)

    total_raw = 0
    total_kept = 0
    total_size_mb = 0.0
    overall_start = time.time()

    for year in YEARS:
        t0 = time.time()
        pattern = f"year={year}/*.parquet"
        files = list(RAW_DIR.glob(pattern))
        if not files:
            print(f"\n[{year}] No files — skipping")
            continue

        print(f"\n[{year}] Loading {len(files)} partitions …")
        df = pl.read_parquet(RAW_DIR / pattern, hive_partitioning=True)
        n_raw = len(df)
        total_raw += n_raw
        print(f"  Raw: {n_raw:,} rows, {df['symbol'].n_unique()} symbols")

        candidates = identify_candidate_contracts(df)
        n_candidates = len(candidates)
        print(f"  Candidate contracts (year-local): {n_candidates:,}")

        # Also include candidates identified in adjacent years whose expirations
        # fall within this year — a contract can enter the 25-35 DTE window in
        # one year and be held across a year boundary. Safer to process all
        # candidates globally in one final join, so do that below.
        candidates.write_parquet(OUT_DIR / f"_candidates_{year}.parquet")
        del df, candidates
        gc.collect()
        print(f"  Year {year} candidate scan: {time.time() - t0:.0f}s")

    # Global candidate union (so cross-year lifecycles are preserved)
    print("\nBuilding global candidate union …")
    candidate_files = [OUT_DIR / f"_candidates_{y}.parquet" for y in YEARS]
    candidate_files = [p for p in candidate_files if p.exists()]
    all_candidates = (
        pl.concat([pl.read_parquet(p) for p in candidate_files])
        .unique()
        .sort(["symbol", "expiration", "strike"])
    )
    print(f"  Total unique candidate contracts: {len(all_candidates):,}")

    # Pass 2: second sweep, keeping every daily observation of any candidate
    # contract (both legs, entry through expiration).
    for year in YEARS:
        pattern = f"year={year}/*.parquet"
        files = list(RAW_DIR.glob(pattern))
        if not files:
            continue

        t0 = time.time()
        print(f"\n[{year}] Filtering to lifecycle observations …")
        df = pl.read_parquet(RAW_DIR / pattern, hive_partitioning=True)
        kept = df.join(all_candidates, on=["symbol", "strike", "expiration"], how="semi").sort(
            ["symbol", "date", "expiration", "call_put", "strike"]
        )

        out_path = OUT_DIR / f"year={year}.parquet"
        kept.write_parquet(out_path, compression="zstd", compression_level=22, statistics=True)
        size_mb = out_path.stat().st_size / 1024 / 1024
        total_kept += len(kept)
        total_size_mb += size_mb
        print(
            f"  {year}: {len(kept):,} rows kept ({len(kept) / len(df):.1%}), "
            f"{size_mb:.1f} MB, {time.time() - t0:.0f}s"
        )
        del df, kept
        gc.collect()

    # Clean up interim candidate files
    for p in candidate_files:
        p.unlink()

    print()
    print("=" * 60)
    print(f"Total raw rows: {total_raw:,}")
    print(f"Total kept rows: {total_kept:,} ({total_kept / total_raw:.1%})")
    print(f"Total output size: {total_size_mb:.1f} MB")
    print(f"Overall elapsed: {time.time() - overall_start:.0f}s")
    print(f"Output: {OUT_DIR}")


if __name__ == "__main__":
    main()

```

在遵守原作品许可的前提下,附作者信息全文展示。 许可协议: MIT

此摘要由 Stratmill 研究智能体根据原文撰写,并非原文副本。