無期限先物の資金調達決済データの収集と整合
コード Machine Learning for Trading
サマリー
この文書では、Binanceの公式USD-M無期限先物の資金調達記録をダウンロードし、標準化して列指向ファイルにキャッシュするパイプラインを説明します。指定した銘柄と日付範囲について、月次アーカイブを並行して取得します。記録をUTCタイムスタンプに変換し、決済項目を標準化して銘柄と時刻で重複を除き、保存前に欠損値や無効値がないか確認します。読み込み時には銘柄と日付を任意で指定できます。
バックテストでは、資金調達記録を価格パネル内の銘柄とタイムスタンプのキーに厳密に合わせます。決済時に保有するポジションに資金調達コストが課され、実行ごとに日付範囲が異なれば対象となる決済も変わるためです。価格に一致する決済がない場合、パイプラインはエラーを返します。この文書が説明するのはデータ処理と時刻の整合であり、取引シグナルや実証的な戦略ではありません。性能を示す証拠もありません。データソースは1つの取引所のUSD-M無期限先物であり、他の取引所やデリバティブを網羅するとは限りません。
主なアイデア
- 銘柄ごと、月ごとに資金調達アーカイブを収集し、UTCの決済記録に標準化します。
- 銘柄とタイムスタンプの組み合わせの重複を除き、無効な料率や決済間隔を除外します。
- バックテストでは、資金調達決済を価格パネルが対象とするキーに正確に合わせます。
- 資金調達コストは決済時のポジションに依存するため、対象記録はバックテスト期間によって異なります。
- 記載されたデータソースはBinanceのUSD-M無期限先物の資金調達データであり、すべての取引所や金融商品を網羅するものではありません。
タグ
全文
# funding_data.py
```py
"""Official Binance USD-M perpetual funding-rate data for the crypto case study."""
from __future__ import annotations
import io
import zipfile
from concurrent.futures import ThreadPoolExecutor, as_completed
from datetime import UTC, date, datetime
from pathlib import Path
import httpx
import polars as pl
import yaml
from data.exceptions import DataNotFoundError
from utils import ML4T_DATA_PATH
from utils.paths import get_case_study_dir
BASE_URL = "https://data.binance.vision/data/futures/um/monthly/fundingRate"
FUNDING_PATH = ML4T_DATA_PATH / "crypto" / "market" / "funding_rate.parquet"
def _month_keys(start_date: str, end_date: str) -> list[str]:
start = datetime.strptime(start_date, "%Y-%m-%d")
end = datetime.strptime(end_date, "%Y-%m-%d")
keys = []
year, month = start.year, start.month
while (year, month) <= (end.year, end.month):
keys.append(f"{year}-{month:02d}")
month += 1
if month == 13:
year += 1
month = 1
return keys
def _fetch_month(symbol: str, month: str) -> tuple[pl.DataFrame | None, bool]:
url = f"{BASE_URL}/{symbol}/{symbol}-fundingRate-{month}.zip"
response = httpx.get(url, timeout=30, follow_redirects=True)
if response.status_code == 404:
return None, True
response.raise_for_status()
with zipfile.ZipFile(io.BytesIO(response.content)) as archive:
names = archive.namelist()
if len(names) != 1:
raise ValueError(f"Expected one CSV in {url}, found {names}")
frame = pl.read_csv(io.BytesIO(archive.read(names[0]))).with_columns(
pl.lit(symbol).alias("symbol")
)
required = {"calc_time", "funding_interval_hours", "last_funding_rate", "symbol"}
missing = required - set(frame.columns)
if missing:
raise ValueError(f"Funding archive schema changed for {url}: missing {sorted(missing)}")
return frame, False
def _normalize_funding(frames: list[pl.DataFrame]) -> pl.DataFrame:
funding = (
pl.concat(frames, how="diagonal_relaxed")
.with_columns(
pl.from_epoch("calc_time", time_unit="ms")
.dt.replace_time_zone("UTC")
.cast(pl.Datetime("ms", "UTC"))
.dt.truncate("1h")
.alias("timestamp"),
pl.col("funding_interval_hours").cast(pl.Int16),
pl.col("last_funding_rate").cast(pl.Float64).alias("funding_rate"),
)
.select("timestamp", "symbol", "funding_interval_hours", "funding_rate")
.unique(["timestamp", "symbol"], maintain_order=False)
.sort("timestamp", "symbol")
)
if funding.is_empty():
raise ValueError("Funding archive produced no rows")
if funding.select(pl.struct("timestamp", "symbol").is_duplicated().any()).item():
raise ValueError("Funding archive contains duplicate timestamp-symbol keys")
invalid = funding.filter(
pl.col("funding_rate").is_nan()
| pl.col("funding_rate").is_infinite()
| (pl.col("funding_interval_hours") <= 0)
| (pl.col("funding_interval_hours") > 8)
)
if invalid.height:
raise ValueError(f"Funding archive contains {invalid.height} invalid rows")
return funding
def download_funding_rates(
symbols: list[str],
start_date: str,
end_date: str,
*,
output_path: Path = FUNDING_PATH,
max_workers: int = 12,
) -> pl.DataFrame:
"""Download monthly official funding settlements and atomically cache them."""
tasks = [(symbol, month) for symbol in symbols for month in _month_keys(start_date, end_date)]
frames: list[pl.DataFrame] = []
missing = 0
print(f"Downloading {len(tasks):,} Binance funding-rate archives", flush=True)
with ThreadPoolExecutor(max_workers=max_workers) as pool:
futures = {
pool.submit(_fetch_month, symbol, month): (symbol, month) for symbol, month in tasks
}
for index, future in enumerate(as_completed(futures), start=1):
frame, not_listed = future.result()
if frame is not None:
frames.append(frame)
elif not_listed:
missing += 1
if index % 100 == 0 or index == len(tasks):
print(
f" completed {index:,}/{len(tasks):,}; files={len(frames):,}; "
f"unavailable={missing:,}",
flush=True,
)
funding = _normalize_funding(frames)
output_path.parent.mkdir(parents=True, exist_ok=True)
temporary_path = output_path.with_suffix(".tmp.parquet")
funding.write_parquet(temporary_path)
temporary_path.replace(output_path)
print(
f"Saved {len(funding):,} settlements for {funding['symbol'].n_unique()} symbols "
f"to {output_path}",
flush=True,
)
return funding
def load_funding_rates(
*,
symbols: list[str] | None = None,
start_date: str | None = None,
end_date: str | None = None,
path: Path = FUNDING_PATH,
) -> pl.DataFrame:
"""Load official funding settlements with optional symbol and date filters."""
if not path.exists():
raise DataNotFoundError(
dataset_name="Binance USD-M Perpetual Funding Rates",
path=path,
download_script="case_studies/crypto_perps_funding/funding_data.py",
readme="case_studies/crypto_perps_funding/README.md",
)
funding = pl.read_parquet(path)
if symbols:
funding = funding.filter(pl.col("symbol").is_in(symbols))
if start_date:
funding = funding.filter(pl.col("timestamp").dt.date() >= pl.lit(start_date).str.to_date())
if end_date:
funding = funding.filter(pl.col("timestamp").dt.date() <= pl.lit(end_date).str.to_date())
return funding.sort("timestamp", "symbol")
def funding_rates_for_prices(prices: pl.DataFrame) -> pl.DataFrame:
"""Official funding settlements restricted to the exact keys ``prices`` covers.
The backtest charges funding against the position held at each settlement, so the
settlements a run is identified by have to be the ones its own price panel spans. A
holdout run slices a different window from the validation run it inherits its
configuration from, which is why this is derived from the frame rather than carried
across with the rest of the specification.
"""
def _day(value: object) -> str:
if isinstance(value, datetime):
return value.date().isoformat()
if isinstance(value, date):
return value.isoformat()
raise TypeError(f"expected a date-like timestamp, got {type(value).__name__}")
timestamp_dtype = prices.schema["timestamp"]
price_keys = prices.select("symbol", "timestamp").unique()
symbols = price_keys.get_column("symbol").unique().to_list()
funding = load_funding_rates(
symbols=symbols,
start_date=_day(price_keys.get_column("timestamp").min()),
end_date=_day(price_keys.get_column("timestamp").max()),
)
funding = funding.with_columns(
pl.col("symbol").cast(price_keys.schema["symbol"]),
pl.col("timestamp").cast(timestamp_dtype),
).join(price_keys, on=["symbol", "timestamp"], how="semi")
if funding.is_empty():
raise ValueError("crypto backtest resolved no official funding settlements")
return funding.sort("timestamp", "symbol")
def main() -> None:
case_dir = get_case_study_dir("crypto_perps_funding")
setup = yaml.safe_load((case_dir / "config" / "setup.yaml").read_text())
download_funding_rates(
list(setup["universe"]["symbols"]),
"2020-01-01",
"2025-12-31",
)
print(f"Completed at {datetime.now(UTC).isoformat()}", flush=True)
if __name__ == "__main__":
main()
```出典を明記したうえで、ライセンスに従って全文を掲載しています。 ライセンス: MIT
この要約は原文をもとにStratmillのリサーチエージェントが作成したもので、出典の複製ではありません。