Coleta e alinhamento de liquidações de funding de futuros perpétuos
Resumo
Este documento descreve um pipeline para baixar registros oficiais de funding perpétuo Binance USD-M, normalizá-los e armazená-los em cache em um arquivo colunar. Arquivos mensais são buscados simultaneamente para os símbolos e o intervalo de datas solicitados. Os registros são convertidos em timestamps UTC, padronizados em campos de liquidação, deduplicados por símbolo e horário e verificados quanto a valores ausentes ou inválidos antes de serem salvos. Os carregadores permitem filtros opcionais por símbolo e data.
Para backtests, os registros de funding são ainda restritos às chaves exatas de símbolo e timestamp presentes no painel de preços. Isso importa porque os custos de funding são cobrados sobre posições mantidas na liquidação, e as liquidações relevantes variam quando a execução usa uma janela de datas diferente. O pipeline gera um erro se nenhuma liquidação corresponder aos preços. O documento explica o tratamento e o alinhamento dos dados, não um sinal de trading ou uma estratégia empírica; não fornece evidências de desempenho. A fonte de dados são contratos perpétuos USD-M de uma única corretora, então os registros descritos não comprovam cobertura de outras plataformas ou derivativos.
Ideias principais
- Arquivos de funding são coletados por símbolo e mês e depois normalizados em registros de liquidação UTC.
- O pipeline deduplica pares de símbolo e timestamp e rejeita taxas ou intervalos de liquidação inválidos.
- Os backtests devem alinhar as liquidações de funding às chaves exatas cobertas por seu painel de preços.
- Os custos de funding dependem das posições mantidas na liquidação; por isso, os registros aplicáveis variam conforme a janela do backtest.
- A fonte descrita cobre dados de funding perpétuo Binance USD-M, e não todas as plataformas ou instrumentos.
Tags
Texto completo
# funding_data.py
```py
"""Official Binance USD-M perpetual funding-rate data for the crypto case study."""
from __future__ import annotations
import io
import zipfile
from concurrent.futures import ThreadPoolExecutor, as_completed
from datetime import UTC, date, datetime
from pathlib import Path
import httpx
import polars as pl
import yaml
from data.exceptions import DataNotFoundError
from utils import ML4T_DATA_PATH
from utils.paths import get_case_study_dir
BASE_URL = "https://data.binance.vision/data/futures/um/monthly/fundingRate"
FUNDING_PATH = ML4T_DATA_PATH / "crypto" / "market" / "funding_rate.parquet"
def _month_keys(start_date: str, end_date: str) -> list[str]:
start = datetime.strptime(start_date, "%Y-%m-%d")
end = datetime.strptime(end_date, "%Y-%m-%d")
keys = []
year, month = start.year, start.month
while (year, month) <= (end.year, end.month):
keys.append(f"{year}-{month:02d}")
month += 1
if month == 13:
year += 1
month = 1
return keys
def _fetch_month(symbol: str, month: str) -> tuple[pl.DataFrame | None, bool]:
url = f"{BASE_URL}/{symbol}/{symbol}-fundingRate-{month}.zip"
response = httpx.get(url, timeout=30, follow_redirects=True)
if response.status_code == 404:
return None, True
response.raise_for_status()
with zipfile.ZipFile(io.BytesIO(response.content)) as archive:
names = archive.namelist()
if len(names) != 1:
raise ValueError(f"Expected one CSV in {url}, found {names}")
frame = pl.read_csv(io.BytesIO(archive.read(names[0]))).with_columns(
pl.lit(symbol).alias("symbol")
)
required = {"calc_time", "funding_interval_hours", "last_funding_rate", "symbol"}
missing = required - set(frame.columns)
if missing:
raise ValueError(f"Funding archive schema changed for {url}: missing {sorted(missing)}")
return frame, False
def _normalize_funding(frames: list[pl.DataFrame]) -> pl.DataFrame:
funding = (
pl.concat(frames, how="diagonal_relaxed")
.with_columns(
pl.from_epoch("calc_time", time_unit="ms")
.dt.replace_time_zone("UTC")
.cast(pl.Datetime("ms", "UTC"))
.dt.truncate("1h")
.alias("timestamp"),
pl.col("funding_interval_hours").cast(pl.Int16),
pl.col("last_funding_rate").cast(pl.Float64).alias("funding_rate"),
)
.select("timestamp", "symbol", "funding_interval_hours", "funding_rate")
.unique(["timestamp", "symbol"], maintain_order=False)
.sort("timestamp", "symbol")
)
if funding.is_empty():
raise ValueError("Funding archive produced no rows")
if funding.select(pl.struct("timestamp", "symbol").is_duplicated().any()).item():
raise ValueError("Funding archive contains duplicate timestamp-symbol keys")
invalid = funding.filter(
pl.col("funding_rate").is_nan()
| pl.col("funding_rate").is_infinite()
| (pl.col("funding_interval_hours") <= 0)
| (pl.col("funding_interval_hours") > 8)
)
if invalid.height:
raise ValueError(f"Funding archive contains {invalid.height} invalid rows")
return funding
def download_funding_rates(
symbols: list[str],
start_date: str,
end_date: str,
*,
output_path: Path = FUNDING_PATH,
max_workers: int = 12,
) -> pl.DataFrame:
"""Download monthly official funding settlements and atomically cache them."""
tasks = [(symbol, month) for symbol in symbols for month in _month_keys(start_date, end_date)]
frames: list[pl.DataFrame] = []
missing = 0
print(f"Downloading {len(tasks):,} Binance funding-rate archives", flush=True)
with ThreadPoolExecutor(max_workers=max_workers) as pool:
futures = {
pool.submit(_fetch_month, symbol, month): (symbol, month) for symbol, month in tasks
}
for index, future in enumerate(as_completed(futures), start=1):
frame, not_listed = future.result()
if frame is not None:
frames.append(frame)
elif not_listed:
missing += 1
if index % 100 == 0 or index == len(tasks):
print(
f" completed {index:,}/{len(tasks):,}; files={len(frames):,}; "
f"unavailable={missing:,}",
flush=True,
)
funding = _normalize_funding(frames)
output_path.parent.mkdir(parents=True, exist_ok=True)
temporary_path = output_path.with_suffix(".tmp.parquet")
funding.write_parquet(temporary_path)
temporary_path.replace(output_path)
print(
f"Saved {len(funding):,} settlements for {funding['symbol'].n_unique()} symbols "
f"to {output_path}",
flush=True,
)
return funding
def load_funding_rates(
*,
symbols: list[str] | None = None,
start_date: str | None = None,
end_date: str | None = None,
path: Path = FUNDING_PATH,
) -> pl.DataFrame:
"""Load official funding settlements with optional symbol and date filters."""
if not path.exists():
raise DataNotFoundError(
dataset_name="Binance USD-M Perpetual Funding Rates",
path=path,
download_script="case_studies/crypto_perps_funding/funding_data.py",
readme="case_studies/crypto_perps_funding/README.md",
)
funding = pl.read_parquet(path)
if symbols:
funding = funding.filter(pl.col("symbol").is_in(symbols))
if start_date:
funding = funding.filter(pl.col("timestamp").dt.date() >= pl.lit(start_date).str.to_date())
if end_date:
funding = funding.filter(pl.col("timestamp").dt.date() <= pl.lit(end_date).str.to_date())
return funding.sort("timestamp", "symbol")
def funding_rates_for_prices(prices: pl.DataFrame) -> pl.DataFrame:
"""Official funding settlements restricted to the exact keys ``prices`` covers.
The backtest charges funding against the position held at each settlement, so the
settlements a run is identified by have to be the ones its own price panel spans. A
holdout run slices a different window from the validation run it inherits its
configuration from, which is why this is derived from the frame rather than carried
across with the rest of the specification.
"""
def _day(value: object) -> str:
if isinstance(value, datetime):
return value.date().isoformat()
if isinstance(value, date):
return value.isoformat()
raise TypeError(f"expected a date-like timestamp, got {type(value).__name__}")
timestamp_dtype = prices.schema["timestamp"]
price_keys = prices.select("symbol", "timestamp").unique()
symbols = price_keys.get_column("symbol").unique().to_list()
funding = load_funding_rates(
symbols=symbols,
start_date=_day(price_keys.get_column("timestamp").min()),
end_date=_day(price_keys.get_column("timestamp").max()),
)
funding = funding.with_columns(
pl.col("symbol").cast(price_keys.schema["symbol"]),
pl.col("timestamp").cast(timestamp_dtype),
).join(price_keys, on=["symbol", "timestamp"], how="semi")
if funding.is_empty():
raise ValueError("crypto backtest resolved no official funding settlements")
return funding.sort("timestamp", "symbol")
def main() -> None:
case_dir = get_case_study_dir("crypto_perps_funding")
setup = yaml.safe_load((case_dir / "config" / "setup.yaml").read_text())
download_funding_rates(
list(setup["universe"]["symbols"]),
"2020-01-01",
"2025-12-31",
)
print(f"Completed at {datetime.now(UTC).isoformat()}", flush=True)
if __name__ == "__main__":
main()
```Exibido na íntegra, com atribuição conforme a licença da fonte. Licença: MIT
Este resumo foi escrito pelo agente de pesquisa da Stratmill com base no original; não é uma cópia da fonte.