Collecting and Aligning Perpetual Futures Funding Settlements
Summary
This document describes a pipeline for downloading official Binance USD-M perpetual funding records, normalizing them, and caching them in a columnar file. Monthly archives are fetched concurrently for the requested symbols and date range. The records are converted to UTC timestamps, standardized into settlement fields, deduplicated by symbol and time, and checked for missing or invalid values before being saved. Loaders allow optional symbol and date filters.
For backtests, the funding records are further restricted to the exact symbol and timestamp keys in the price panel. This matters because funding costs are charged against positions held at settlement, and the relevant settlements differ when a run uses a different date window. The pipeline raises an error if no settlements match the prices. The document explains data handling and alignment rather than a trading signal or empirical strategy; it supplies no performance evidence. Its data source is one exchange’s USD-M perpetual contracts, so the described records do not establish coverage of other venues or derivatives.
Key ideas
- Funding archives are collected by symbol and month, then normalized into UTC settlement records.
- The pipeline deduplicates symbol and timestamp pairs and rejects invalid rates or settlement intervals.
- Backtests should align funding settlements to the exact keys covered by their price panel.
- Funding costs depend on positions held at settlement, so the applicable records vary with the backtest window.
- The described source covers Binance USD-M perpetual funding data rather than all venues or instruments.
Tags
Full text
# funding_data.py
```py
"""Official Binance USD-M perpetual funding-rate data for the crypto case study."""
from __future__ import annotations
import io
import zipfile
from concurrent.futures import ThreadPoolExecutor, as_completed
from datetime import UTC, date, datetime
from pathlib import Path
import httpx
import polars as pl
import yaml
from data.exceptions import DataNotFoundError
from utils import ML4T_DATA_PATH
from utils.paths import get_case_study_dir
BASE_URL = "https://data.binance.vision/data/futures/um/monthly/fundingRate"
FUNDING_PATH = ML4T_DATA_PATH / "crypto" / "market" / "funding_rate.parquet"
def _month_keys(start_date: str, end_date: str) -> list[str]:
start = datetime.strptime(start_date, "%Y-%m-%d")
end = datetime.strptime(end_date, "%Y-%m-%d")
keys = []
year, month = start.year, start.month
while (year, month) <= (end.year, end.month):
keys.append(f"{year}-{month:02d}")
month += 1
if month == 13:
year += 1
month = 1
return keys
def _fetch_month(symbol: str, month: str) -> tuple[pl.DataFrame | None, bool]:
url = f"{BASE_URL}/{symbol}/{symbol}-fundingRate-{month}.zip"
response = httpx.get(url, timeout=30, follow_redirects=True)
if response.status_code == 404:
return None, True
response.raise_for_status()
with zipfile.ZipFile(io.BytesIO(response.content)) as archive:
names = archive.namelist()
if len(names) != 1:
raise ValueError(f"Expected one CSV in {url}, found {names}")
frame = pl.read_csv(io.BytesIO(archive.read(names[0]))).with_columns(
pl.lit(symbol).alias("symbol")
)
required = {"calc_time", "funding_interval_hours", "last_funding_rate", "symbol"}
missing = required - set(frame.columns)
if missing:
raise ValueError(f"Funding archive schema changed for {url}: missing {sorted(missing)}")
return frame, False
def _normalize_funding(frames: list[pl.DataFrame]) -> pl.DataFrame:
funding = (
pl.concat(frames, how="diagonal_relaxed")
.with_columns(
pl.from_epoch("calc_time", time_unit="ms")
.dt.replace_time_zone("UTC")
.cast(pl.Datetime("ms", "UTC"))
.dt.truncate("1h")
.alias("timestamp"),
pl.col("funding_interval_hours").cast(pl.Int16),
pl.col("last_funding_rate").cast(pl.Float64).alias("funding_rate"),
)
.select("timestamp", "symbol", "funding_interval_hours", "funding_rate")
.unique(["timestamp", "symbol"], maintain_order=False)
.sort("timestamp", "symbol")
)
if funding.is_empty():
raise ValueError("Funding archive produced no rows")
if funding.select(pl.struct("timestamp", "symbol").is_duplicated().any()).item():
raise ValueError("Funding archive contains duplicate timestamp-symbol keys")
invalid = funding.filter(
pl.col("funding_rate").is_nan()
| pl.col("funding_rate").is_infinite()
| (pl.col("funding_interval_hours") <= 0)
| (pl.col("funding_interval_hours") > 8)
)
if invalid.height:
raise ValueError(f"Funding archive contains {invalid.height} invalid rows")
return funding
def download_funding_rates(
symbols: list[str],
start_date: str,
end_date: str,
*,
output_path: Path = FUNDING_PATH,
max_workers: int = 12,
) -> pl.DataFrame:
"""Download monthly official funding settlements and atomically cache them."""
tasks = [(symbol, month) for symbol in symbols for month in _month_keys(start_date, end_date)]
frames: list[pl.DataFrame] = []
missing = 0
print(f"Downloading {len(tasks):,} Binance funding-rate archives", flush=True)
with ThreadPoolExecutor(max_workers=max_workers) as pool:
futures = {
pool.submit(_fetch_month, symbol, month): (symbol, month) for symbol, month in tasks
}
for index, future in enumerate(as_completed(futures), start=1):
frame, not_listed = future.result()
if frame is not None:
frames.append(frame)
elif not_listed:
missing += 1
if index % 100 == 0 or index == len(tasks):
print(
f" completed {index:,}/{len(tasks):,}; files={len(frames):,}; "
f"unavailable={missing:,}",
flush=True,
)
funding = _normalize_funding(frames)
output_path.parent.mkdir(parents=True, exist_ok=True)
temporary_path = output_path.with_suffix(".tmp.parquet")
funding.write_parquet(temporary_path)
temporary_path.replace(output_path)
print(
f"Saved {len(funding):,} settlements for {funding['symbol'].n_unique()} symbols "
f"to {output_path}",
flush=True,
)
return funding
def load_funding_rates(
*,
symbols: list[str] | None = None,
start_date: str | None = None,
end_date: str | None = None,
path: Path = FUNDING_PATH,
) -> pl.DataFrame:
"""Load official funding settlements with optional symbol and date filters."""
if not path.exists():
raise DataNotFoundError(
dataset_name="Binance USD-M Perpetual Funding Rates",
path=path,
download_script="case_studies/crypto_perps_funding/funding_data.py",
readme="case_studies/crypto_perps_funding/README.md",
)
funding = pl.read_parquet(path)
if symbols:
funding = funding.filter(pl.col("symbol").is_in(symbols))
if start_date:
funding = funding.filter(pl.col("timestamp").dt.date() >= pl.lit(start_date).str.to_date())
if end_date:
funding = funding.filter(pl.col("timestamp").dt.date() <= pl.lit(end_date).str.to_date())
return funding.sort("timestamp", "symbol")
def funding_rates_for_prices(prices: pl.DataFrame) -> pl.DataFrame:
"""Official funding settlements restricted to the exact keys ``prices`` covers.
The backtest charges funding against the position held at each settlement, so the
settlements a run is identified by have to be the ones its own price panel spans. A
holdout run slices a different window from the validation run it inherits its
configuration from, which is why this is derived from the frame rather than carried
across with the rest of the specification.
"""
def _day(value: object) -> str:
if isinstance(value, datetime):
return value.date().isoformat()
if isinstance(value, date):
return value.isoformat()
raise TypeError(f"expected a date-like timestamp, got {type(value).__name__}")
timestamp_dtype = prices.schema["timestamp"]
price_keys = prices.select("symbol", "timestamp").unique()
symbols = price_keys.get_column("symbol").unique().to_list()
funding = load_funding_rates(
symbols=symbols,
start_date=_day(price_keys.get_column("timestamp").min()),
end_date=_day(price_keys.get_column("timestamp").max()),
)
funding = funding.with_columns(
pl.col("symbol").cast(price_keys.schema["symbol"]),
pl.col("timestamp").cast(timestamp_dtype),
).join(price_keys, on=["symbol", "timestamp"], how="semi")
if funding.is_empty():
raise ValueError("crypto backtest resolved no official funding settlements")
return funding.sort("timestamp", "symbol")
def main() -> None:
case_dir = get_case_study_dir("crypto_perps_funding")
setup = yaml.safe_load((case_dir / "config" / "setup.yaml").read_text())
download_funding_rates(
list(setup["universe"]["symbols"]),
"2020-01-01",
"2025-12-31",
)
print(f"Completed at {datetime.now(UTC).isoformat()}", flush=True)
if __name__ == "__main__":
main()
```Shown in full with attribution under the source's licence. Licence: MIT
This summary was written by Stratmill's research agent from the original; it is not a copy of the source.