본문으로 건너뛰기
라이브러리 문서 전체

횡단면 패널 및 특성 관리 포트폴리오 준비

코드 Machine Learning for Trading

요약

이 문서는 날짜가 있는 엔터티 기록을 잠재 팩터 사례 연구용 배열로 변환하는 방법을 설명합니다. 한 방법은 적격성 데이터셋에서 학습한 커버리지 기준을 충족하는 엔터티만 남깁니다. 누락된 관측값은 NaN으로 유지해 일관된 엔터티 축을 만듭니다. 두 번째 방법은 불규칙한 횡단면을 만들고 각 날짜에 관측된 최대 크기까지 채웁니다. 슬롯 위치는 날짜별로만 유효하며 날짜가 바뀌어도 동일한 엔터티를 나타내지는 않습니다.

또한 각 날짜 안에서 특성 순위를 제한된 척도로 변환하고, 각 특성별로 수익률을 회귀해 대각 특성 관리 포트폴리오를 구성하며, 과거 기준 시점 조인으로 거시 특성을 정렬하는 방법을 다룹니다. 구현에는 실용적인 한계가 드러납니다. 적격 엔터티가 없는 날짜는 오류를 일으키고, 과거 거시 데이터가 부족하면 거부되며, 특성이나 수익률이 누락된 행은 특성 관리 포트폴리오 계산에서 빠집니다. 이는 데이터 준비 참고 자료이지 실증 트레이딩 연구가 아니며 성과 근거를 제시하지 않습니다.

핵심 아이디어

  • 고정 패널은 학습 데이터에서 정한 커버리지 기준으로 엔터티를 선택하고 안정적인 엔터티 축을 유지합니다.
  • 불규칙 패널은 날짜별로 관측된 엔터티만 나타냅니다. 채워 넣은 슬롯은 날짜가 바뀌어도 엔터티 식별을 유지하지 않습니다.
  • 횡단면 순위 정규화는 각 날짜의 유한한 특성값을 날짜별로 제한된 척도에 매핑합니다.
  • 대각 특성 관리 포트폴리오는 각 특성별 수익률 가중 익스포저를 따로 추정합니다.
  • 거시 특성은 이용 가능한 과거 데이터 범위 안에서 과거 기준 시점 조인으로 패널 날짜에 정렬할 수 있습니다.

태그

전문
# panel.py


```py
"""Data preparation utilities for latent factor case studies."""

from __future__ import annotations

from collections.abc import Sequence
from typing import Any

import numpy as np
import polars as pl
from scipy import stats

# One definition, in a module the coverage guard can import without torch - see that
# module's docstring for why it is not in this file.
from case_studies.utils.persistent_panel import (
    DEFAULT_MIN_COVERAGE,
    PERSISTENT_PANEL_MODELS,
    eligible_persistent_entities,
)


def prepare_ragged_panel_data(
    dataset: pl.DataFrame,
    feature_names: list[str],
    label_col: str,
    date_col: str,
    entity_col: str,
    max_entities: int = 0,
    eval_label_col: str | None = None,
    macro_panel: pl.DataFrame | None = None,
) -> dict[str, Any]:
    """Build a dated cross-sectional panel with per-date observed assets only.

    The returned arrays are padded to the maximum cross-section size within the
    input window. The slot axis is date-local and does not imply stable entity
    identity across time.
    """
    df = _sort_panel_frame(dataset, date_col=date_col, entity_col=entity_col)
    if max_entities > 0:
        df = _limit_entities(df, entity_col=entity_col, max_entities=max_entities)

    groups = df.partition_by(date_col, maintain_order=True)
    if not groups:
        raise ValueError("Dataset produced no dated cross-sections")

    dates = [group[date_col][0] for group in groups]
    counts = np.asarray([group.height for group in groups], dtype=np.int32)
    n_dates = len(groups)
    n_slots = int(counts.max())
    n_features = len(feature_names)

    chars = np.full((n_dates, n_slots, n_features), np.nan, dtype=np.float32)
    returns = np.full((n_dates, n_slots), np.nan, dtype=np.float32)
    eval_returns = np.full((n_dates, n_slots), np.nan, dtype=np.float32) if eval_label_col else None
    entities = np.full((n_dates, n_slots), None, dtype=object)

    for date_idx, group in enumerate(groups):
        n_obs = group.height
        chars[date_idx, :n_obs] = group.select(feature_names).to_numpy().astype(np.float32)
        returns[date_idx, :n_obs] = (
            group.select(label_col).to_numpy().reshape(-1).astype(np.float32)
        )
        if eval_returns is not None:
            eval_returns[date_idx, :n_obs] = (
                group.select(eval_label_col).to_numpy().reshape(-1).astype(np.float32)
            )
        entities[date_idx, :n_obs] = np.asarray(group[entity_col].to_list(), dtype=object)

    macro = None
    macro_features: list[str] | None = None
    if macro_panel is not None:
        macro, macro_features = align_macro_to_dates(macro_panel, dates, date_col)

    return {
        "chars": chars,
        "returns": returns,
        "eval_returns": eval_returns,
        "dates": np.asarray(dates, dtype="datetime64[ns]"),
        "entities": entities,
        "counts": counts,
        "entity_col": entity_col,
        "macro": macro,
        "macro_features": macro_features,
    }


def prepare_panel_data(
    dataset: pl.DataFrame,
    feature_names: list[str],
    label_col: str,
    date_col: str,
    entity_col: str,
    *,
    eligibility_dataset: pl.DataFrame,
    max_entities: int = 0,
    min_coverage: float = DEFAULT_MIN_COVERAGE,
    eval_label_col: str | None = None,
    macro_panel: pl.DataFrame | None = None,
) -> dict[str, Any]:
    """Build a persistent-entity panel with eligibility learned from training data."""
    df = _sort_panel_frame(dataset, date_col=date_col, entity_col=entity_col)
    eligibility_df = _sort_panel_frame(
        eligibility_dataset,
        date_col=date_col,
        entity_col=entity_col,
    )

    eligible = eligible_persistent_entities(
        eligibility_df,
        entity_col=entity_col,
        date_col=date_col,
        min_coverage=min_coverage,
    )
    if max_entities > 0:
        eligible = eligible.head(max_entities)

    entities = sorted(eligible[entity_col].to_list())
    if not entities:
        raise ValueError("No entities met the persistent-panel coverage requirement")

    df = df.filter(pl.col(entity_col).is_in(entities)).sort(date_col, entity_col)
    dates = sorted(df[date_col].unique().to_list())

    n_dates = len(dates)
    n_entities = len(entities)
    n_features = len(feature_names)

    chars = np.full((n_dates, n_entities, n_features), np.nan, dtype=np.float32)
    returns = np.full((n_dates, n_entities), np.nan, dtype=np.float32)
    eval_returns = (
        np.full((n_dates, n_entities), np.nan, dtype=np.float32) if eval_label_col else None
    )
    date_values = np.asarray(dates, dtype="datetime64[ns]")
    entity_values = np.asarray(entities, dtype=object)
    date_idx = np.searchsorted(date_values, df[date_col].to_numpy())
    entity_idx = np.searchsorted(entity_values, df[entity_col].to_numpy())

    chars[date_idx, entity_idx] = (
        df.select(feature_names)
        .to_numpy()
        .astype(
            np.float32,
            copy=False,
        )
    )
    returns[date_idx, entity_idx] = df[label_col].to_numpy().astype(np.float32, copy=False)
    if eval_returns is not None:
        eval_returns[date_idx, entity_idx] = (
            df[eval_label_col]
            .to_numpy()
            .astype(
                np.float32,
                copy=False,
            )
        )

    macro = None
    macro_features: list[str] | None = None
    if macro_panel is not None:
        macro, macro_features = align_macro_to_dates(macro_panel, dates, date_col)

    return {
        "chars": chars,
        "returns": returns,
        "eval_returns": eval_returns,
        "dates": date_values,
        "entities": entity_values,
        "entity_col": entity_col,
        "macro": macro,
        "macro_features": macro_features,
    }


def rank_normalize_cross_section(chars: np.ndarray) -> np.ndarray:
    """Rank-normalize each date's characteristics to the [-0.5, 0.5] interval."""
    arr = np.asarray(chars, dtype=np.float32)
    original_ndim = arr.ndim
    if original_ndim == 2:
        arr = arr[None, :, :]
    if arr.ndim != 3:
        raise ValueError(f"chars must be 2D or 3D; got shape {arr.shape}")

    ranked = np.zeros_like(arr, dtype=np.float32)
    _, _, n_features = arr.shape

    for date_idx in range(arr.shape[0]):
        for feature_idx in range(n_features):
            values = arr[date_idx, :, feature_idx]
            valid = np.isfinite(values)
            n_valid = int(valid.sum())
            if n_valid == 0:
                continue
            if n_valid == 1:
                ranked[date_idx, valid, feature_idx] = 0.0
                continue
            ranks = stats.rankdata(values[valid], method="average")
            ranked[date_idx, valid, feature_idx] = ((ranks - 1.0) / (n_valid - 1.0) - 0.5).astype(
                np.float32
            )

    return ranked[0] if original_ndim == 2 else ranked


def compute_managed_portfolios(
    chars: np.ndarray,
    returns: np.ndarray,
) -> np.ndarray:
    """Compute diagonal characteristic-managed portfolios for each date."""
    if chars.ndim != 3:
        raise ValueError(f"chars must be 3D (T, N, L); got shape {chars.shape}")
    if returns.ndim != 2:
        raise ValueError(f"returns must be 2D (T, N); got shape {returns.shape}")
    if chars.shape[:2] != returns.shape:
        raise ValueError(
            f"chars and returns disagree on (T, N): {chars.shape[:2]} vs {returns.shape}"
        )

    n_dates, n_slots, n_features = chars.shape
    ones = np.ones((n_dates, n_slots, 1), dtype=np.float32)
    chars_aug = np.concatenate([chars.astype(np.float32, copy=False), ones], axis=2)
    portfolios = np.zeros((n_dates, n_slots, n_features + 1), dtype=np.float32)
    eps = 1e-8

    for date_idx in range(n_dates):
        z_t = chars_aug[date_idx]
        r_t = returns[date_idx]
        valid = np.isfinite(r_t) & np.isfinite(z_t).all(axis=1)
        if not valid.any():
            continue
        z_valid = z_t[valid].astype(np.float64)
        r_valid = r_t[valid].astype(np.float64)
        numerator = (z_valid * r_valid[:, None]).sum(axis=0)
        denominator = (z_valid**2).sum(axis=0)
        x_t = numerator / np.maximum(denominator, eps)
        portfolios[date_idx] = np.broadcast_to(
            x_t.astype(np.float32)[None, :],
            (n_slots, n_features + 1),
        )

    return portfolios


def align_macro_to_dates(
    macro_panel: pl.DataFrame,
    dates: Sequence[object],
    date_col: str = "timestamp",
) -> tuple[np.ndarray, list[str]]:
    """Align macro features to case-study dates with backward as-of joins."""
    macro = macro_panel.clone()
    if hasattr(macro[date_col].dtype, "time_zone") and macro[date_col].dtype.time_zone is not None:
        macro = macro.with_columns(pl.col(date_col).dt.replace_time_zone(None))

    feature_cols = [column for column in macro.columns if column != date_col]
    if not feature_cols:
        return np.zeros((len(dates), 0), dtype=np.float32), []

    date_frame = (
        pl.DataFrame(pl.Series(date_col, dates))
        .with_columns(pl.col(date_col).cast(macro.schema[date_col]))
        .sort(date_col)
    )
    aligned = date_frame.join_asof(
        macro.sort(date_col), on=date_col, strategy="backward"
    ).fill_null(strategy="forward")
    null_counts = aligned.select(feature_cols).null_count().row(0)
    if any(null_counts):
        missing = [name for name, count in zip(feature_cols, null_counts, strict=True) if count]
        raise ValueError(
            f"Macro context is unavailable on or before the first requested date for: {missing}"
        )
    return aligned.select(feature_cols).to_numpy().astype(np.float32), feature_cols


def _sort_panel_frame(
    dataset: pl.DataFrame,
    *,
    date_col: str,
    entity_col: str,
) -> pl.DataFrame:
    df = dataset.sort(date_col, entity_col)
    if hasattr(df[date_col].dtype, "time_zone") and df[date_col].dtype.time_zone is not None:
        df = df.with_columns(pl.col(date_col).dt.replace_time_zone(None))
    return df


def _limit_entities(
    dataset: pl.DataFrame,
    *,
    entity_col: str,
    max_entities: int,
) -> pl.DataFrame:
    top_entities = (
        dataset.group_by(entity_col)
        .len()
        .sort(["len", entity_col], descending=[True, False])
        .head(max_entities)[entity_col]
        .to_list()
    )
    return dataset.filter(pl.col(entity_col).is_in(top_entities))


__all__ = [
    "DEFAULT_MIN_COVERAGE",
    "PERSISTENT_PANEL_MODELS",
    "align_macro_to_dates",
    "compute_managed_portfolios",
    "eligible_persistent_entities",
    "prepare_panel_data",
    "prepare_ragged_panel_data",
    "rank_normalize_cross_section",
]

```

출처의 라이선스에 따라 출처를 표시하고 전문을 공개합니다. 라이선스: MIT

이 요약은 원문을 바탕으로 Stratmill의 리서치 에이전트가 작성했으며, 원문을 복사한 것이 아닙니다.