Skip to content

Data and features

Configuration

config

Typed configuration loaded from TOML.

TOML is read with the stdlib tomllib, so config costs no dependency. (It landed in 3.11; the project's tested floor is 3.13 — see requires-python.) Every experiment is fully described by one file in configs/ — that file is the unit of reproducibility. If a number influenced a result, it belongs here and not in a function default.

SizingConfig dataclass

SizingConfig(long_entry: float = 0.56, long_exit: float = 0.5, short_entry: float = 0.44, short_exit: float = 0.5, allow_short: bool = True, min_hold: int = 6, max_leverage: float = 1.0, vol_target: float = 0.0)

Signal -> position. This is where a real edge is kept or destroyed.

load_config

load_config(path: str | Path) -> RunConfig

Load and validate one experiment's config.

A missing or malformed file is a ConfigError, not a traceback. nullres run -c typo.toml used to end in a raw FileNotFoundError from deep inside pathlib, which tells the reader where Python gave up rather than what they got wrong.

Source code in nullres/config.py
def load_config(path: str | Path) -> RunConfig:
    """Load and validate one experiment's config.

    A missing or malformed file is a ConfigError, not a traceback. `nullres run
    -c typo.toml` used to end in a raw FileNotFoundError from deep inside
    pathlib, which tells the reader where Python gave up rather than what they
    got wrong.
    """
    path = Path(path)
    try:
        with path.open("rb") as fh:
            raw = tomllib.load(fh)
    except FileNotFoundError:
        raise ConfigError(f"no config file at {path}") from None
    except IsADirectoryError:
        raise ConfigError(f"{path} is a directory, not a config file") from None
    except tomllib.TOMLDecodeError as exc:
        raise ConfigError(f"{path} is not valid TOML: {exc}") from exc
    cfg = _build(RunConfig, raw)
    if cfg.name == "unnamed":
        cfg.name = path.stem
    return cfg

Loading

data

Market data loading.

Everything here returns the same contract: a DataFrame indexed by UTC timestamp with float columns [open, high, low, close, volume, trades], strictly increasing index, no duplicates. load_bars is the only entry point callers should need.

fetch_month

fetch_month(symbol: str, interval: str, month: str, cache_dir: str = 'data', retries: int = 3, market: str = 'spot') -> DataFrame | None

Return one month of klines, from cache when present. None if unavailable.

Source code in nullres/data/binance.py
def fetch_month(symbol: str, interval: str, month: str, cache_dir: str = "data",
                retries: int = 3, market: str = "spot") -> pd.DataFrame | None:
    """Return one month of klines, from cache when present. None if unavailable."""
    cache = Path(cache_dir)
    cache.mkdir(parents=True, exist_ok=True)
    # Futures files are namespaced so spot and perp bars never collide in cache.
    tag = "" if market == "spot" else f"{market}-"
    cached = cache / f"{tag}{symbol}-{interval}-{month}.parquet"
    if cached.exists():
        hit = read_parquet_or_discard(cached)
        if hit is not None:
            return hit

    if market not in MARKET_URLS:
        raise ConfigError(f"unknown market {market!r}; choose from {sorted(MARKET_URLS)}")
    base = MARKET_URLS[market]
    url = f"{base}/{symbol}/{interval}/{symbol}-{interval}-{month}.zip"
    for attempt in range(retries):
        try:
            resp = requests.get(url, timeout=60)
        except requests.RequestException as exc:
            if attempt == retries - 1:
                log.warning("  fail %s: %s", month, exc)
                return None
            time.sleep(2 ** attempt)
            continue

        if resp.status_code == 404:
            # Month predates the listing or is not yet published. Not an error.
            log.info("  miss %s (404)", month)
            return None
        if resp.status_code != 200:
            if attempt == retries - 1:
                log.warning("  fail %s (HTTP %s)", month, resp.status_code)
                return None
            time.sleep(2 ** attempt)
            continue

        with zipfile.ZipFile(io.BytesIO(resp.content)) as zf:
            raw = zf.read(zf.namelist()[0]).decode()

        # Archives from ~2025 onward ship a header row; older ones do not.
        header = 0 if raw.lstrip().lower().startswith("open_time") else None
        df = pd.read_csv(io.StringIO(raw), header=header, names=KLINE_COLS)
        write_parquet_atomic(df, cached)
        log.info("  ok   %s  (%s bars)", month, f"{len(df):,}")
        return df
    return None

load_binance

load_binance(symbol: str, interval: str, start: str, end: str, cache_dir: str = 'data', verbose: bool = True, market: str = 'spot', required: bool = True) -> DataFrame | None

Load a contiguous range of months and validate the result.

Parameters:

Name Type Description Default
required bool

when False, return None instead of raising if the symbol has no data at all. Delisted assets must be loadable-or-absent rather than fatal — excluding them because they stopped existing is the definition of survivorship bias.

True
Source code in nullres/data/binance.py
def load_binance(symbol: str, interval: str, start: str, end: str,
                 cache_dir: str = "data", verbose: bool = True,
                 market: str = "spot", required: bool = True) -> pd.DataFrame | None:
    """Load a contiguous range of months and validate the result.

    Args:
        required: when False, return None instead of raising if the symbol has
            no data at all. Delisted assets must be loadable-or-absent rather
            than fatal — excluding them because they stopped existing is the
            definition of survivorship bias.
    """
    if verbose:
        log.info("Loading %s %s %s..%s (%s)", symbol, interval, start, end, market)
    months = [d.strftime("%Y-%m") for d in pd.date_range(start, end, freq="MS")]
    parts = [p for p in (fetch_month(symbol, interval, m, cache_dir, market=market)
                         for m in months)
             if p is not None and len(p)]
    if not parts:
        if not required:
            return None
        raise DataUnavailableError(
            f"No data for {symbol} {interval} {start}..{end}. "
            f"Check the symbol spelling and that the range is not in the future."
        )

    df = pd.concat(parts, ignore_index=True)

    # Binance switched open_time from milliseconds to microseconds during 2025.
    # Detect per row rather than per file: a single concat can straddle both.
    ot = df["open_time"].astype("int64")
    unit_us = ot > 1e15
    ts = pd.Series(pd.NaT, index=df.index, dtype="datetime64[ns]")
    if unit_us.any():
        ts[unit_us] = pd.to_datetime(ot[unit_us], unit="us")
    if (~unit_us).any():
        ts[~unit_us] = pd.to_datetime(ot[~unit_us], unit="ms")
    df["ts"] = ts

    df = df[["ts"] + KEEP].astype({c: "float64" for c in KEEP})
    df = df.drop_duplicates("ts").sort_values("ts").set_index("ts")
    df.index.name = "ts"

    _validate(df, interval, verbose)
    return df

load_funding

load_funding(symbol: str, start: str, end: str, cache_dir: str = 'data', verbose: bool = True) -> DataFrame

8-hourly funding rates, indexed by settlement time (UTC).

The index is the moment the rate was SETTLED, which is the moment it became known. Callers must join with that in mind — see features/derivatives.py.

Source code in nullres/data/futures.py
def load_funding(symbol: str, start: str, end: str, cache_dir: str = "data",
                 verbose: bool = True) -> pd.DataFrame:
    """8-hourly funding rates, indexed by settlement time (UTC).

    The index is the moment the rate was SETTLED, which is the moment it became
    known. Callers must join with that in mind — see `features/derivatives.py`.
    """
    cache = Path(cache_dir)
    cache.mkdir(parents=True, exist_ok=True)
    months = [d.strftime("%Y-%m") for d in pd.date_range(start, end, freq="MS")]
    parts = []

    for month in months:
        cached = cache / f"{symbol}-funding-{month}.parquet"
        if cached.exists():
            hit = read_parquet_or_discard(cached)
            if hit is not None:
                parts.append(hit)
                continue
        resp = _get(f"{FUNDING_URL}/{symbol}/{symbol}-fundingRate-{month}.zip")
        if resp is None:
            if verbose:
                log.info("  funding miss %s", month)
            continue
        df = _read_zip_csv(resp.content)
        write_parquet_atomic(df, cached)
        parts.append(df)

    if not parts:
        raise DataUnavailableError(f"no funding data for {symbol} {start}..{end}")

    df = pd.concat(parts, ignore_index=True)
    # Force nanosecond resolution. pandas 3.0 infers the unit from the source,
    # so string timestamps land as datetime64[ms] while kline data is [ns], and
    # merge_asof refuses to join across resolutions.
    df["ts"] = pd.to_datetime(df["calc_time"], unit="ms").astype("datetime64[ns]")
    out = (df[["ts", "last_funding_rate", "funding_interval_hours"]]
           .rename(columns={"last_funding_rate": "funding_rate",
                            "funding_interval_hours": "funding_hours"})
           .drop_duplicates("ts").sort_values("ts").set_index("ts"))
    out = out.astype({"funding_rate": "float64", "funding_hours": "float64"})
    if verbose:
        log.info("  funding: %s settlements %s..%s", f"{len(out):,}",
                 f"{out.index[0]:%Y-%m-%d}", f"{out.index[-1]:%Y-%m-%d}")
    return out

load_metrics

load_metrics(symbol: str, start: str, end: str, cache_dir: str = 'data', workers: int = 8, verbose: bool = True) -> DataFrame

Open interest and long/short ratios at 5-minute granularity.

Binance publishes these as one archive PER DAY (the monthly path 404s), so a six-year range is ~2,000 requests. They are fetched concurrently and cached one parquet per month — caching per day would leave 2,000 files in data/ for no benefit.

Source code in nullres/data/futures.py
def load_metrics(symbol: str, start: str, end: str, cache_dir: str = "data",
                 workers: int = 8, verbose: bool = True) -> pd.DataFrame:
    """Open interest and long/short ratios at 5-minute granularity.

    Binance publishes these as one archive PER DAY (the monthly path 404s), so
    a six-year range is ~2,000 requests. They are fetched concurrently and
    cached one parquet per month — caching per day would leave 2,000 files in
    `data/` for no benefit.
    """
    cache = Path(cache_dir)
    cache.mkdir(parents=True, exist_ok=True)
    months = pd.date_range(start, end, freq="MS")
    parts = []

    for month_start in months:
        tag = month_start.strftime("%Y-%m")
        cached = cache / f"{symbol}-metrics-{tag}.parquet"
        if cached.exists():
            hit = read_parquet_or_discard(cached)
            if hit is not None:
                parts.append(hit)
                continue

        days = pd.date_range(month_start, month_start + pd.offsets.MonthEnd(0), freq="D")
        with requests.Session() as session:
            with ThreadPoolExecutor(max_workers=workers) as pool:
                frames = list(pool.map(
                    lambda d: _fetch_metrics_day(symbol, d.strftime("%Y-%m-%d"), session),
                    days,
                ))
        frames = [f for f in frames if f is not None and len(f)]
        if not frames:
            if verbose:
                log.info("  metrics miss %s", tag)
            continue

        month_df = pd.concat(frames, ignore_index=True)
        write_parquet_atomic(month_df, cached)
        parts.append(month_df)
        if verbose:
            log.info("  metrics ok   %s  (%s rows from %d days)", tag,
                     f"{len(month_df):,}", len(frames))

    if not parts:
        raise DataUnavailableError(
            f"no metrics data for {symbol} {start}..{end}. "
            f"BTCUSDT metrics begin 2020-09; other symbols start later."
        )

    df = pd.concat(parts, ignore_index=True)
    df["ts"] = pd.to_datetime(df["create_time"]).astype("datetime64[ns]")
    keep = ["ts"] + [c for c in METRIC_COLS if c in df.columns]
    out = (df[keep].rename(columns=METRIC_COLS)
           .drop_duplicates("ts").sort_values("ts").set_index("ts"))
    out = out.astype("float64")
    if verbose:
        log.info("  metrics: %s rows %s..%s", f"{len(out):,}",
                 f"{out.index[0]:%Y-%m-%d}", f"{out.index[-1]:%Y-%m-%d}")
    return out

synthetic_bars

synthetic_bars(n: int = 40000, seed: int = 0, interval: str = '1h', sigma: float = 0.004, edge: float = 0.0, start: str = '2020-01-01') -> DataFrame

Geometric random walk in OHLCV form.

Parameters:

Name Type Description Default
edge float

AR(1) coefficient on log returns. 0.0 means a pure martingale — unpredictable by construction. ~0.05 is a faint but genuinely learnable edge; real liquid markets sit near 0.0 to 0.02.

0.0
Source code in nullres/data/synthetic.py
def synthetic_bars(n: int = 40_000, seed: int = 0, interval: str = "1h",
                   sigma: float = 0.004, edge: float = 0.0,
                   start: str = "2020-01-01") -> pd.DataFrame:
    """Geometric random walk in OHLCV form.

    Args:
        edge: AR(1) coefficient on log returns. 0.0 means a pure martingale —
            unpredictable by construction. ~0.05 is a faint but genuinely
            learnable edge; real liquid markets sit near 0.0 to 0.02.
    """
    rng = np.random.default_rng(seed)
    shocks = rng.normal(0.0, sigma, n)

    if edge:
        logret = np.empty(n)
        logret[0] = shocks[0]
        for i in range(1, n):
            logret[i] = edge * logret[i - 1] + shocks[i]
    else:
        logret = shocks

    close = 30_000.0 * np.exp(np.cumsum(logret))

    # Build O/H/L around the close so OHLC invariants always hold.
    open_ = np.empty(n)
    open_[0] = close[0] * (1 + rng.normal(0, sigma / 8))
    open_[1:] = close[:-1] * (1 + rng.normal(0, sigma / 8, n - 1))
    hi_wick = np.abs(rng.normal(0, sigma / 2, n))
    lo_wick = np.abs(rng.normal(0, sigma / 2, n))
    high = np.maximum(open_, close) * (1 + hi_wick)
    low = np.minimum(open_, close) * (1 - lo_wick)

    return pd.DataFrame(
        {
            "open": open_,
            "high": high,
            "low": low,
            "close": close,
            "volume": np.abs(rng.normal(500, 150, n)),
            "trades": np.abs(rng.normal(3_000, 800, n)),
        },
        index=pd.date_range(start, periods=n, freq=_interval_freq(interval), name="ts"),
    )

synthetic_funding

synthetic_funding(bars: DataFrame, seed: int = 0) -> DataFrame

Funding settlements every 8h, uncorrelated with future returns.

Exists so the null control exercises the SAME code path as a real run. Without it the random-walk check runs on 32 features while the live config runs on 47, and a broken funding join would never reach the one test whose whole job is to catch fabricated edge.

The values are noise by construction, so any strategy that profits from them on this data has found a bug in the join, not a signal.

Source code in nullres/data/synthetic.py
def synthetic_funding(bars: pd.DataFrame, seed: int = 0) -> pd.DataFrame:
    """Funding settlements every 8h, uncorrelated with future returns.

    Exists so the null control exercises the SAME code path as a real run.
    Without it the random-walk check runs on 32 features while the live config
    runs on 47, and a broken funding join would never reach the one test whose
    whole job is to catch fabricated edge.

    The values are noise by construction, so any strategy that profits from
    them on this data has found a bug in the join, not a signal.
    """
    rng = np.random.default_rng(seed + 101)
    idx = pd.date_range(bars.index[0], bars.index[-1] + pd.Timedelta("8h"), freq="8h")
    return pd.DataFrame(
        {
            "funding_rate": rng.normal(0.0001, 0.0003, len(idx)),
            "funding_hours": np.full(len(idx), 8.0),
        },
        index=pd.DatetimeIndex(idx, name="ts"),
    )

synthetic_metrics

synthetic_metrics(bars: DataFrame, seed: int = 0) -> DataFrame

Open interest and positioning ratios, also pure noise.

Open interest is generated as a random walk with drift so it is non-stationary like the real thing — that way the stationarity discipline in features/derivatives.py is genuinely exercised.

Source code in nullres/data/synthetic.py
def synthetic_metrics(bars: pd.DataFrame, seed: int = 0) -> pd.DataFrame:
    """Open interest and positioning ratios, also pure noise.

    Open interest is generated as a random walk with drift so it is
    non-stationary like the real thing — that way the stationarity discipline
    in `features/derivatives.py` is genuinely exercised.
    """
    rng = np.random.default_rng(seed + 202)
    idx = pd.date_range(bars.index[0], bars.index[-1] + pd.Timedelta("1h"), freq="1h")
    n = len(idx)
    oi = 1e5 * np.exp(np.cumsum(rng.normal(0.0002, 0.01, n)))
    return pd.DataFrame(
        {
            "open_interest": oi,
            "oi_value": oi * 30_000,
            "top_trader_accounts_ls": np.abs(rng.normal(2.5, 0.4, n)),
            "top_trader_positions_ls": np.abs(rng.normal(1.2, 0.2, n)),
            "all_accounts_ls": np.abs(rng.normal(2.0, 0.3, n)),
            "taker_buy_sell_ratio": np.abs(rng.normal(1.0, 0.2, n)),
        },
        index=pd.DatetimeIndex(idx, name="ts"),
    )

load_bars

load_bars(cfg)

Dispatch on cfg.source and return bars matching the OHLCV contract.

Source code in nullres/data/__init__.py
def load_bars(cfg):
    """Dispatch on cfg.source and return bars matching the OHLCV contract."""
    if cfg.source == "synthetic":
        return synthetic_bars(interval=cfg.interval)
    if cfg.source == "binance":
        return load_binance(cfg.symbol, cfg.interval, cfg.start, cfg.end, cfg.cache_dir)
    raise ConfigError(f"unknown data source {cfg.source!r}")

load_auxiliary

load_auxiliary(cfg, verbose: bool = True, bars=None)

Return (funding, metrics), either of which may be None.

For synthetic data the auxiliary frames are generated as pure noise rather than skipped. That keeps the null control running the SAME feature pipeline as a live config — otherwise the random-walk check would exercise 32 features while the real run uses 46, and a broken funding join would never reach the one test designed to catch fabricated edge.

Source code in nullres/data/__init__.py
def load_auxiliary(cfg, verbose: bool = True, bars=None):
    """Return (funding, metrics), either of which may be None.

    For synthetic data the auxiliary frames are generated as pure noise rather
    than skipped. That keeps the null control running the SAME feature pipeline
    as a live config — otherwise the random-walk check would exercise 32
    features while the real run uses 46, and a broken funding join would never
    reach the one test designed to catch fabricated edge.
    """
    if cfg.source == "synthetic":
        if bars is None or not (cfg.funding or cfg.metrics):
            return None, None
        return (
            synthetic_funding(bars) if cfg.funding else None,
            synthetic_metrics(bars) if cfg.metrics else None,
        )
    if cfg.source != "binance":
        return None, None
    funding = (load_funding(cfg.symbol, cfg.start, cfg.end, cfg.cache_dir, verbose)
               if cfg.funding else None)
    metrics = (load_metrics(cfg.symbol, cfg.start, cfg.end, cfg.cache_dir,
                            verbose=verbose) if cfg.metrics else None)
    return funding, metrics

binance

Binance public monthly kline archives (data.binance.vision).

No API key, no rate limit, no exchange account. Each month is cached to parquet so a re-run is offline and instant.

fetch_month

fetch_month(symbol: str, interval: str, month: str, cache_dir: str = 'data', retries: int = 3, market: str = 'spot') -> DataFrame | None

Return one month of klines, from cache when present. None if unavailable.

Source code in nullres/data/binance.py
def fetch_month(symbol: str, interval: str, month: str, cache_dir: str = "data",
                retries: int = 3, market: str = "spot") -> pd.DataFrame | None:
    """Return one month of klines, from cache when present. None if unavailable."""
    cache = Path(cache_dir)
    cache.mkdir(parents=True, exist_ok=True)
    # Futures files are namespaced so spot and perp bars never collide in cache.
    tag = "" if market == "spot" else f"{market}-"
    cached = cache / f"{tag}{symbol}-{interval}-{month}.parquet"
    if cached.exists():
        hit = read_parquet_or_discard(cached)
        if hit is not None:
            return hit

    if market not in MARKET_URLS:
        raise ConfigError(f"unknown market {market!r}; choose from {sorted(MARKET_URLS)}")
    base = MARKET_URLS[market]
    url = f"{base}/{symbol}/{interval}/{symbol}-{interval}-{month}.zip"
    for attempt in range(retries):
        try:
            resp = requests.get(url, timeout=60)
        except requests.RequestException as exc:
            if attempt == retries - 1:
                log.warning("  fail %s: %s", month, exc)
                return None
            time.sleep(2 ** attempt)
            continue

        if resp.status_code == 404:
            # Month predates the listing or is not yet published. Not an error.
            log.info("  miss %s (404)", month)
            return None
        if resp.status_code != 200:
            if attempt == retries - 1:
                log.warning("  fail %s (HTTP %s)", month, resp.status_code)
                return None
            time.sleep(2 ** attempt)
            continue

        with zipfile.ZipFile(io.BytesIO(resp.content)) as zf:
            raw = zf.read(zf.namelist()[0]).decode()

        # Archives from ~2025 onward ship a header row; older ones do not.
        header = 0 if raw.lstrip().lower().startswith("open_time") else None
        df = pd.read_csv(io.StringIO(raw), header=header, names=KLINE_COLS)
        write_parquet_atomic(df, cached)
        log.info("  ok   %s  (%s bars)", month, f"{len(df):,}")
        return df
    return None

load_binance

load_binance(symbol: str, interval: str, start: str, end: str, cache_dir: str = 'data', verbose: bool = True, market: str = 'spot', required: bool = True) -> DataFrame | None

Load a contiguous range of months and validate the result.

Parameters:

Name Type Description Default
required bool

when False, return None instead of raising if the symbol has no data at all. Delisted assets must be loadable-or-absent rather than fatal — excluding them because they stopped existing is the definition of survivorship bias.

True
Source code in nullres/data/binance.py
def load_binance(symbol: str, interval: str, start: str, end: str,
                 cache_dir: str = "data", verbose: bool = True,
                 market: str = "spot", required: bool = True) -> pd.DataFrame | None:
    """Load a contiguous range of months and validate the result.

    Args:
        required: when False, return None instead of raising if the symbol has
            no data at all. Delisted assets must be loadable-or-absent rather
            than fatal — excluding them because they stopped existing is the
            definition of survivorship bias.
    """
    if verbose:
        log.info("Loading %s %s %s..%s (%s)", symbol, interval, start, end, market)
    months = [d.strftime("%Y-%m") for d in pd.date_range(start, end, freq="MS")]
    parts = [p for p in (fetch_month(symbol, interval, m, cache_dir, market=market)
                         for m in months)
             if p is not None and len(p)]
    if not parts:
        if not required:
            return None
        raise DataUnavailableError(
            f"No data for {symbol} {interval} {start}..{end}. "
            f"Check the symbol spelling and that the range is not in the future."
        )

    df = pd.concat(parts, ignore_index=True)

    # Binance switched open_time from milliseconds to microseconds during 2025.
    # Detect per row rather than per file: a single concat can straddle both.
    ot = df["open_time"].astype("int64")
    unit_us = ot > 1e15
    ts = pd.Series(pd.NaT, index=df.index, dtype="datetime64[ns]")
    if unit_us.any():
        ts[unit_us] = pd.to_datetime(ot[unit_us], unit="us")
    if (~unit_us).any():
        ts[~unit_us] = pd.to_datetime(ot[~unit_us], unit="ms")
    df["ts"] = ts

    df = df[["ts"] + KEEP].astype({c: "float64" for c in KEEP})
    df = df.drop_duplicates("ts").sort_values("ts").set_index("ts")
    df.index.name = "ts"

    _validate(df, interval, verbose)
    return df

futures

Binance USD-M futures data: funding rates and open-interest metrics.

This is the first information in the repo that is NOT a transform of OHLCV. Everything in features/technical.py is thirty-two views of four numbers; nullres features showed almost none of them carry out of sample. Funding and open interest measure something the price series cannot express: how much leverage is deployed, and on which side.

FUNDING RATE   Perpetual futures have no expiry, so an 8-hourly payment
               tethers them to spot. Positive funding means longs pay
               shorts — the crowd is long and paying to stay there. It is a
               direct read on positioning, and it is a PRICE, so it is set
               by people with money at risk.

OPEN INTEREST  Total outstanding contracts. Rising OI into a rally means
               new money; falling OI means an unwind. Same price move,
               opposite implication.

LONG/SHORT     Binance publishes account- and position-weighted long/short
RATIOS         ratios, including a top-trader subset.

Availability (probed, not assumed): funding monthly archives, BTCUSDT from 2020-01 metrics DAILY archives only, BTCUSDT from 2020-09; monthly 404s

Caveat worth stating plainly: these describe the PERPETUAL market, while the bars elsewhere in this repo are spot. That is a legitimate pairing — futures positioning predicting spot price is the whole idea — but they are different venues, and the perp can dislocate from spot precisely when it matters most.

load_funding

load_funding(symbol: str, start: str, end: str, cache_dir: str = 'data', verbose: bool = True) -> DataFrame

8-hourly funding rates, indexed by settlement time (UTC).

The index is the moment the rate was SETTLED, which is the moment it became known. Callers must join with that in mind — see features/derivatives.py.

Source code in nullres/data/futures.py
def load_funding(symbol: str, start: str, end: str, cache_dir: str = "data",
                 verbose: bool = True) -> pd.DataFrame:
    """8-hourly funding rates, indexed by settlement time (UTC).

    The index is the moment the rate was SETTLED, which is the moment it became
    known. Callers must join with that in mind — see `features/derivatives.py`.
    """
    cache = Path(cache_dir)
    cache.mkdir(parents=True, exist_ok=True)
    months = [d.strftime("%Y-%m") for d in pd.date_range(start, end, freq="MS")]
    parts = []

    for month in months:
        cached = cache / f"{symbol}-funding-{month}.parquet"
        if cached.exists():
            hit = read_parquet_or_discard(cached)
            if hit is not None:
                parts.append(hit)
                continue
        resp = _get(f"{FUNDING_URL}/{symbol}/{symbol}-fundingRate-{month}.zip")
        if resp is None:
            if verbose:
                log.info("  funding miss %s", month)
            continue
        df = _read_zip_csv(resp.content)
        write_parquet_atomic(df, cached)
        parts.append(df)

    if not parts:
        raise DataUnavailableError(f"no funding data for {symbol} {start}..{end}")

    df = pd.concat(parts, ignore_index=True)
    # Force nanosecond resolution. pandas 3.0 infers the unit from the source,
    # so string timestamps land as datetime64[ms] while kline data is [ns], and
    # merge_asof refuses to join across resolutions.
    df["ts"] = pd.to_datetime(df["calc_time"], unit="ms").astype("datetime64[ns]")
    out = (df[["ts", "last_funding_rate", "funding_interval_hours"]]
           .rename(columns={"last_funding_rate": "funding_rate",
                            "funding_interval_hours": "funding_hours"})
           .drop_duplicates("ts").sort_values("ts").set_index("ts"))
    out = out.astype({"funding_rate": "float64", "funding_hours": "float64"})
    if verbose:
        log.info("  funding: %s settlements %s..%s", f"{len(out):,}",
                 f"{out.index[0]:%Y-%m-%d}", f"{out.index[-1]:%Y-%m-%d}")
    return out

load_metrics

load_metrics(symbol: str, start: str, end: str, cache_dir: str = 'data', workers: int = 8, verbose: bool = True) -> DataFrame

Open interest and long/short ratios at 5-minute granularity.

Binance publishes these as one archive PER DAY (the monthly path 404s), so a six-year range is ~2,000 requests. They are fetched concurrently and cached one parquet per month — caching per day would leave 2,000 files in data/ for no benefit.

Source code in nullres/data/futures.py
def load_metrics(symbol: str, start: str, end: str, cache_dir: str = "data",
                 workers: int = 8, verbose: bool = True) -> pd.DataFrame:
    """Open interest and long/short ratios at 5-minute granularity.

    Binance publishes these as one archive PER DAY (the monthly path 404s), so
    a six-year range is ~2,000 requests. They are fetched concurrently and
    cached one parquet per month — caching per day would leave 2,000 files in
    `data/` for no benefit.
    """
    cache = Path(cache_dir)
    cache.mkdir(parents=True, exist_ok=True)
    months = pd.date_range(start, end, freq="MS")
    parts = []

    for month_start in months:
        tag = month_start.strftime("%Y-%m")
        cached = cache / f"{symbol}-metrics-{tag}.parquet"
        if cached.exists():
            hit = read_parquet_or_discard(cached)
            if hit is not None:
                parts.append(hit)
                continue

        days = pd.date_range(month_start, month_start + pd.offsets.MonthEnd(0), freq="D")
        with requests.Session() as session:
            with ThreadPoolExecutor(max_workers=workers) as pool:
                frames = list(pool.map(
                    lambda d: _fetch_metrics_day(symbol, d.strftime("%Y-%m-%d"), session),
                    days,
                ))
        frames = [f for f in frames if f is not None and len(f)]
        if not frames:
            if verbose:
                log.info("  metrics miss %s", tag)
            continue

        month_df = pd.concat(frames, ignore_index=True)
        write_parquet_atomic(month_df, cached)
        parts.append(month_df)
        if verbose:
            log.info("  metrics ok   %s  (%s rows from %d days)", tag,
                     f"{len(month_df):,}", len(frames))

    if not parts:
        raise DataUnavailableError(
            f"no metrics data for {symbol} {start}..{end}. "
            f"BTCUSDT metrics begin 2020-09; other symbols start later."
        )

    df = pd.concat(parts, ignore_index=True)
    df["ts"] = pd.to_datetime(df["create_time"]).astype("datetime64[ns]")
    keep = ["ts"] + [c for c in METRIC_COLS if c in df.columns]
    out = (df[keep].rename(columns=METRIC_COLS)
           .drop_duplicates("ts").sort_values("ts").set_index("ts"))
    out = out.astype("float64")
    if verbose:
        log.info("  metrics: %s rows %s..%s", f"{len(out):,}",
                 f"{out.index[0]:%Y-%m-%d}", f"{out.index[-1]:%Y-%m-%d}")
    return out

universe

Point-in-time universe construction.

Writing down a list of symbols from memory is hindsight dressed as data — you will recall the ones that survived. The universe here is built mechanically: enumerate everything the archive holds, then ask each symbol whether it had data in a given month. A coin that listed in 2023 fails that test; a coin that died in 2022 passes it, and belongs in the sample.

Liquidity screening is a separate problem and a subtler one. Ranking by full-sample average volume is lookahead — it knows which coins would go on to matter. liquidity_screen ranks on a TRAILING window only, so the universe at each bar is the one you could actually have chosen at that bar.

list_symbols

list_symbols(market: str = 'um', pattern: str = '[A-Z0-9]+USDT') -> list[str]

Every symbol the archive holds for market.

Source code in nullres/data/universe.py
def list_symbols(market: str = "um", pattern: str = r"[A-Z0-9]+USDT") -> list[str]:
    """Every symbol the archive holds for `market`."""
    prefix = PREFIXES[market]
    out, marker = [], None
    while True:
        params = {"delimiter": "/", "prefix": prefix}
        if marker:
            params["marker"] = marker
        resp = requests.get(LISTING, params=params, timeout=60)
        resp.raise_for_status()
        root = ET.fromstring(resp.text)
        ns = {"s3": root.tag.split("}")[0].strip("{")}
        got = [p.find("s3:Prefix", ns).text for p in root.findall("s3:CommonPrefixes", ns)]
        out.extend(s[len(prefix):].strip("/") for s in got)
        truncated = root.find("s3:IsTruncated", ns)
        if truncated is None or truncated.text != "true" or not got:
            break
        marker = got[-1]
    return sorted(s for s in out if re.fullmatch(pattern, s))

universe_as_of

universe_as_of(month: str, interval: str = '4h', market: str = 'um', workers: int = 24, cache_dir: str = 'data', verbose: bool = True) -> list[str]

Symbols that were trading in month — nothing about what came after.

Cached, because 787 HEAD requests is rude to repeat and the answer for a past month never changes.

Source code in nullres/data/universe.py
def universe_as_of(month: str, interval: str = "4h", market: str = "um",
                   workers: int = 24, cache_dir: str = "data",
                   verbose: bool = True) -> list[str]:
    """Symbols that were trading in `month` — nothing about what came after.

    Cached, because 787 HEAD requests is rude to repeat and the answer for a
    past month never changes.
    """
    cache = Path(cache_dir) / f"universe-{market}-{interval}-{month}.txt"
    if cache.exists():
        return [s for s in cache.read_text().split() if s]

    candidates = list_symbols(market)
    if verbose:
        log.info("  %s symbols in archive; testing %s", f"{len(candidates):,}", month)

    base = KLINES[market]

    def existed(symbol: str) -> str | None:
        url = f"{base}/{symbol}/{interval}/{symbol}-{interval}-{month}.zip"
        try:
            return symbol if requests.head(url, timeout=30).status_code == 200 else None
        except requests.RequestException:
            return None

    with ThreadPoolExecutor(max_workers=workers) as pool:
        live = [s for s in pool.map(existed, candidates) if s]

    cache.parent.mkdir(parents=True, exist_ok=True)
    cache.write_text("\n".join(live))
    if verbose:
        log.info("  %s were trading in %s", f"{len(live):,}", month)
    return live

delisted_from_cache

delisted_from_cache(symbols: list[str], interval: str, end: str, cache_dir: str = 'data', market: str = 'um', grace_months: int = 2) -> dict[str, str]

Symbols whose cached archive stops well before the sample ends.

Works entirely off local parquet files, so the survivorship check runs offline and costs nothing. grace_months absorbs the normal lag between the end of a range and the archive catching up — without it, every symbol looks delisted in the current month.

Source code in nullres/data/universe.py
def delisted_from_cache(symbols: list[str], interval: str, end: str,
                        cache_dir: str = "data", market: str = "um",
                        grace_months: int = 2) -> dict[str, str]:
    """Symbols whose cached archive stops well before the sample ends.

    Works entirely off local parquet files, so the survivorship check runs
    offline and costs nothing. `grace_months` absorbs the normal lag between
    the end of a range and the archive catching up — without it, every symbol
    looks delisted in the current month.
    """
    tag = "" if market == "spot" else f"{market}-"
    cutoff = (pd.Period(end, freq="M") - grace_months).strftime("%Y-%m")

    out: dict[str, str] = {}
    for symbol in symbols:
        months = sorted(
            p.stem.rsplit("-", 2)[-2] + "-" + p.stem.rsplit("-", 2)[-1]
            for p in Path(cache_dir).glob(f"{tag}{symbol}-{interval}-*.parquet")
        )
        if months and months[-1] < cutoff:
            out[symbol] = months[-1]
    return out

liquidity_screen

liquidity_screen(volumes: DataFrame, top_n: int = 40, window: int = 180, min_history: int = 180) -> DataFrame

Boolean mask: is this symbol in the top-N by TRAILING dollar volume?

The trailing window is what makes this point-in-time. Screening on full-sample volume would quietly select the coins that went on to become important — a survivorship bias that hides inside what looks like ordinary data hygiene.

Parameters:

Name Type Description Default
volumes DataFrame

ts x symbol quote volume per bar.

required
window int

bars of history the ranking is computed over.

180
min_history int

a symbol needs at least this much history to be eligible, so newly listed coins are not ranked on three days of launch hype.

180
Source code in nullres/data/universe.py
def liquidity_screen(volumes: pd.DataFrame, top_n: int = 40,
                     window: int = 180, min_history: int = 180) -> pd.DataFrame:
    """Boolean mask: is this symbol in the top-N by TRAILING dollar volume?

    The trailing window is what makes this point-in-time. Screening on
    full-sample volume would quietly select the coins that went on to become
    important — a survivorship bias that hides inside what looks like ordinary
    data hygiene.

    Args:
        volumes: ts x symbol quote volume per bar.
        window: bars of history the ranking is computed over.
        min_history: a symbol needs at least this much history to be eligible,
            so newly listed coins are not ranked on three days of launch hype.
    """
    trailing = volumes.rolling(window, min_periods=min_history).mean()
    ranks = trailing.rank(axis=1, ascending=False, na_option="bottom")
    eligible = trailing.notna()
    return (ranks <= top_n) & eligible

synthetic

Synthetic bars with known ground truth.

Two uses, both essential:

synthetic_bars — a geometric random walk. By construction there is NO edge. Any strategy that profits on this after costs has a bug. This is the single most useful test in the repo.

synthetic_bars(edge=...) — a walk with a real, known autocorrelation. If your pipeline CANNOT find this, it is too weak or mis-wired, and a null result on real data tells you nothing.

Run both before trusting any result. A harness that fails either is not measuring what you think it is.

synthetic_funding

synthetic_funding(bars: DataFrame, seed: int = 0) -> DataFrame

Funding settlements every 8h, uncorrelated with future returns.

Exists so the null control exercises the SAME code path as a real run. Without it the random-walk check runs on 32 features while the live config runs on 47, and a broken funding join would never reach the one test whose whole job is to catch fabricated edge.

The values are noise by construction, so any strategy that profits from them on this data has found a bug in the join, not a signal.

Source code in nullres/data/synthetic.py
def synthetic_funding(bars: pd.DataFrame, seed: int = 0) -> pd.DataFrame:
    """Funding settlements every 8h, uncorrelated with future returns.

    Exists so the null control exercises the SAME code path as a real run.
    Without it the random-walk check runs on 32 features while the live config
    runs on 47, and a broken funding join would never reach the one test whose
    whole job is to catch fabricated edge.

    The values are noise by construction, so any strategy that profits from
    them on this data has found a bug in the join, not a signal.
    """
    rng = np.random.default_rng(seed + 101)
    idx = pd.date_range(bars.index[0], bars.index[-1] + pd.Timedelta("8h"), freq="8h")
    return pd.DataFrame(
        {
            "funding_rate": rng.normal(0.0001, 0.0003, len(idx)),
            "funding_hours": np.full(len(idx), 8.0),
        },
        index=pd.DatetimeIndex(idx, name="ts"),
    )

synthetic_metrics

synthetic_metrics(bars: DataFrame, seed: int = 0) -> DataFrame

Open interest and positioning ratios, also pure noise.

Open interest is generated as a random walk with drift so it is non-stationary like the real thing — that way the stationarity discipline in features/derivatives.py is genuinely exercised.

Source code in nullres/data/synthetic.py
def synthetic_metrics(bars: pd.DataFrame, seed: int = 0) -> pd.DataFrame:
    """Open interest and positioning ratios, also pure noise.

    Open interest is generated as a random walk with drift so it is
    non-stationary like the real thing — that way the stationarity discipline
    in `features/derivatives.py` is genuinely exercised.
    """
    rng = np.random.default_rng(seed + 202)
    idx = pd.date_range(bars.index[0], bars.index[-1] + pd.Timedelta("1h"), freq="1h")
    n = len(idx)
    oi = 1e5 * np.exp(np.cumsum(rng.normal(0.0002, 0.01, n)))
    return pd.DataFrame(
        {
            "open_interest": oi,
            "oi_value": oi * 30_000,
            "top_trader_accounts_ls": np.abs(rng.normal(2.5, 0.4, n)),
            "top_trader_positions_ls": np.abs(rng.normal(1.2, 0.2, n)),
            "all_accounts_ls": np.abs(rng.normal(2.0, 0.3, n)),
            "taker_buy_sell_ratio": np.abs(rng.normal(1.0, 0.2, n)),
        },
        index=pd.DatetimeIndex(idx, name="ts"),
    )

synthetic_bars

synthetic_bars(n: int = 40000, seed: int = 0, interval: str = '1h', sigma: float = 0.004, edge: float = 0.0, start: str = '2020-01-01') -> DataFrame

Geometric random walk in OHLCV form.

Parameters:

Name Type Description Default
edge float

AR(1) coefficient on log returns. 0.0 means a pure martingale — unpredictable by construction. ~0.05 is a faint but genuinely learnable edge; real liquid markets sit near 0.0 to 0.02.

0.0
Source code in nullres/data/synthetic.py
def synthetic_bars(n: int = 40_000, seed: int = 0, interval: str = "1h",
                   sigma: float = 0.004, edge: float = 0.0,
                   start: str = "2020-01-01") -> pd.DataFrame:
    """Geometric random walk in OHLCV form.

    Args:
        edge: AR(1) coefficient on log returns. 0.0 means a pure martingale —
            unpredictable by construction. ~0.05 is a faint but genuinely
            learnable edge; real liquid markets sit near 0.0 to 0.02.
    """
    rng = np.random.default_rng(seed)
    shocks = rng.normal(0.0, sigma, n)

    if edge:
        logret = np.empty(n)
        logret[0] = shocks[0]
        for i in range(1, n):
            logret[i] = edge * logret[i - 1] + shocks[i]
    else:
        logret = shocks

    close = 30_000.0 * np.exp(np.cumsum(logret))

    # Build O/H/L around the close so OHLC invariants always hold.
    open_ = np.empty(n)
    open_[0] = close[0] * (1 + rng.normal(0, sigma / 8))
    open_[1:] = close[:-1] * (1 + rng.normal(0, sigma / 8, n - 1))
    hi_wick = np.abs(rng.normal(0, sigma / 2, n))
    lo_wick = np.abs(rng.normal(0, sigma / 2, n))
    high = np.maximum(open_, close) * (1 + hi_wick)
    low = np.minimum(open_, close) * (1 - lo_wick)

    return pd.DataFrame(
        {
            "open": open_,
            "high": high,
            "low": low,
            "close": close,
            "volume": np.abs(rng.normal(500, 150, n)),
            "trades": np.abs(rng.normal(3_000, 800, n)),
        },
        index=pd.date_range(start, periods=n, freq=_interval_freq(interval), name="ts"),
    )

cache

Writing to the parquet cache without leaving corpses behind.

crosssec._guard_metrics_fetch refuses to start a download it estimates at four hours, on the grounds that a silent multi-hour fetch is not something a tool should do to you. The corollary went unhandled: a download that long WILL be interrupted — Ctrl-C, a laptop lid, an OOM kill — and DataFrame.to_parquet writes in place. An interrupt part-way through leaves a truncated file at exactly the path every later run treats as authoritative.

That failure is worse than a missing file in three ways. It is permanent, since nothing ever rewrites a path that already exists. It is silent until the next read, which may be days later. And it surfaces as a parquet decode error deep inside pyarrow, naming neither the cache nor the download that produced it — so the obvious reading is "the library is broken", not "delete this one file".

Writing to a temporary file in the same directory and renaming it into place fixes it. os.replace is atomic on POSIX and on Windows, so a reader sees either the previous file or the complete new one, never a partial write.

write_parquet_atomic

write_parquet_atomic(df: DataFrame, path: str | Path) -> Path

Write df to path so that readers never observe a partial file.

The temporary file is created beside the target rather than in the system temp directory, because os.replace is only atomic within one filesystem and a cache directory may well be on a different mount.

Source code in nullres/data/cache.py
def write_parquet_atomic(df: pd.DataFrame, path: str | Path) -> Path:
    """Write `df` to `path` so that readers never observe a partial file.

    The temporary file is created beside the target rather than in the system
    temp directory, because `os.replace` is only atomic within one filesystem
    and a cache directory may well be on a different mount.
    """
    path = Path(path)
    path.parent.mkdir(parents=True, exist_ok=True)
    tmp = path.with_name(f"{path.name}.{os.getpid()}.tmp")
    try:
        df.to_parquet(tmp)
        os.replace(tmp, path)
    except BaseException:
        # BaseException, not Exception: KeyboardInterrupt is the single most
        # likely way to land here, and it is not an Exception subclass. Leaving
        # the temp file behind on Ctrl-C would recreate the litter this exists
        # to prevent, one `.tmp` per abandoned download.
        tmp.unlink(missing_ok=True)
        raise
    return path

read_parquet_or_discard

read_parquet_or_discard(path: str | Path) -> DataFrame | None

Read a cached parquet, deleting and reporting it if it is unreadable.

Files written before write_parquet_atomic may already be truncated, and a corrupt cache entry should cost one re-download rather than an afternoon of reading tracebacks. Returning None means "treat this as a cache miss".

Source code in nullres/data/cache.py
def read_parquet_or_discard(path: str | Path) -> pd.DataFrame | None:
    """Read a cached parquet, deleting and reporting it if it is unreadable.

    Files written before `write_parquet_atomic` may already be truncated, and a
    corrupt cache entry should cost one re-download rather than an afternoon of
    reading tracebacks. Returning None means "treat this as a cache miss".
    """
    path = Path(path)
    try:
        return pd.read_parquet(path)
    except Exception as exc:                              # noqa: BLE001
        log.warning("  discarding unreadable cache file %s (%s: %s); "
                    "it will be re-fetched", path.name, type(exc).__name__, exc)
        path.unlink(missing_ok=True)
        return None

Features

technical

Feature engineering.

Two hard rules, both enforced by nullres.audit:

  1. POINT-IN-TIME. Every value at bar t uses only bars <= t. In practice this means: rolling windows only, never .shift(-k), never an expanding stat over the full sample, never fillna(method="bfill"), never a global mean/std for scaling. audit.check_point_in_time recomputes features on truncated data and asserts the last row is unchanged.

  2. STATIONARY. No raw price levels. BTC ran 4k -> 100k over this sample; a tree that learned "close > 60000" learned the calendar, not the market. Everything below is a ratio, a z-score, or a bounded oscillator.

rsi

rsi(close: Series, n: int = 14) -> Series

Wilder RSI, bounded 0..100.

The zero-loss case has to be handled explicitly. up / dn divides by zero whenever the window contains no down moves, and mapping that to NaN — the obvious defensive spelling — is wrong twice over. RSI is 100 there by definition, not unknown; and because pipeline.prepare keeps only rows where EVERY feature is present, one NaN here evicts the whole bar and the other 45 features with it. A silent divide-by-zero became silent sample loss, concentrated in exactly the strong uptrends a momentum feature is supposed to describe.

Source code in nullres/features/technical.py
def rsi(close: pd.Series, n: int = 14) -> pd.Series:
    """Wilder RSI, bounded 0..100.

    The zero-loss case has to be handled explicitly. `up / dn` divides by zero
    whenever the window contains no down moves, and mapping that to NaN — the
    obvious defensive spelling — is wrong twice over. RSI is 100 there by
    definition, not unknown; and because `pipeline.prepare` keeps only rows
    where EVERY feature is present, one NaN here evicts the whole bar and the
    other 45 features with it. A silent divide-by-zero became silent sample
    loss, concentrated in exactly the strong uptrends a momentum feature is
    supposed to describe.
    """
    delta = close.diff()
    up = delta.clip(lower=0).ewm(alpha=1 / n, adjust=False).mean()
    dn = (-delta.clip(upper=0)).ewm(alpha=1 / n, adjust=False).mean()

    out = 100 - 100 / (1 + up / dn.replace(0, np.nan))
    # NaN comparisons are False, so the warmup rows stay NaN as intended.
    out = out.mask((dn == 0) & (up > 0), 100.0)
    # No movement in either direction: neutral by convention, not missing.
    return out.mask((dn == 0) & (up == 0), 50.0)

build_features

build_features(df: DataFrame, funding=None, metrics=None) -> DataFrame

Return the feature matrix, indexed identically to df.

Leading rows are NaN until the longest window fills; callers drop them.

funding and metrics are optional Binance futures frames. When supplied, derivative features are appended — see features/derivatives.py, where the point-in-time join is the part that matters.

Source code in nullres/features/technical.py
def build_features(df: pd.DataFrame, funding=None, metrics=None) -> pd.DataFrame:
    """Return the feature matrix, indexed identically to `df`.

    Leading rows are NaN until the longest window fills; callers drop them.

    `funding` and `metrics` are optional Binance futures frames. When supplied,
    derivative features are appended — see `features/derivatives.py`, where the
    point-in-time join is the part that matters.
    """
    f = pd.DataFrame(index=df.index)
    close, high, low = df["close"], df["high"], df["low"]
    logret = np.log(close).diff()

    # --- momentum over several horizons -----------------------------------
    for lag in (1, 2, 3, 6, 12, 24, 72):
        f[f"ret_{lag}"] = logret.rolling(lag).sum()

    # --- volatility regime --------------------------------------------------
    for w in (12, 24, 72, 168):
        f[f"vol_{w}"] = logret.rolling(w).std()
    f["volratio"] = f["vol_12"] / f["vol_168"]

    # --- stretch / mean-reversion ------------------------------------------
    for w in (12, 24, 72, 168):
        f[f"z_{w}"] = _zscore(close, w)

    f["rsi_14"] = rsi(close)

    # --- bar shape ----------------------------------------------------------
    rng = (high - low).replace(0, np.nan)
    body_top = df[["open", "close"]].max(axis=1)
    body_bot = df[["open", "close"]].min(axis=1)
    f["atr_pct"] = atr(df) / close
    f["hl_range"] = rng / close
    f["upper_wick"] = (high - body_top) / rng
    f["lower_wick"] = (body_bot - low) / rng
    f["body"] = (close - df["open"]) / rng

    # --- channel position / breakout ---------------------------------------
    for w in (24, 72):
        hh = high.rolling(w).max()
        ll = low.rolling(w).min()
        f[f"donch_{w}"] = (close - ll) / (hh - ll).replace(0, np.nan)

    # --- trend acceleration -------------------------------------------------
    ema_fast = close.ewm(span=12, adjust=False).mean()
    ema_slow = close.ewm(span=26, adjust=False).mean()
    macd = ema_fast - ema_slow
    f["macd_n"] = (macd - macd.ewm(span=9, adjust=False).mean()) / close

    # --- flow ---------------------------------------------------------------
    f["vol_z"] = _zscore(df["volume"], 72)
    f["trade_z"] = _zscore(df["trades"], 72)
    f["avg_trade"] = _zscore(df["volume"] / df["trades"].replace(0, np.nan), 72)
    f["amihud"] = _zscore(logret.abs() / df["volume"].replace(0, np.nan), 72)

    # --- higher moments -----------------------------------------------------
    f["ret_skew_72"] = logret.rolling(72).skew()

    # --- calendar -----------------------------------------------------------
    f["hour"] = df.index.hour.astype("float64")
    f["dow"] = df.index.dayofweek.astype("float64")

    f = f.replace([np.inf, -np.inf], np.nan)

    if funding is not None or metrics is not None:
        from nullres.features.derivatives import build_derivative_features

        f = f.join(build_derivative_features(df, funding, metrics))
    return f

derivatives

Features from funding rates and open interest.

THE JOIN IS THE WHOLE PROBLEM. Everything else here is arithmetic.

A bar indexed at time T covers [T, T+interval) and CLOSES at T+interval. So a funding settlement or OI reading may be used for bar T only if its timestamp is strictly before T+interval. Joining on the bar's OPEN time throws away a bar of information; joining on anything at or after the close is lookahead, and it is the kind that produces a beautiful equity curve.

We use merge_asof(direction="backward", allow_exact_matches=False) against the bar's close instant. Exact matches are excluded because a settlement stamped exactly at the close is simultaneous with it, and "simultaneous" is not "available".

The auxiliary frames are also CLIPPED to the bar range before joining. That is not cosmetic: audit.check_point_in_time truncates the bars and recomputes, and without clipping the funding frame would still hold future rows, so a bad join direction would silently produce identical output and the check would pass. With clipping, the audit covers this surface too — tests/test_derivatives.py proves it by injecting a forward join and asserting it gets caught.

build_derivative_features

build_derivative_features(bars: DataFrame, funding: DataFrame | None = None, metrics: DataFrame | None = None) -> DataFrame

Stationary features from funding and open-interest data.

Levels are avoided throughout. Open interest grew ~10x over this sample; a model that learned "OI > 400k" learned the calendar, exactly as it would have from raw price.

Source code in nullres/features/derivatives.py
def build_derivative_features(bars: pd.DataFrame,
                              funding: pd.DataFrame | None = None,
                              metrics: pd.DataFrame | None = None) -> pd.DataFrame:
    """Stationary features from funding and open-interest data.

    Levels are avoided throughout. Open interest grew ~10x over this sample; a
    model that learned "OI > 400k" learned the calendar, exactly as it would
    have from raw price.
    """
    f = pd.DataFrame(index=bars.index)
    logclose = np.log(bars["close"])

    if funding is not None and len(funding):
        joined = _asof(bars, funding, ["funding_rate"])
        rate = joined["funding_rate"]
        f["funding"] = rate
        # Settlements are 8-hourly; on a 4h bar 3 settlements is ~24h. Windows
        # are expressed in BARS, so they scale with the configured timeframe.
        f["funding_ma_3"] = rate.rolling(3).mean()
        f["funding_ma_21"] = rate.rolling(21).mean()
        f["funding_z"] = (
            (rate - rate.rolling(90).mean()) / rate.rolling(90).std()
        )
        f["funding_cum_7d"] = rate.rolling(42).sum()

    if metrics is not None and len(metrics):
        cols = [c for c in ["open_interest", "all_accounts_ls",
                            "top_trader_positions_ls", "taker_buy_sell_ratio"]
                if c in metrics.columns]
        joined = _asof(bars, metrics, cols)

        if "open_interest" in joined:
            oi = joined["open_interest"].replace(0, np.nan)
            log_oi = np.log(oi)
            f["oi_chg_6"] = log_oi.diff(6)
            f["oi_chg_24"] = log_oi.diff(24)
            f["oi_z"] = (log_oi - log_oi.rolling(168).mean()) / log_oi.rolling(168).std()
            # Same price move, opposite meaning depending on whether positions
            # are being opened or closed.
            f["oi_price_div"] = np.sign(log_oi.diff(24)) * np.sign(logclose.diff(24))

        if "all_accounts_ls" in joined:
            f["ls_accounts"] = joined["all_accounts_ls"]
        if "top_trader_positions_ls" in joined:
            f["ls_top_positions"] = joined["top_trader_positions_ls"]
        if {"all_accounts_ls", "top_trader_positions_ls"} <= set(joined.columns):
            f["ls_spread"] = (joined["top_trader_positions_ls"]
                              - joined["all_accounts_ls"])
        if "taker_buy_sell_ratio" in joined:
            taker = joined["taker_buy_sell_ratio"]
            f["taker_ratio"] = taker
            f["taker_z"] = (
                (taker - taker.rolling(72).mean()) / taker.rolling(72).std()
            )

    return f.replace([np.inf, -np.inf], np.nan)

Labels

targets

Label construction.

Every label returns a frame with a uniform contract:

y        int    0/1 target, NaN where the bar is unlabelled (dropped later)
t_end    int    positional index of the bar at which the label RESOLVES
ret      float  the log return the label is derived from, for diagnostics
sigma    float  volatility estimate at decision time, known at bar t

t_end is the load-bearing column. A label spanning bars t..t+20 must not sit in a training set whose test window begins at t+5 — the training label already contains the answer to the test period. nullres.validation purges on this column. A fixed purge constant is only correct when every label has the same horizon, which stops being true the moment you use barriers.

On label choice: next_bar_sign is the honest version of the baseline's label, and it is almost pure noise. A 1h BTC bar's empirical mean absolute move is ~0.40% against a ~0.24% round trip, so you are asking a model to call a coin flip well enough to clear 60% of the move. (nullres budget quotes 45% instead, because it uses the Gaussian E|move| of 0.54%; fat tails make the real move smaller and the real bar higher — see docs/03.) triple_barrier instead asks a question worth answering — "does price travel 1.5 sigma up before it travels 1.5 sigma down" — which has a real, if small, autocorrelation structure and a payoff that exceeds costs.

next_bar_sign

next_bar_sign(df: DataFrame, cfg) -> DataFrame

1 if the next bar's close exceeds this one's. The baseline's label.

Kept for comparison, not recommended. Resolves one bar ahead.

Source code in nullres/labels/targets.py
def next_bar_sign(df: pd.DataFrame, cfg) -> pd.DataFrame:
    """1 if the next bar's close exceeds this one's. The baseline's label.

    Kept for comparison, not recommended. Resolves one bar ahead.
    """
    logret = np.log(df["close"]).diff()
    fwd = logret.shift(-1)
    n = len(df)
    return pd.DataFrame(
        {
            "y": (fwd > 0).astype("float64").where(fwd.notna()),
            "t_end": np.minimum(np.arange(n) + 1, n - 1),
            "ret": fwd,
            "sigma": _sigma(df["close"], cfg.vol_window),
        },
        index=df.index,
    )

fwd_return

fwd_return(df: DataFrame, cfg) -> DataFrame

Sign of the vol-scaled return over horizon bars.

Bars whose move is smaller than deadband sigma are left unlabelled. That matters: without it, roughly half the training set is noise the model tries to fit, and the fit it finds is spurious.

Source code in nullres/labels/targets.py
def fwd_return(df: pd.DataFrame, cfg) -> pd.DataFrame:
    """Sign of the vol-scaled return over `horizon` bars.

    Bars whose move is smaller than `deadband` sigma are left unlabelled. That
    matters: without it, roughly half the training set is noise the model tries
    to fit, and the fit it finds is spurious.
    """
    n = len(df)
    logclose = np.log(df["close"])
    fwd = logclose.shift(-cfg.horizon) - logclose
    sigma = _sigma(df["close"], cfg.vol_window)
    scale = sigma * np.sqrt(cfg.horizon)

    scaled = fwd / scale.replace(0, np.nan)
    y = pd.Series(np.nan, index=df.index, dtype="float64")
    y[scaled > cfg.deadband] = 1.0
    y[scaled < -cfg.deadband] = 0.0
    y[fwd.isna()] = np.nan

    return pd.DataFrame(
        {
            "y": y,
            "t_end": np.minimum(np.arange(n) + cfg.horizon, n - 1),
            "ret": fwd,
            "sigma": sigma,
        },
        index=df.index,
    )

triple_barrier

triple_barrier(df: DataFrame, cfg) -> DataFrame

López de Prado triple barrier, vectorised over the horizon.

From the close of bar t, place a profit barrier at +uppersigma and a stop at -lowersigma, plus a vertical barrier horizon bars out. Label 1 if the upper barrier is touched first, 0 if the lower is, and by the sign of the realised return if the vertical barrier is reached first.

The barriers are volatility-scaled, so the label means the same thing in a calm 2023 and a violent March 2020 — a fixed 1% target is a different question in each regime, and mixing the two is why fixed-percent labels train models that only work in the regime that dominated the sample.

When both barriers fall inside one bar, OHLC cannot tell us which came first. We assume the STOP hit first. That is pessimistic by design: the alternative silently inflates every result you will ever produce here.

Source code in nullres/labels/targets.py
def triple_barrier(df: pd.DataFrame, cfg) -> pd.DataFrame:
    """López de Prado triple barrier, vectorised over the horizon.

    From the close of bar t, place a profit barrier at +upper*sigma and a stop
    at -lower*sigma, plus a vertical barrier `horizon` bars out. Label 1 if the
    upper barrier is touched first, 0 if the lower is, and by the sign of the
    realised return if the vertical barrier is reached first.

    The barriers are volatility-scaled, so the label means the same thing in a
    calm 2023 and a violent March 2020 — a fixed 1% target is a different
    question in each regime, and mixing the two is why fixed-percent labels
    train models that only work in the regime that dominated the sample.

    When both barriers fall inside one bar, OHLC cannot tell us which came
    first. We assume the STOP hit first. That is pessimistic by design: the
    alternative silently inflates every result you will ever produce here.
    """
    n = len(df)
    close = df["close"].to_numpy(dtype="float64")
    high = df["high"].to_numpy(dtype="float64")
    low = df["low"].to_numpy(dtype="float64")
    sigma = _sigma(df["close"], cfg.vol_window)
    sig = sigma.to_numpy(dtype="float64")

    upper = close * np.exp(cfg.upper * sig)
    lower = close * np.exp(-cfg.lower * sig)

    idx = np.arange(n)
    side = np.zeros(n)                     # +1 profit, -1 stop, 0 unresolved
    t_end = np.minimum(idx + cfg.horizon, n - 1)
    open_ = np.ones(n, dtype=bool) & np.isfinite(sig)

    for h in range(1, cfg.horizon + 1):
        j = np.minimum(idx + h, n - 1)
        in_range = (idx + h) < n
        live = open_ & in_range
        if not live.any():
            break

        hit_dn = live & (low[j] <= lower)
        hit_up = live & (high[j] >= upper) & ~hit_dn   # stop wins ties

        newly = hit_up | hit_dn
        side[newly] = np.where(hit_up[newly], 1.0, -1.0)
        t_end[newly] = j[newly]
        open_ &= ~newly

    # Vertical barrier: unresolved paths fall back to the sign of the return.
    logclose = np.log(close)
    ret = logclose[t_end] - logclose
    y = np.where(side > 0, 1.0, np.where(side < 0, 0.0, (ret > 0).astype(float)))

    # A label that runs off the end of the sample never resolved — drop it.
    unresolved_tail = (idx + cfg.horizon) >= n
    y = np.where(unresolved_tail | ~np.isfinite(sig), np.nan, y)

    return pd.DataFrame(
        {"y": y, "t_end": t_end, "ret": ret, "sigma": sigma.to_numpy()},
        index=df.index,
    )

Models

classifier

Model construction and out-of-sample prediction.

Every .fit() in this repository is a place that could accidentally train on the future, so the set of them is kept small, deliberate, and pinned by tests/test_packaging.py::test_no_unaudited_fit_sites. There are three:

classifier.fit_predict_walk_forward   the single-asset walk-forward
classifier.feature_importance         refits the last fold to permute it
crosssec.fit_predict_panel            the panel walk-forward, split on TIME

The third is easy to miss and long went unmentioned — the docs claimed a single call site while the cross-sectional path, which produced the strongest result in the project, had its own. Each is purged independently, so nothing leaks; the risk was that a fourth could appear without anyone noticing. The test now fails if one does.

make_model

make_model(cfg)

Build an unfitted estimator from a ModelConfig.

Source code in nullres/models/classifier.py
def make_model(cfg):
    """Build an unfitted estimator from a ModelConfig."""
    if cfg.kind == "hgb":
        return HistGradientBoostingClassifier(
            max_iter=cfg.max_iter,
            learning_rate=cfg.learning_rate,
            max_depth=cfg.max_depth,
            l2_regularization=cfg.l2,
            min_samples_leaf=cfg.min_samples_leaf,
            random_state=cfg.seed,
            early_stopping=False,
        )
    if cfg.kind == "logistic":
        # Scaling must be fitted inside the fold, hence the pipeline: fitting a
        # scaler on the whole sample leaks test-period mean and variance into
        # training. It is a small leak, and it is still a leak.
        return make_pipeline(
            StandardScaler(),
            LogisticRegression(C=1.0 / max(cfg.l2, 1e-6), max_iter=1_000,
                               random_state=cfg.seed),
        )
    raise ConfigError(f"unknown model kind {cfg.kind!r}; choose hgb or logistic")

fit_predict_walk_forward

fit_predict_walk_forward(X: DataFrame, y: Series, t_end: ndarray, split_cfg, model_cfg, use_uniqueness: bool = True, verbose: bool = True) -> tuple[Series, list[dict]]

Out-of-sample P(class 1) for every bar in a test fold.

Bars outside every test window stay NaN — they are training-only and must never appear in a backtest. Rows with a NaN label are predicted but not trained on, which is how the deadband in fwd_return works.

Returns (proba, fold_reports).

Source code in nullres/models/classifier.py
def fit_predict_walk_forward(
    X: pd.DataFrame,
    y: pd.Series,
    t_end: np.ndarray,
    split_cfg,
    model_cfg,
    use_uniqueness: bool = True,
    verbose: bool = True,
) -> tuple[pd.Series, list[dict]]:
    """Out-of-sample P(class 1) for every bar in a test fold.

    Bars outside every test window stay NaN — they are training-only and must
    never appear in a backtest. Rows with a NaN label are predicted but not
    trained on, which is how the deadband in `fwd_return` works.

    Returns (proba, fold_reports).
    """
    proba = pd.Series(np.nan, index=X.index, dtype="float64")
    y_arr = y.to_numpy(dtype="float64")
    weights = uniqueness_weights(t_end, len(X)) if use_uniqueness else np.ones(len(X))
    reports: list[dict] = []

    for k, (train, test) in enumerate(purged_walk_forward(t_end, split_cfg), start=1):
        labelled = train[np.isfinite(y_arr[train])]
        if labelled.size < 100:
            continue
        classes = np.unique(y_arr[labelled])
        if classes.size < 2:
            if verbose:
                log.info("  fold %d: only one class in training set, skipped", k)
            continue

        model = make_model(model_cfg)
        model.fit(
            X.iloc[labelled],
            y_arr[labelled].astype(int),
            **{"sample_weight": weights[labelled]} if use_uniqueness else {},
        )
        p = model.predict_proba(X.iloc[test])[:, 1]
        proba.iloc[test] = p

        y_test = y_arr[test]
        scored = np.isfinite(y_test)
        acc = (float(((p[scored] > 0.5) == (y_test[scored] > 0.5)).mean())
               if scored.any() else float("nan"))

        # AUC is the better read on whether ANY signal exists: accuracy at a
        # fixed 0.5 cut hides a model that ranks well but is poorly calibrated.
        auc = float("nan")
        if scored.sum() > 10 and np.unique(y_test[scored]).size == 2:
            from sklearn.metrics import roc_auc_score
            auc = float(roc_auc_score(y_test[scored].astype(int), p[scored]))

        report = {
            "fold": k,
            "train": int(labelled.size),
            "test": int(test.size),
            "base_rate": float(np.nanmean(y_arr[labelled])),
            "acc": acc,
            "auc": auc,
            "test_from": str(X.index[test[0]])[:10],
            "test_to": str(X.index[test[-1]])[:10],
        }
        reports.append(report)
        if verbose:
            log.info("  fold %d: train %s  test %s  [%s..%s]  acc %.4f  auc %.4f",
                     k, f"{labelled.size:>7,}", f"{test.size:>6,}",
                     report["test_from"], report["test_to"], acc, auc)

    if not reports:
        raise InsufficientDataError(
            "no fold produced predictions — check split.min_train and label config"
        )
    return proba, reports

feature_importance

feature_importance(X: DataFrame, y: Series, t_end: ndarray, split_cfg, model_cfg, n_repeats: int = 3) -> Series

Permutation importance on the LAST fold's test window only.

In-sample importances tell you what the model memorised. This tells you what actually carried out of sample, which is a much shorter list.

Source code in nullres/models/classifier.py
def feature_importance(X: pd.DataFrame, y: pd.Series, t_end: np.ndarray,
                       split_cfg, model_cfg, n_repeats: int = 3) -> pd.Series:
    """Permutation importance on the LAST fold's test window only.

    In-sample importances tell you what the model memorised. This tells you what
    actually carried out of sample, which is a much shorter list.
    """
    from sklearn.inspection import permutation_importance

    folds = list(purged_walk_forward(t_end, split_cfg))
    train, test = folds[-1]
    y_arr = y.to_numpy(dtype="float64")
    labelled = train[np.isfinite(y_arr[train])]
    scored = test[np.isfinite(y_arr[test])]

    model = make_model(model_cfg)
    model.fit(X.iloc[labelled], y_arr[labelled].astype(int))
    result = permutation_importance(
        model, X.iloc[scored], y_arr[scored].astype(int),
        n_repeats=n_repeats, random_state=model_cfg.seed, scoring="roc_auc",
    )
    return pd.Series(result.importances_mean, index=X.columns).sort_values(ascending=False)

Strategies

base

Context dataclass

Context(bars: DataFrame, features: DataFrame, label: DataFrame, cfg: object, oos_mask: Series, diagnostics: dict = dict(), verbose: bool = True)

Everything a strategy is allowed to see.

Note what is absent: there is no handle on the future, and oos_mask marks the bars a strategy is permitted to be judged on. Rule strategies could in principle trade the whole sample, but they are masked to the same window as the ML strategies so the comparison is fair — a rule evaluated over six years against a model evaluated over five is not a comparison.

Strategy

Bases: Protocol

positions
positions(ctx: Context) -> Series

Target position per bar, decided at that bar's close.

Source code in nullres/strategies/base.py
def positions(self, ctx: Context) -> pd.Series:
    """Target position per bar, decided at that bar's close."""
    ...

strategy_fingerprint

strategy_fingerprint(obj) -> str

A stable identity for a strategy instance, including its parameters.

repr will not do: these are plain classes, so the default repr embeds id(obj) and changes every run. This reads the instance dictionary instead, recursing into nested strategies — MLMeta holds a primary rule whose own parameters decide what the model is trained on.

Source code in nullres/strategies/base.py
def strategy_fingerprint(obj) -> str:
    """A stable identity for a strategy instance, including its parameters.

    `repr` will not do: these are plain classes, so the default repr embeds
    `id(obj)` and changes every run. This reads the instance dictionary
    instead, recursing into nested strategies — `MLMeta` holds a primary rule
    whose own parameters decide what the model is trained on.
    """
    parts = []
    for name, value in sorted(vars(obj).items()):
        inner = (strategy_fingerprint(value)
                 if hasattr(value, "positions") else repr(value))
        parts.append(f"{name}={inner}")
    return f"{type(obj).__name__}({','.join(parts)})"

cached_proba

cached_proba(ctx: Context, key: str, compute, extra: str = '')

Memoise walk-forward predictions across runs that share a context.

nullres sweep varies only sizing, which cannot change the model's output, so refitting 25 times would be pure waste. The cache key includes the label, split and model config, so any change that WOULD alter the predictions misses the cache instead of silently returning stale ones.

extra is for anything else that feeds the feature matrix. It exists because the fingerprint was incomplete: MLMeta appends primary_side to the features, which depends on the parameters of its primary rule, and none of those appeared in the key. Two MLMeta strategies with different primaries, evaluated against one prepared context, would have taken each other's predictions — silently, since a cache hit looks exactly like a fast computation. Nothing in the shipped configs varies the primary, so this was a trap rather than a live bug, which is the kind that survives longest.

Source code in nullres/strategies/base.py
def cached_proba(ctx: Context, key: str, compute, extra: str = ""):
    """Memoise walk-forward predictions across runs that share a context.

    `nullres sweep` varies only sizing, which cannot change the model's output,
    so refitting 25 times would be pure waste. The cache key includes the
    label, split and model config, so any change that WOULD alter the
    predictions misses the cache instead of silently returning stale ones.

    `extra` is for anything else that feeds the feature matrix. It exists
    because the fingerprint was incomplete: `MLMeta` appends `primary_side` to
    the features, which depends on the parameters of its primary rule, and none
    of those appeared in the key. Two `MLMeta` strategies with different
    primaries, evaluated against one prepared context, would have taken each
    other's predictions — silently, since a cache hit looks exactly like a fast
    computation. Nothing in the shipped configs varies the primary, so this was
    a trap rather than a live bug, which is the kind that survives longest.
    """
    fingerprint = (repr(ctx.cfg.label), repr(ctx.cfg.split), repr(ctx.cfg.model),
                   extra)
    slot = ctx.diagnostics.setdefault(key, {})
    if slot.get("fingerprint") == fingerprint and "proba" in slot:
        return slot["proba"], slot.get("folds", [])

    proba, folds = compute()
    slot.update(fingerprint=fingerprint, proba=proba, folds=folds)
    return proba, folds

mask_to_oos

mask_to_oos(pos: Series, ctx: Context) -> Series

Zero out any position outside the out-of-sample window.

Source code in nullres/strategies/base.py
def mask_to_oos(pos: pd.Series, ctx: Context) -> pd.Series:
    """Zero out any position outside the out-of-sample window."""
    return pos.where(ctx.oos_mask, 0.0).fillna(0.0)

crossover_state

crossover_state(fast: Series, slow: Series) -> Series

+1 while fast is above slow, -1 while below. Point-in-time safe.

Source code in nullres/strategies/base.py
def crossover_state(fast: pd.Series, slow: pd.Series) -> pd.Series:
    """+1 while fast is above slow, -1 while below. Point-in-time safe."""
    return pd.Series(np.where(fast > slow, 1.0, -1.0), index=fast.index)

rules

Rule-based strategies — the benchmarks any model has to clear.

These are deliberately simple and deliberately not tuned. Their job is to set the bar. A tuned rule is not a benchmark, it is another overfit strategy with fewer parameters.

SMACross

SMACross(fast: int = 50, slow: int = 200, allow_short: bool = False)

Long when the fast average is above the slow one, flat otherwise.

The oldest systematic strategy there is. It trades rarely, so costs barely register, which is exactly why it is hard to beat.

Source code in nullres/strategies/rules.py
def __init__(self, fast: int = 50, slow: int = 200, allow_short: bool = False):
    self.fast, self.slow, self.allow_short = fast, slow, allow_short

DonchianBreakout

DonchianBreakout(entry: int = 96, exit: int = 48)

Long on a new N-bar high, flat on a new M-bar low. Classic trend following.

Source code in nullres/strategies/rules.py
def __init__(self, entry: int = 96, exit: int = 48):
    self.entry, self.exit = entry, exit

VolTargetHold

VolTargetHold(target: float = 0.5, vol_window: int = 30, band: float = 0.1, max_leverage: float = 1.0)

Always long, but sized so that RISK is constant rather than notional.

This strategy makes no directional claim at all. It exists because of a measured asymmetry in the data:

lag-1 autocorrelation of returns      -0.029    (noise)
lag-1 autocorrelation of |returns|    +0.227    (strong)
lag-1 autocorrelation of 30-bar vol   +0.992    (near-deterministic)

Direction is unpredictable; volatility is extremely persistent. So rather than guessing which way the market goes, hold it continuously and vary the size by 1/sigma — cutting exposure when the market is violent and restoring it when it calms.

Note what this can and cannot do. It does not improve expected return; a lower-volatility path with the same drift compounds better, but the edge comes from risk management, not prediction. Judge it on Sharpe and drawdown, and expect total return at or slightly below buy & hold.

max_leverage=1.0 by default, so in calm regimes it is simply long and never borrows. That makes it deliverable in a spot account with no margin.

Source code in nullres/strategies/rules.py
def __init__(self, target: float = 0.50, vol_window: int = 30,
             band: float = 0.10, max_leverage: float = 1.0):
    self.target = target
    self.vol_window = vol_window
    self.band = band
    self.max_leverage = max_leverage

MeanReversionZ

MeanReversionZ(window: int = 72, entry: float = 2.0, exit: float = 0.5)

Fade stretched moves: long when the z-score is deeply negative, and vice versa.

Works in ranging regimes, gets destroyed in trending ones. Included partly as a benchmark and partly because its failure mode is instructive.

Source code in nullres/strategies/rules.py
def __init__(self, window: int = 72, entry: float = 2.0, exit: float = 0.5):
    self.window, self.entry, self.exit = window, entry, exit

ml

Machine-learning strategies.

Two formulations, and the difference between them matters more than the model:

MLDirection Predict the direction. The model must answer "which way", which on liquid intraday crypto is close to unanswerable.

MLMeta Meta-labelling. A simple rule decides WHICH WAY to trade; the model only decides WHETHER TO TAKE the trade. This is a far easier question — the model is allowed to say "I don't know" by declining, and declining is free. It also turns an unbalanced 3-class problem into a clean binary one, and the model's output maps naturally onto position size.

If you only take one structural idea from this repo, take the second one.

MLMeta

MLMeta(primary=None)

Meta-labelling on top of a moving-average trend filter.

The primary rule supplies the side. The label becomes "was the rule right?", which is trained only on bars where the rule actually had a position — the model never wastes capacity on bars it will not trade.

Source code in nullres/strategies/ml.py
def __init__(self, primary=None):
    self.primary = primary or SMACross(fast=24, slow=120, allow_short=True)

Orchestration

pipeline

End-to-end orchestration: bars -> features -> labels -> positions -> metrics.

One rule governs the ordering here. Features and labels are built on the FULL frame first, and only then are rows dropped and positions renumbered. Building them per-fold would be slower and no safer; building them after dropping rows would silently shorten every rolling window across the gaps.

prepare

prepare(cfg, verbose: bool = True) -> Context

Load data, build features and labels, align them, and mark the OOS window.

Source code in nullres/pipeline.py
def prepare(cfg, verbose: bool = True) -> Context:
    """Load data, build features and labels, align them, and mark the OOS window."""
    bars = load_bars(cfg.data)
    funding, metrics = load_auxiliary(cfg.data, verbose=verbose, bars=bars)
    features = build_features(bars, funding=funding, metrics=metrics)
    label = build_label(bars, cfg.label)

    # Drop the warmup period where rolling windows have not filled, plus any bar
    # with no volatility estimate. Rows with a NaN target are KEPT: the model
    # predicts on them, it just does not train on them.
    keep = features.notna().all(axis=1) & label["sigma"].notna() & label["ret"].notna()
    keep_arr = keep.to_numpy()
    if keep_arr.sum() < 1_000:
        raise InsufficientDataError(
            f"only {int(keep_arr.sum())} usable bars after alignment — "
            f"widen the date range or shorten the feature windows"
        )

    t_end = remap_t_end(label["t_end"].to_numpy(dtype=np.int64), keep_arr)
    bars, features, label = bars[keep], features[keep], label[keep].copy()
    label["t_end"] = t_end

    # The out-of-sample window is the union of every walk-forward test fold.
    # Every strategy, rules included, is judged only here.
    oos = np.zeros(len(bars), dtype=bool)
    for _, test in purged_walk_forward(t_end, cfg.split):
        oos[test] = True
    oos_mask = pd.Series(oos, index=bars.index)

    if verbose:
        labelled = label["y"].notna()
        log.info("\n%s usable bars | %d features | %s labelled | base rate %.3f",
                 f"{len(bars):,}", features.shape[1],
                 f"{int(labelled.sum()):,}", label["y"].mean())
        log.info("out-of-sample: %s bars (%s .. %s)", f"{int(oos.sum()):,}",
                 f"{bars.index[oos][0]:%Y-%m-%d}", f"{bars.index[oos][-1]:%Y-%m-%d}")
        for row in describe_folds(t_end, cfg.split, bars.index):
            log.info("  fold %d: train %s (purged %s)  test %s  [%s..%s]",
                     row["fold"], f"{row['train']:>7,}", f"{row['purged']:>4,}",
                     f"{row['test']:>6,}", row["test_from"], row["test_to"])

        for warning in coherence_warnings(cfg, bars):
            log.warning("  WARNING: %s", warning)

    return Context(bars=bars, features=features, label=label, cfg=cfg,
                   oos_mask=oos_mask, verbose=verbose)

ablate

ablate(ctx: Context, group: str) -> Context

Drop a feature group AFTER row alignment, for a matched-sample ablation.

Turning the data off in the config is not a controlled comparison: the derivative features carry their own warmup (oi_z needs 168 bars), so disabling them changes which rows survive the NaN mask, which changes the fold boundaries and the out-of-sample window. The two runs then differ in their samples as well as their features, and even buy & hold moves.

This drops the columns from an already-prepared context, so the rows, the splits and the benchmark are byte-identical and the only variable is the feature set.

Source code in nullres/pipeline.py
def ablate(ctx: Context, group: str) -> Context:
    """Drop a feature group AFTER row alignment, for a matched-sample ablation.

    Turning the data off in the config is not a controlled comparison: the
    derivative features carry their own warmup (`oi_z` needs 168 bars), so
    disabling them changes which rows survive the NaN mask, which changes the
    fold boundaries and the out-of-sample window. The two runs then differ in
    their samples as well as their features, and even buy & hold moves.

    This drops the columns from an already-prepared context, so the rows, the
    splits and the benchmark are byte-identical and the only variable is the
    feature set.
    """
    from nullres.features import DERIVATIVE_DOC

    groups = {"derivatives": set(DERIVATIVE_DOC)}
    if group not in groups:
        raise ConfigError(f"unknown feature group {group!r}; choose from {sorted(groups)}")

    drop = [c for c in ctx.features.columns if c in groups[group]]
    if not drop:
        raise ConfigError(f"no {group} features present to ablate")
    ctx.features = ctx.features.drop(columns=drop)
    ctx.diagnostics.clear()          # cached predictions are now stale
    return ctx

coherence_warnings

coherence_warnings(cfg, bars: DataFrame) -> list[str]

Catch configurations that cannot work, before spending compute on them.

These are not style notes. Each one describes a setup where the backtest will produce a number that means nothing.

Source code in nullres/pipeline.py
def coherence_warnings(cfg, bars: pd.DataFrame) -> list[str]:
    """Catch configurations that cannot work, before spending compute on them.

    These are not style notes. Each one describes a setup where the backtest
    will produce a number that means nothing.
    """
    from nullres.costs import breakeven_hold, required_accuracy

    out = []
    horizon = cfg.label.horizon
    hold = max(cfg.sizing.min_hold, 1)

    # A model trained to predict 24 bars ahead tells you nothing about whether
    # to keep a position for 500 — and vice versa.
    if hold > 4 * horizon or horizon > 4 * hold:
        out.append(
            f"label.horizon={horizon} but sizing.min_hold={hold}. The model "
            f"predicts a {horizon}-bar outcome while the strategy holds for "
            f"{hold} bars; these should be within a factor of ~2."
        )

    sigma = float(np.log(bars["close"]).diff().std())
    need = required_accuracy(sigma, hold, cfg.cost.fee_bps, cfg.cost.slippage_bps)
    if need > 0.56:
        be = breakeven_hold(sigma, 0.52, cfg.cost.fee_bps, cfg.cost.slippage_bps)
        target = "impossible at any accuracy" if need > 1.0 else f"{need:.1%} accuracy"
        out.append(
            f"at min_hold={hold} this strategy needs {target} just to break even. "
            f"A realistic 52% model would need to hold ~{be:,.0f} bars. "
            f"Run `nullres budget` for the full table."
        )
    return out

trials_so_far

trials_so_far(cfg, extra: int = 0, command: str | None = None) -> int

Multiple-testing exposure: everything looked at before reporting this.

Reads the run ledger rather than counting strategies in the current run. Counting only the current run is the mistake this replaces — it reported n_trials=6 for a project that had explored well over a hundred parameter combinations, which made every deflated Sharpe too generous.

extra is how many variants the run about to happen will evaluate. It is folded into the ledger's own dedupe rather than added on top — see runlog.count_trials — so verifying an existing result does not inflate that result's own correction.

Source code in nullres/pipeline.py
def trials_so_far(cfg, extra: int = 0, command: str | None = None) -> int:
    """Multiple-testing exposure: everything looked at before reporting this.

    Reads the run ledger rather than counting strategies in the current run.
    Counting only the current run is the mistake this replaces — it reported
    `n_trials=6` for a project that had explored well over a hundred parameter
    combinations, which made every deflated Sharpe too generous.

    `extra` is how many variants the run about to happen will evaluate. It is
    folded into the ledger's own dedupe rather than added on top — see
    `runlog.count_trials` — so verifying an existing result does not inflate
    that result's own correction.
    """
    from nullres.runlog import config_hash, count_trials, load_runs

    try:
        history = load_runs()
    except OSError:
        history = []
    pending = (config_hash(cfg), command, extra) if command and extra else None
    return max(count_trials(history, prior=getattr(cfg, "prior_trials", 0),
                            pending=pending) + (0 if pending else extra), 1)

trials_caveat

trials_caveat() -> str

Anything that makes the trial count a FLOOR rather than a measurement.

Source code in nullres/pipeline.py
def trials_caveat() -> str:
    """Anything that makes the trial count a FLOOR rather than a measurement."""
    from nullres.runlog import load_runs, unrecorded_variants

    try:
        unknown = unrecorded_variants(load_runs())
    except OSError:
        return ""
    if not unknown:
        return ""
    return (f"  NOTE: {unknown} ledger record(s) predate variant recording and "
            f"count as 1 trial each.\n        The true exposure is higher, so "
            f"the deflated Sharpe below is generous.")

run_pipeline

run_pipeline(cfg, verbose: bool = True, ctx: Context | None = None, n_trials: int | None = None) -> dict[str, dict]

Run every configured strategy and return {name: metrics}.

A caller may pass a prepared ctx to avoid recomputing features when only sizing or cost parameters change (see nullres sweep). The context's cfg is repointed at cfg so those overrides actually take effect — strategies read their parameters from ctx.cfg, not from the closure.

Source code in nullres/pipeline.py
def run_pipeline(cfg, verbose: bool = True, ctx: Context | None = None,
                 n_trials: int | None = None) -> dict[str, dict]:
    """Run every configured strategy and return {name: metrics}.

    A caller may pass a prepared `ctx` to avoid recomputing features when only
    sizing or cost parameters change (see `nullres sweep`). The context's cfg is
    repointed at `cfg` so those overrides actually take effect — strategies read
    their parameters from ctx.cfg, not from the closure.
    """
    if ctx is None:
        ctx = prepare(cfg, verbose=verbose)
    else:
        ctx.cfg = cfg
        ctx.verbose = verbose

    names = list(dict.fromkeys(["buy_hold", *cfg.strategies]))
    if n_trials is None:
        n_trials = trials_so_far(cfg, extra=len(names))
    results: dict[str, dict] = {}

    for name in names:
        if verbose:
            log.info("\n-> %s", name)
        strategy = build_strategy(name, cfg.params.get(name))
        positions = strategy.positions(ctx)
        result = backtest(ctx.bars, positions, cfg.cost)
        # Measured on the out-of-sample window only. Positions are zeroed
        # outside it, and averaging over those zeros deflates every Sharpe by
        # sqrt(oos fraction) — uniformly, so it hid in the rankings.
        metrics = summarize(result, cfg.data.bars_per_year, n_trials=n_trials,
                            mask=ctx.oos_mask)
        results[name] = metrics
        ctx.diagnostics.setdefault(name, {})["result"] = result

    return results