# ============================================================================= # Project : anomaly-atlas # File : benchmarks/synthetic/generators.py # Purpose : Synthetic series with KNOWN properties — the "test the tests" set # Author : Simon-Pierre Boucher # Contact : contact@spboucher.ai # Data src : hfmarketdata.io (sole data source) # Created : 2026-08-12 # Modified : 2026-08-12 # Platform : macOS / Apple Silicon (arm64) # License : All rights reserved (research code) # ============================================================================= """Synthetic price/return series with planted, analytically-known properties. Charter §8.1: before any detector touches real data it must (a) find NOTHING in a pure random walk, (b) recover every planted effect, and (c) flag a pure bid-ask-bounce series as an artifact, not an anomaly. These generators are the ground truth for that gate (see test_synthetic_gate.py). All series are generated from an explicit seed; no global RNG state. """ from __future__ import annotations import numpy as np def random_walk(n: int, sigma: float = 0.001, seed: int = 0) -> np.ndarray: """Pure log-price random walk. Ground truth: VR(q)=1, AC1(returns)=0.""" rng = np.random.default_rng(seed) return np.cumsum(rng.normal(0.0, sigma, n)) def ou_prices(n: int, kappa: float, sigma: float = 0.001, seed: int = 0) -> np.ndarray: """Mean-reverting (Ornstein-Uhlenbeck) log-price around 0. Discrete: p_t = (1 - kappa) * p_{t-1} + eps. Ground truth half-life = ln(2) / -ln(1 - kappa); return AC1 < 0; VR(q) < 1 for q >= 2. """ rng = np.random.default_rng(seed) p = np.empty(n) p[0] = 0.0 eps = rng.normal(0.0, sigma, n) for t in range(1, n): p[t] = (1.0 - kappa) * p[t - 1] + eps[t] return p def roll_bounce_prices(n: int, spread: float, sigma: float = 0.001, seed: int = 0) -> np.ndarray: """Roll (1984) model: observed log-price = random-walk mid ± spread/2. Ground truth: Cov(r_t, r_{t-1}) = -spread^2/4, implied Roll spread = 2*sqrt(-cov) = spread, and the WHOLE negative AC1 is artifact. """ rng = np.random.default_rng(seed) mid = np.cumsum(rng.normal(0.0, sigma, n)) q = rng.choice([-1.0, 1.0], size=n) return mid + (spread / 2.0) * q def leadlag_pair( n: int, beta: float, lag: int, sigma: float = 0.001, seed: int = 0 ) -> tuple[np.ndarray, np.ndarray]: """Return series (x, y) where x truly leads y by `lag` steps. y_t = beta * x_{t-lag} + noise. Ground truth: cross-corr peaks at `lag` with corr ≈ beta*sd(x)/sd(y); zero at all other lags. """ rng = np.random.default_rng(seed) x = rng.normal(0.0, sigma, n) noise = rng.normal(0.0, sigma, n) y = noise.copy() y[lag:] += beta * x[: n - lag] return x, y def seasonal_returns( n: int, period: int, hot_phase: int, amplitude: float, sigma: float = 0.001, seed: int = 0, ) -> np.ndarray: """Returns with a planted calendar effect: mean = amplitude on one phase. Ground truth: mean(returns | t % period == hot_phase) = amplitude, all other phases 0. """ rng = np.random.default_rng(seed) r = rng.normal(0.0, sigma, n) r[np.arange(n) % period == hot_phase] += amplitude return r def stale_observe( prices: np.ndarray, p_observe: float, seed: int = 0 ) -> tuple[np.ndarray, np.ndarray]: """Simulate an illiquid ticker: each price prints with prob p_observe, otherwise the last print is carried forward (LOCF). Returns (locf_prices, observed_mask). Ground truth: LOCF returns of a random walk gain SPURIOUS positive lag-1 autocorrelation, and a fully observed correlated series appears to LEAD the stale one. """ rng = np.random.default_rng(seed) observed = rng.random(len(prices)) < p_observe observed[0] = True locf = prices.copy() for t in range(1, len(prices)): if not observed[t]: locf[t] = locf[t - 1] return locf, observed def correlated_pair( n: int, rho: float, sigma: float = 0.001, seed: int = 0 ) -> tuple[np.ndarray, np.ndarray]: """Two random-walk log-prices with contemporaneously correlated innovations. Ground truth: corr(r_x, r_y) = rho at lag 0, zero at every nonzero lag — any measured lead-lag after LOCF is pure artifact. """ rng = np.random.default_rng(seed) z1 = rng.normal(0.0, sigma, n) z2 = rng.normal(0.0, sigma, n) rx = z1 ry = rho * z1 + np.sqrt(1.0 - rho**2) * z2 return np.cumsum(rx), np.cumsum(ry)