# ============================================================================= # Project : anomaly-atlas # File : experiments/micro/expE_calendar_scan/benchmark.py # Purpose : Calendar scan with pre-counted budget + permuted-calendar null # Author : Simon-Pierre Boucher # Contact : contact@spboucher.ai # Data src : hfmarketdata.io (sole data source) # Created : 2026-08-12 # Modified : 2026-08-12 # Platform : macOS / Apple Silicon (arm64) # License : All rights reserved (research code) # ============================================================================= """Experiment E — calendar scan (protocol + 8-test budget pre-specified in hypothesis.md). Level 0 throughout. Also measures the H20 intraday artifact profile (taxonomy input, not a hypothesis test).""" from __future__ import annotations import json import sys from datetime import UTC, date, datetime from pathlib import Path import numpy as np REPO_ROOT = Path(__file__).resolve().parents[3] sys.path.insert(0, str(REPO_ROOT / "benchmarks")) sys.path.insert(0, str(REPO_ROOT / "src")) from hardware_manifest import collect_manifest # noqa: E402 from anomaly_atlas.data.hf_client import HFMarketDataClient # noqa: E402 from anomaly_atlas.data.universe import LIQUID_ETF, LIQUID_STOCK, TRAIN # noqa: E402 from anomaly_atlas.stats.multiple_testing import benjamini_hochberg # noqa: E402 from anomaly_atlas.validation.artifacts import edge_spread # noqa: E402 N_PERM, PERM_SEED = 2000, 42 UMICRO_WINDOW = ("2014-01-01", "2016-01-01") # H20 profile, cached from expC/D HALF_HOURS = [f"{h:02d}:{m:02d}" for h, m in [(9, 30), (10, 0), (10, 30), (11, 0), (11, 30), (12, 0), (12, 30), (13, 0), (13, 30), (14, 0), (14, 30), (15, 0), (15, 30)]] def weekday(d: str) -> int: return date(int(d[:4]), int(d[5:7]), int(d[8:10])).weekday() def class_masks(dates: list[str]) -> dict[str, np.ndarray]: """The 8 pre-declared calendar classes as boolean masks over days.""" n = len(dates) wd = np.array([weekday(d) for d in dates]) masks = {name: wd == i for i, name in enumerate(["mon", "tue", "wed", "thu", "fri"])} month = np.array([d[:7] for d in dates]) tom = np.zeros(n, dtype=bool) for i in range(n): if i + 1 < n and month[i + 1] != month[i]: tom[i] = True # last trading day of month (-1) if i > 0 and month[i - 1] != month[i]: for j in range(i, min(i + 3, n)): # first 3 trading days (+1..+3) if month[j] == month[i]: tom[j] = True masks["turn_of_month"] = tom # holidays: a non-weekend gap in the trading calendar pre = np.zeros(n, dtype=bool) post = np.zeros(n, dtype=bool) for i in range(n - 1): d0 = date(int(dates[i][:4]), int(dates[i][5:7]), int(dates[i][8:10])) d1 = date(int(dates[i + 1][:4]), int(dates[i + 1][5:7]), int(dates[i + 1][8:10])) gap_weekdays = np.busday_count(d0.isoformat(), d1.isoformat()) - 1 if gap_weekdays >= 1: pre[i] = True post[i + 1] = True masks["pre_holiday"] = pre masks["post_holiday"] = post return masks def stats_for(returns: np.ndarray, masks: dict[str, np.ndarray]) -> dict[str, float]: mu = returns.mean() return {name: float(returns[m].mean() - mu) if m.sum() >= 20 else float("nan") for name, m in masks.items()} def permutation_test(returns: np.ndarray, years: np.ndarray, masks: dict[str, np.ndarray]) -> tuple[dict, dict]: """Within-year permutation: marginal p per class + family-wise max-stat p.""" rng = np.random.default_rng(PERM_SEED) observed = stats_for(returns, masks) names = list(masks) exceed = dict.fromkeys(names, 0) fw_exceed = dict.fromkeys(names, 0) perm_dist: dict[str, list[float]] = {k: [] for k in names} year_idx = [np.where(years == y)[0] for y in np.unique(years)] for _ in range(N_PERM): perm = returns.copy() for idx in year_idx: perm[idx] = perm[idx][rng.permutation(len(idx))] s = stats_for(perm, masks) max_abs = max(abs(v) for v in s.values() if np.isfinite(v)) for name in names: perm_dist[name].append(s[name]) if np.isfinite(s[name]) and abs(s[name]) >= abs(observed[name]): exceed[name] += 1 if np.isfinite(observed[name]) and max_abs >= abs(observed[name]): fw_exceed[name] += 1 marg = {k: (exceed[k] + 1) / (N_PERM + 1) for k in names} fw = {k: (fw_exceed[k] + 1) / (N_PERM + 1) for k in names} band = {k: [round(float(np.percentile(perm_dist[k], q)) * 1e4, 3) for q in (2.5, 97.5)] for k in names} return {"observed_bp": {k: round(v * 1e4, 3) for k, v in observed.items()}, "perm_band95_bp": band, "p_marginal": {k: round(v, 4) for k, v in marg.items()}, "p_familywise": {k: round(v, 4) for k, v in fw.items()}}, observed def h20_profile(client: HFMarketDataClient) -> dict: """Intraday half-hour profile of |return|, EDGE spread, staleness.""" tickers = [("stock", t) for t in LIQUID_STOCK] + [("etf", t) for t in LIQUID_ETF] buckets = {hh: {"absret": [], "bars": [], "minutes": 0, "present": 0} for hh in HALF_HOURS} s, e = UMICRO_WINDOW for asset, ticker in tickers: bars = client.get_bars(asset, ticker, "1min", "adj_split", s, e) by_day: dict[str, list[dict]] = {} for b in bars: t = b["datetime"][11:16] if "09:30" <= t < "16:00": by_day.setdefault(b["datetime"][:10], []).append(b) n_days = len(by_day) for hh_i, hh in enumerate(HALF_HOURS): hi = HALF_HOURS[hh_i + 1] if hh_i + 1 < len(HALF_HOURS) else "16:00" sel = [b for bs in by_day.values() for b in bs if hh <= b["datetime"][11:16] < hi] closes = np.array([b["close"] for b in sel]) if len(closes) > 100: r = np.abs(np.diff(np.log(closes))) buckets[hh]["absret"].append(float(np.median(r))) buckets[hh]["bars"].append(sel) buckets[hh]["minutes"] += 30 * n_days buckets[hh]["present"] += len(sel) profile = [] for hh in HALF_HOURS: b = buckets[hh] spreads = [] for sel in b["bars"]: o = np.array([x["open"] for x in sel]) h = np.array([x["high"] for x in sel]) lo = np.array([x["low"] for x in sel]) c = np.array([x["close"] for x in sel]) sp = edge_spread(o, h, lo, c) if np.isfinite(sp): spreads.append(sp) profile.append({ "bucket": hh, "median_abs_1min_ret_bp": round(float(np.median(b["absret"])) * 1e4, 3) if b["absret"] else None, "median_edge_spread_bp": round(float(np.median(spreads)) * 1e4, 3) if spreads else None, "staleness": round(1 - b["present"] / b["minutes"], 4) if b["minutes"] else None, }) return {"window": UMICRO_WINDOW, "universe": LIQUID_STOCK + LIQUID_ETF, "profile": profile} def main() -> None: run_utc = datetime.now(UTC) client = HFMarketDataClient() bars = client.get_bars("etf", "SPY", "1day", "adj_splitdiv", TRAIN[0], TRAIN[1]) dates = [b["datetime"][:10] for b in bars] closes = np.array([b["close"] for b in bars]) returns = np.diff(np.log(closes)) dates = dates[1:] # return dates years = np.array([d[:4] for d in dates]) masks = class_masks(dates) table, observed = permutation_test(returns, years, masks) fdr = benjamini_hochberg(np.array([table["p_marginal"][k] for k in masks]), alpha=0.05) table["fdr_marginal"] = {k: bool(r) for k, r in zip(masks, fdr, strict=True)} # sub-period stability, descriptive subs = {} for name, (lo, hi) in {"2000-2007": ("2000", "2007"), "2008-2015": ("2008", "2015")}.items(): sel = (years >= lo) & (years <= hi) subs[name] = {k: round(v * 1e4, 3) for k, v in stats_for(returns[sel], {k: m[sel] for k, m in masks.items()}).items()} results = { "experiment": "expE_calendar_scan", "run_utc": run_utc.isoformat(), "author": "Simon-Pierre Boucher", "contact": "contact@spboucher.ai", "data_source": "hfmarketdata.io", "confidence_level": 0, "protocol": {"instrument": "SPY 1day adj_splitdiv", "train": TRAIN, "budget_tests": 8, "n_perm": N_PERM, "perm_seed": PERM_SEED, "class_counts": {k: int(m.sum()) for k, m in masks.items()}, "n_days": int(len(returns))}, "calendar_tests": table, "subperiods_bp": subs, "h20_intraday_profile": h20_profile(client), "client_stats": vars(client.stats) | {"refreshes": list(client.stats.refreshes)}, "manifest": collect_manifest(), } out_dir = REPO_ROOT / "results" / "expE_calendar_scan" / run_utc.strftime("%Y%m%dT%H%M%SZ") out_dir.mkdir(parents=True) (out_dir / "results.json").write_text(json.dumps(results, indent=2) + "\n") print(f"wrote {out_dir.relative_to(REPO_ROOT)}/results.json") print(json.dumps({k: results["calendar_tests"][k] for k in ("observed_bp", "p_marginal", "p_familywise", "fdr_marginal")}, indent=1)) print("H20 profile:", json.dumps(results["h20_intraday_profile"]["profile"], indent=1)[:800]) if __name__ == "__main__": main()