diff --git a/README.md b/README.md index 8d0fb97..875e70c 100644 --- a/README.md +++ b/README.md @@ -18,19 +18,26 @@ From the directory containing `backend/`, `frontend/`, and `scripts/` (run `cd h ./scripts/dev ``` -Open [http://127.0.0.1:5173](http://127.0.0.1:5173). Ctrl+C stops both processes. +`./scripts/dev` requires a clean `main` checkout with `origin` pointing to +`hypertrial/hyperoptions`. It checks `origin/main` before installing locked +dependencies, fast-forwards a clean checkout, and then starts both services. +If the remote cannot be reached, it warns that the version is unverified and +starts the clean local checkout. A feature branch, local changes, or divergent +history stops startup with a repair message. While running, the UI checks for +a newer remote revision; restart `./scripts/dev` to apply it. Open +[http://127.0.0.1:5173](http://127.0.0.1:5173). Ctrl+C stops both processes. For separate terminals: ```bash cd backend -uv sync --group dev --group research +uv sync --frozen --group dev --group research uv run --group research uvicorn options_api.main:app --reload --no-access-log --host 127.0.0.1 --port 8000 ``` ```bash cd frontend -npm install +npm ci npm run generate:api npm run dev ``` @@ -105,6 +112,11 @@ After the expiry session completes, the separate **Expiry · close-based result* `GET /api/health`, `GET /api/tickers?q=&limit=`, `GET /api/covered-calls/{ticker}?moneyness=`, and `GET /api/cash-secured-puts/{ticker}?moneyness=` serve the chain. Eligible chain rows include a backend-generated `watch_key`; ineligible rows include `watchability_reason`. The watchlist uses `POST /api/watchlist` with that key, `GET /api/watchlist`, `DELETE /api/watchlist/{id}`, and `POST /api/watchlist/refresh`; job progress is available from `GET /api/jobs/{id}`. The manual `/api/research/*` routes have been removed. Tickers must match `^[A-Z]{1,5}$` and belong to the Nasdaq-listed universe. The universe fails closed: if it cannot be loaded, ticker search and chain routes return 503. Unknown symbols return 404. Host validation admits loopback hosts only, including IPv6 `::1`. CORS allows the local Vite origin. API writes require that origin and JSON, and write bodies are capped at 16 KiB. +`GET /api/version?frontend_sha=` reports the frozen running revision, branch, +remote revision or offline status, check time, and optional frontend/backend +revision match. Its remote check is cached for 60 seconds and never changes +the running checkout. + The checked-in `frontend/openapi.json` is generated from FastAPI. `npm run generate:api` regenerates TypeScript types and Zod page validators under `frontend/src/generated/`. `npm run check:api` fails if that output drifts. Nasdaq fetches use a bounded policy: the required option chain has two attempts, a 15s read timeout, and a 22s overall deadline. Optional stock info and history use a shorter 8s deadline so a degraded page does not wait through two full chain timeouts. HTTP 429 is not retried; numeric `Retry-After` is forwarded. The ticker screener uses the same 15s read timeout as the chain. diff --git a/backend/pyproject.toml b/backend/pyproject.toml index 37f1820..881436a 100644 --- a/backend/pyproject.toml +++ b/backend/pyproject.toml @@ -4,6 +4,7 @@ version = "0.1.0" description = "Local Nasdaq covered-call and cash-secured-put options workstation" requires-python = ">=3.12,<3.13" dependencies = [ + "arch>=8,<9", "duckdb>=1.2", "exchange-calendars>=4.11,<5", "fastapi>=0.115", @@ -27,9 +28,7 @@ dev = [ "httpx2>=2.9", "ruff>=0.15", ] -research = [ - "arch>=8,<9", -] +research = [] [tool.ruff] target-version = "py312" diff --git a/backend/scripts/evaluate_intraday_prospective.py b/backend/scripts/evaluate_intraday_prospective.py index ed5196a..13a1cfe 100644 --- a/backend/scripts/evaluate_intraday_prospective.py +++ b/backend/scripts/evaluate_intraday_prospective.py @@ -9,343 +9,12 @@ import argparse import json -from collections import Counter, defaultdict -from datetime import UTC, date, datetime, time, timedelta +from datetime import date from pathlib import Path -from statistics import mean -from zoneinfo import ZoneInfo -import numpy as np - -from options_api.market_calendar import _calendar, session_close from stocksweeper.config import load_settings +from stocksweeper.forecast.intraday_evidence import evaluate from stocksweeper.forecast.ledger import ForecastLedger -from stocksweeper.forecast.physical_evaluation import _quantile, _score - -_NY = ZoneInfo("America/New_York") -_WINDOWS = ("10:00", "13:00", "15:30") -_MODELS = ("dated_close", "quote_reanchored_comparator", "intraday_shadow") -_CHALLENGERS = _MODELS[1:] - - -def _cell_key(row: dict[str, object]) -> tuple[object, ...]: - return ( - row["contract_key"], - row["root"], - row["side"], - row["expiration"], - row["expiry_session"], - row["strike_exact"], - row["terms_note"], - row["contract_since"], - row["issued_at"].astimezone(_NY).date(), - row["snapshot_window"], - ) - - -def _vintage(row: dict[str, object]) -> tuple[object, ...]: - # A refetch can change retrieval time without changing the model input. - return row["input_session"], row["data_hash"] - - -def _order(row: dict[str, object]) -> tuple[datetime, str]: - return row["issued_at"], row["idempotency_key"] - - -def _calendar_stratum(day: date) -> str: - calendar = _calendar() - if not calendar.is_session(day.isoformat()): - return "holiday_or_non_session" - session = calendar.date_to_session(day.isoformat(), direction="none") - duration = calendar.session_close(session) - calendar.session_open(session) - return "early_close" if duration < timedelta(hours=6, minutes=30) else "regular_session" - - -def _summary(cells: list[dict[str, object]], *, bootstrap: bool = False) -> dict[str, object]: - available: Counter[str] = Counter() - issued: Counter[str] = Counter() - reasons: Counter[str] = Counter() - latencies: dict[str, list[float]] = defaultdict(list) - quote_ages: list[float] = [] - target_lags: list[float] = [] - units: dict[tuple[object, ...], list[dict[str, tuple[float, float]]]] = defaultdict(list) - for cell in cells: - attempts = cell["attempts"] - for model, attempt in attempts.items(): - if attempt is None: - continue - issued[model] += 1 - if attempt["status"] == "available": - available[model] += 1 - latency = attempt["lookup_ms"] - if ( - model in _CHALLENGERS - and latency is not None - and np.isfinite(latency) - and latency >= 0 - ): - latencies[model].append(float(latency)) - reason = cell["reason"] - if reason is not None: - reasons[reason] += 1 - if cell["scores"] is not None: - units[(cell["ticker"], cell["day"], cell["window"], cell["expiry_session"])].append( - cell["scores"] - ) - if cell["quote_age_ms"] is not None: - quote_ages.append(cell["quote_age_ms"]) - if cell["target_lag_ms"] is not None: - target_lags.append(cell["target_lag_ms"]) - unit_scores = [ - { - model: ( - mean(contract[model][0] for contract in contracts), - mean(contract[model][1] for contract in contracts), - ) - for model in _MODELS - } - for contracts in units.values() - ] - means = { - metric: { - model: mean(score[model][index] for score in unit_scores) if unit_scores else None - for model in _MODELS - } - for index, metric in enumerate(("brier", "log_loss")) - } - deltas = { - metric: { - reference: ( - means[metric]["intraday_shadow"] - means[metric][reference] if unit_scores else None - ) - for reference in _MODELS[:2] - } - for metric in ("brier", "log_loss") - } - intervals = None - if bootstrap and len({key[1] for key in units}) >= 2: - by_day: dict[date, list[dict[str, tuple[float, float]]]] = defaultdict(list) - for key, contracts in units.items(): - by_day[key[1]].append( - { - model: ( - mean(contract[model][0] for contract in contracts), - mean(contract[model][1] for contract in contracts), - ) - for model in _MODELS - } - ) - days = sorted(by_day) - counts = np.asarray([len(by_day[day]) for day in days]) - sampled = np.random.default_rng(20260927).integers(0, len(days), size=(2000, len(days))) - sampled_counts = counts[sampled].sum(axis=1) - intervals = {} - for index, metric in enumerate(("brier", "log_loss")): - intervals[metric] = {} - for reference in _MODELS[:2]: - sums = np.asarray( - [ - sum( - score["intraday_shadow"][index] - score[reference][index] - for score in by_day[day] - ) - for day in days - ] - ) - values = sums[sampled].sum(axis=1) / sampled_counts - intervals[metric][reference] = [ - _quantile(values.tolist(), 0.025), - _quantile(values.tolist(), 0.975), - ] - return { - "recorded_contract_windows": len(cells), - "recorded_contracts": len({cell["contract_key"] for cell in cells}), - "recorded_tickers": len({cell["ticker"] for cell in cells}), - "recorded_dates": len({cell["day"] for cell in cells}), - "forecast_present_cells": {model: issued[model] for model in _MODELS}, - "forecast_available_cells": {model: available[model] for model in _MODELS}, - "triple_available_contract_windows": sum( - all( - attempt is not None and attempt["status"] == "available" - for attempt in cell["attempts"].values() - ) - for cell in cells - ), - "scored_contract_windows": sum(cell["scores"] is not None for cell in cells), - "scored_ticker_date_window_expiry_units": len(units), - "scored_tickers": len({key[0] for key in units}), - "scored_date_blocks": len({key[1] for key in units}), - "mean": means, - "paired_shadow_minus_reference": deltas, - "paired_calendar_date_bootstrap_95": intervals, - "rejection_reasons": dict(sorted(reasons.items())), - "latency_ms": { - model: { - "p50": _quantile(latencies[model], 0.5), - "p95": _quantile(latencies[model], 0.95), - } - for model in _CHALLENGERS - }, - "quote_age_ms": {"p50": _quantile(quote_ages, 0.5), "p95": _quantile(quote_ages, 0.95)}, - "target_to_issue_lag_ms": { - "p50": _quantile(target_lags, 0.5), - "p95": _quantile(target_lags, 0.95), - }, - } - - -def evaluate(rows: list[dict[str, object]], *, as_of: datetime | None = None) -> dict[str, object]: - """Use first prospective attempt per contract/window; never select by outcome.""" - now = as_of or datetime.now(UTC) - if now.tzinfo is None: - raise ValueError("as_of must be timezone-aware") - primary: dict[tuple[object, ...], list[dict[str, object]]] = defaultdict(list) - candidates: dict[tuple[object, ...], dict[str, list[dict[str, object]]]] = defaultdict( - lambda: defaultdict(list) - ) - for row in rows: - if row["provenance"] != "as_issued": - continue - if row["method"] in _CHALLENGERS and row["snapshot_window"] in _WINDOWS: - candidates[_cell_key(row)][row["method"]].append(row) - elif row["price_basis"] == "completed_close" and row["snapshot_window"] is None: - key = _cell_key({**row, "snapshot_window": None})[:8] - primary[key].append(row) - cells: list[dict[str, object]] = [] - duplicate_attempts = 0 - for key, attempts in sorted(candidates.items(), key=lambda pair: tuple(map(str, pair[0]))): - contract = key[:8] - day, window = key[-2:] - target = datetime.combine(day, time.fromisoformat(window), _NY).astimezone(UTC) - selected = { - model: min(attempts[model], key=_order) if attempts[model] else None - for model in _CHALLENGERS - } - duplicate_attempts += sum(max(0, len(attempts[model]) - 1) for model in _CHALLENGERS) - first = min((row for row in selected.values() if row is not None), key=_order) - last = max((row for row in selected.values() if row is not None), key=_order) - # Primary issues are idempotent across unchanged data; a window refresh - # may retain the earlier as-issued close forecast rather than insert one. - same_contract = [row for row in primary[contract] if row["issued_at"] <= first["issued_at"]] - same_vintage = [row for row in same_contract if _vintage(row) == _vintage(first)] - baseline = max(same_vintage, key=_order) if same_vintage else None - trio = {"dated_close": baseline, **selected} - reason = None - if len(selected) != 2 or any(row is None for row in selected.values()): - reason = "missing_challenger_attempt" - elif selected[_CHALLENGERS[0]]["issued_at"] != selected[_CHALLENGERS[1]]["issued_at"]: - reason = "challenger_capture_time_mismatch" - elif _vintage(selected[_CHALLENGERS[0]]) != _vintage(selected[_CHALLENGERS[1]]): - reason = "challenger_input_vintage_mismatch" - elif baseline is None: - reason = "baseline_input_vintage_mismatch" if same_contract else "baseline_not_issued" - elif any(row["status"] != "available" for row in trio.values()): - reason = next( - row["unavailable_reason"] or f"{model}_unavailable" - for model, row in trio.items() - if row["status"] != "available" - ) - elif any( - row["itm_probability"] is None or not 0 <= row["itm_probability"] <= 1 - for row in trio.values() - ): - reason = "invalid_probability" - elif ( - selected[_CHALLENGERS[0]]["quote_digest"] is None - or selected[_CHALLENGERS[0]]["quote_digest"] - != selected[_CHALLENGERS[1]]["quote_digest"] - ): - reason = "quote_vintage_missing_or_mismatch" - elif any(row["price_basis"] != "underlying_quote" for row in selected.values()): - reason = "quote_price_basis_mismatch" - elif ( - _calendar_stratum(day) == "holiday_or_non_session" - or target >= session_close(day) - or any( - not target <= row["issued_at"] < target + timedelta(minutes=5) - for row in selected.values() - ) - ): - reason = "snapshot_outside_regular_session" - elif any(row["label_status"] != "valid" for row in trio.values()): - reason = next( - row["label_reason"] or row["label_status"] or "label_missing" - for row in trio.values() - if row["label_status"] != "valid" - ) - elif ( - None in {row["observed_itm"] for row in trio.values()} - or len({row["observed_itm"] for row in trio.values()}) != 1 - ): - reason = "conflicting_exact_labels" - elif any(row["issued_at"] >= session_close(row["expiry_session"]) for row in trio.values()): - reason = "issued_after_expiry_close" - elif ( - first["expiry_session"] != last["expiry_session"] - or session_close(first["expiry_session"]) > now - or any( - row["label_checked_at"] is None - or row["label_checked_at"] < session_close(row["expiry_session"]) - or row["label_checked_at"] <= row["issued_at"] - or row["label_checked_at"] > now - for row in trio.values() - ) - ): - reason = "label_not_mature_at_scoring" - scores = None - if reason is None: - observed = bool(first["observed_itm"]) - scores = { - model: _score(float(row["itm_probability"]), observed) - for model, row in trio.items() - } - quote_time = first["quote_time"] - cells.append( - { - "contract_key": first["contract_key"], - "ticker": first["ticker"], - "day": day, - "window": window, - "expiry_session": first["expiry_session"], - "calendar": _calendar_stratum(day), - "expiry": "same_day" if first["expiry_session"] == day else "future", - "attempts": trio, - "reason": reason, - "scores": scores, - "quote_age_ms": ( - (first["issued_at"] - quote_time).total_seconds() * 1000 - if quote_time is not None and first["issued_at"] >= quote_time - else None - ), - "target_lag_ms": ( - (first["issued_at"] - target).total_seconds() * 1000 - if first["issued_at"] >= target - else None - ), - } - ) - return { - "source": "append-only forecast ledger", - "provenance": "as_issued", - "scope": "recorded watchlist intraday contract windows only", - "denominator_limit": "Missed windows are absent from the issuance ledger.", - "metric": "binary ITM at exact expiry-session close; equality is ATM", - "selection": "First challenger attempt, preceding matching close forecast", - "duplicate_challenger_attempts_excluded": duplicate_attempts, - "overall": _summary(cells, bootstrap=True), - "by_window": { - window: _summary([cell for cell in cells if cell["window"] == window]) - for window in _WINDOWS - }, - "by_expiry": { - expiry: _summary([cell for cell in cells if cell["expiry"] == expiry]) - for expiry in ("same_day", "future") - }, - "by_calendar": { - calendar: _summary([cell for cell in cells if cell["calendar"] == calendar]) - for calendar in ("regular_session", "early_close", "holiday_or_non_session") - }, - } def main() -> None: diff --git a/backend/scripts/evaluate_predictive.py b/backend/scripts/evaluate_predictive.py index bb04ad1..4d72b6d 100644 --- a/backend/scripts/evaluate_predictive.py +++ b/backend/scripts/evaluate_predictive.py @@ -16,17 +16,16 @@ import argparse import json from datetime import UTC, date, datetime, time, timedelta -from decimal import Decimal from math import exp, sqrt from pathlib import Path from time import perf_counter import polars as pl -from options_api.market_calendar import session_close from stocksweeper.config import load_settings from stocksweeper.forecast.audit import read_audit_cohort from stocksweeper.forecast.calendar import SessionCalendar +from stocksweeper.forecast.evidence_reports import ledger_contest, save_replay_report from stocksweeper.forecast.ledger import ForecastLedger from stocksweeper.forecast.market import ForecastPriceStore, YahooForecastProvider from stocksweeper.forecast.physical_contest import PhysicalShadowForecaster @@ -84,6 +83,8 @@ def _replay_cohort( shadows[member.ticker] = PhysicalShadowForecaster(forecaster) for band, horizons in _BANDS.items(): rows: list[ContestRow] = [] + band_skipped: dict[str, int] = {} + scheduled_units = baseline_available_units = 0 for ticker, frame in frames.items(): closes = dict(frame.select("ts", "close").iter_rows()) split_dates = { @@ -93,22 +94,31 @@ def _replay_cohort( for origin in origins: when = datetime.combine(origin, time(23), tzinfo=UTC) for horizon in horizons: + scheduled_units += 1 expiry = calendar.offset(origin, horizon) results = shadow.forecast_candidates(ticker, when, expiry) baseline = results["lognormal_ewma"].distribution if baseline is None: reason = results["lognormal_ewma"].reason or "baseline_unavailable" skipped[reason] = skipped.get(reason, 0) + 1 + band_skipped[reason] = band_skipped.get(reason, 0) + 1 continue + baseline_available_units += 1 assert baseline.spot is not None and baseline.daily_volatility is not None outcome = closes.get(expiry) if outcome is None: skipped["maturity_close_missing"] = ( skipped.get("maturity_close_missing", 0) + 1 ) + band_skipped["maturity_close_missing"] = ( + band_skipped.get("maturity_close_missing", 0) + 1 + ) elif any(origin < day <= expiry for day in split_dates): outcome = None skipped["split_affected_label"] = skipped.get("split_affected_label", 0) + 1 + band_skipped["split_affected_label"] = ( + band_skipped.get("split_affected_label", 0) + 1 + ) regime = ( "low" if baseline.daily_volatility < 0.02 @@ -171,6 +181,9 @@ def _replay_cohort( reports[band] = evaluate_band( rows, candidate, band, holdout_start=cutoff, period=period, calendar=calendar ) + reports[band]["replay_scheduled_units"] = scheduled_units + reports[band]["replay_baseline_available_units"] = baseline_available_units + reports[band]["replay_rejection_reasons"] = band_skipped return { "source": "immutable current-vintage Yahoo audit snapshots", "provenance": "immutable_replay", @@ -190,114 +203,6 @@ def _replay_cohort( } -def _ledger_band_rows( - ledger: ForecastLedger, - calendar: SessionCalendar, - provenance: str, - candidate: str, - holdout_start: date | None, - period: str, - band: str, -) -> list[ContestRow]: - contest_rows: list[ContestRow] = [] - horizons = _BANDS[band] - issues = ledger.iter_evaluation_rows( - provenance=provenance, - since=holdout_start if period == "holdout" else None, - expiry_before=holdout_start if period == "screen" else None, - methods=("lognormal_ewma", candidate), - horizon_range=(horizons.start, horizons.stop - 1), - with_crps=True, - ) - for item in issues: - origin = item["input_session"] - method = item["method"] - if origin is None or method is None: - continue - label_valid = ( - item["label_status"] == "valid" - and item["label_checked_at"] > item["issued_at"] - and item["issued_at"] < session_close(item["expiry_session"]) - ) - probability = item["itm_probability"] if item["status"] == "available" else None - observed = bool(item["observed_itm"]) if label_valid else None - score = item["crps"] if label_valid else None - strike = Decimal(item["strike_exact"]) - spot = Decimal(item["spot_exact"]) if item["spot_exact"] else None - relative = abs(float(strike / spot - 1)) if spot else None - moneyness = ( - "unknown" - if relative is None - else "near_atm" - if relative <= 0.05 - else "moderate" - if relative <= 0.15 - else "tail" - ) - contest_rows.append( - ContestRow( - ticker=item["ticker"], - origin=origin, - expiry_session=item["expiry_session"], - horizon=calendar.horizon(origin, item["expiration"]), - strike=item["strike_exact"], - side=item["side"], - method=method, - probability=probability, - observed_itm=observed, - provenance=provenance, - moneyness=moneyness, - volatility_regime=item["volatility_regime"] or "unknown", - event_status=item["known_event_status"] or "unknown", - reason=item["unavailable_reason"] or item["label_reason"], - input_vintage=item["data_hash"], - issued_at=item["issued_at"], - issuance_key=item["idempotency_key"], - contract_id=item["contract_key"], - crps=score, - prepare_ms=item["prepare_ms"], - lookup_ms=item["lookup_ms"], - ) - ) - return contest_rows - - -def _ledger_contest( - ledger: ForecastLedger, - calendar: SessionCalendar, - provenance: str, - candidate: str, - holdout_start: date | None, - period: str, -) -> dict[str, object]: - return { - "source": "append-only forecast ledger", - "provenance": provenance, - "skipped_attempts": ledger.evaluation_skipped_attempts(provenance), - "prospective_panel_coverage": ( - ledger.panel_coverage( - since=holdout_start if period == "holdout" else None, - before=holdout_start if period == "screen" else None, - ) - if provenance == "as_issued" - else None - ), - "bands": { - band: evaluate_band( - _ledger_band_rows( - ledger, calendar, provenance, candidate, holdout_start, period, band - ), - candidate, - band, - holdout_start=holdout_start, - period=period, - calendar=calendar, - ) - for band in ("1", "2-5", "6-25") - }, - } - - def main() -> None: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("ticker", nargs="?", help="Nasdaq ticker with prepared Yahoo Close history") @@ -316,6 +221,10 @@ def main() -> None: ) parser.add_argument("--max-origins", type=int, default=20) parser.add_argument("--max-tickers", type=int, default=50) + parser.add_argument( + "--save-evidence", action="store_true", + help="atomically save a replay report for the local app comparison view", + ) parser.add_argument( "--candidate", choices=("empirical_scaled", "student_t_ewma", "gjr_garch_t") ) @@ -342,14 +251,18 @@ def main() -> None: ) except ValueError as exc: parser.error(str(exc)) + if args.save_evidence: + save_replay_report(data_dir, args.candidate, report) print(json.dumps(report, sort_keys=True, indent=2, allow_nan=False)) return + if args.save_evidence: + parser.error("--save-evidence requires --replay-cohort") if args.ledger_contest: if args.ticker or args.refresh or not args.candidate: parser.error("--ledger-contest needs --candidate and no ticker or --refresh") if args.period != "all" and args.holdout_start is None: parser.error("--period screen/holdout needs --holdout-start") - report = _ledger_contest( + report = ledger_contest( ForecastLedger(data_dir), calendar, args.provenance, diff --git a/backend/src/options_api/live_quant.py b/backend/src/options_api/live_quant.py index 90236f5..711e74d 100644 --- a/backend/src/options_api/live_quant.py +++ b/backend/src/options_api/live_quant.py @@ -14,20 +14,36 @@ years_until_expiry_close, ) from options_api.hypothetical_risk import compute_hypothetical_risk +from options_api.intraday_shadow import _VERSION as INTRADAY_VERSION from options_api.market_calendar import session_close from options_api.market_watch import MarketWatchOdds from options_api.models import ( HypotheticalRiskView, MarketOddsView, MarketSource, + PhysicalModel, PredictiveOddsView, Side, ) from options_api.money import to_pct_tenths from options_api.outcomes import TERMS_NOTE from options_api.predictive_watch import PredictiveWatchOdds +from options_api.physical_shadow_capture import PhysicalShadowCapture +from stocksweeper.forecast.calibration import horizon_band from stocksweeper.forecast.ledger import ForecastIssuance from stocksweeper.forecast.predictive import PredictiveDistribution +from stocksweeper.forecast.predictive import BASELINE_VERSION +from stocksweeper.forecast.physical_contest import ( + EMPIRICAL_SHADOW_VERSION, GJR_VERSION, STUDENT_VERSION, +) + +_MODEL_VERSIONS = { + "lognormal_ewma": BASELINE_VERSION, + "empirical_scaled": EMPIRICAL_SHADOW_VERSION, + "student_t_ewma": STUDENT_VERSION, + "gjr_garch_t": GJR_VERSION, + "intraday_shadow": INTRADAY_VERSION, +} @dataclass(frozen=True) @@ -36,6 +52,8 @@ class LiveQuant: predictive: PredictiveOddsView risk: HypotheticalRiskView greeks: ContractGreeks + physical_models: tuple[PredictiveOddsView, ...] = () + market_models: tuple[MarketOddsView, ...] = () greeks_rate_pct_tenths: int | None = None greeks_rate_as_of_session: date | None = None last_available_market: MarketOddsView | None = None @@ -122,6 +140,8 @@ def quant_for_contract( displayed_chain_fetched_at: datetime | None = None, displayed_chain_source: MarketSource | None = None, terms_note: str = TERMS_NOTE, + physical_shadow: PhysicalShadowCapture | None = None, + forecast_model: PhysicalModel = "lognormal_ewma", ) -> LiveQuant: expiry_text = expiry.isoformat() market = market_odds.lookup(ticker, side, expiry_text, strike, root) @@ -147,6 +167,94 @@ def quant_for_contract( contract_since=contract_since, terms_note=terms_note, ) + physical_models: tuple[PredictiveOddsView, ...] = () + selected_distribution = distribution + if physical_shadow is not None: + model_views = [] + quote_for_model = market_odds.underlying_quote(ticker) + for method in ( + "lognormal_ewma", "empirical_scaled", "student_t_ewma", "gjr_garch_t", + "intraday_shadow", + ): + if predictive.status == "unavailable": + view = PredictiveOddsView( + method=method, status="unavailable", reason=predictive.reason, + as_of_session=distribution.as_of, + expiry_session=distribution.expiry_session, + model_version=_MODEL_VERSIONS[method], + ) + else: + candidate = physical_shadow.candidate( + distribution, method, expiry, contract_since or distribution.as_of, + quote_for_model if method == "intraday_shadow" else None, + ) + choice = candidate.distribution + if choice is None: + view = PredictiveOddsView( + method=method, + status=( + "pending" if candidate.reason == "candidate_not_prepared" + else "unavailable" + ), + reason=candidate.reason, + as_of_session=distribution.as_of, + expiry_session=distribution.expiry_session, + data_hash=distribution.data_hash, + model_version=_MODEL_VERSIONS[method], + ) + elif ( + choice.as_of != distribution.as_of + or choice.expiry_session != distribution.expiry_session + or (method != "intraday_shadow" and choice.data_hash != distribution.data_hash) + ): + view = PredictiveOddsView( + method=method, status="unavailable", reason="input_vintage_changed", + model_version=_MODEL_VERSIONS[method], + ) + else: + view = predictive_odds.view_for_distribution( + choice, side, strike, + price_basis=( + "validated_underlying_quote" + if method == "intraday_shadow" else "completed_close" + ), + price_as_of=( + quote_for_model.quote_time + if method == "intraday_shadow" and quote_for_model is not None + else None + ), + ) + if method == forecast_model and view.status == "available": + selected_distribution = choice + if view.evidence_key is None and ( + band := horizon_band(distribution.horizon_sessions) + ): + view = view.model_copy(update={ + "evidence_key": f"{method}:{band}" + }) + model_views.append(view) + physical_models = tuple(model_views) + predictive = next(view for view in physical_models if view.method == forecast_model) + market = market.model_copy(update={"method": "regimelib"}) + market_report = market_odds.curve_shadow_report(ticker) if physical_shadow is not None else None + market = market.model_copy(update={"model_evidence": { + "held_out_inside": ( + market_report.get("benchmark_held_out_inside") if market_report else None + ), + "held_out_count": ( + market_report.get("benchmark_held_out_predicted") if market_report else None + ), + "paired_held_out_count": ( + market_report.get("paired_held_out_count") if market_report else None + ), + "fit_ms": market_report.get("benchmark_ms") if market_report else None, + "one_tick_stable": None, + "bid_ask_fit": None, + }}) + market_models = ( + (market, market_odds.lookup_curve(ticker, side, expiry_text, strike, root)) + if physical_shadow is not None else (market,) + ) last_good = ( market_odds.lookup_last_good(ticker, side, expiry_text, strike, root) if watched and market.status != "available" @@ -165,8 +273,12 @@ def quant_for_contract( return LiveQuant( market, predictive, - HypotheticalRiskView(reason="Coherent entry quotes are unavailable"), + HypotheticalRiskView( + reason="Coherent entry quotes are unavailable", forecast_method=forecast_model + ), empty_greeks(), + physical_models=physical_models, + market_models=market_models, last_available_market=last_good, issuance=issuance, ) @@ -188,8 +300,8 @@ def quant_for_contract( ) entry_spot = quote.stock_ask if side == "call" else quote.spot coherent_window = ( - distribution.as_of <= quote.session_date <= distribution.expiry_session - and quote.valuation_time < session_close(distribution.expiry_session) + selected_distribution.as_of <= quote.session_date <= selected_distribution.expiry_session + and quote.valuation_time < session_close(selected_distribution.expiry_session) ) if entry_spot is None: risk = HypotheticalRiskView(reason="A stock purchase quote is unavailable") @@ -199,7 +311,7 @@ def quant_for_contract( risk = HypotheticalRiskView(reason="Forecast and entry quote dates do not align") else: risk = compute_hypothetical_risk( - distribution, + selected_distribution, side=side, strike=strike, spot=entry_spot, @@ -208,11 +320,16 @@ def quant_for_contract( quote_session=quote.session_date, reanchor=False, ) + if forecast_model == "intraday_shadow" and risk.status == "available": + risk = risk.model_copy(update={"forecast_price_basis": "intraday_quote"}) + risk = risk.model_copy(update={"forecast_method": forecast_model}) return LiveQuant( market, predictive, risk, greeks, + physical_models=physical_models, + market_models=market_models, greeks_rate_pct_tenths=( to_pct_tenths(quote.rate * 100) if greeks.source is not None else None ), diff --git a/backend/src/options_api/main.py b/backend/src/options_api/main.py index 6e3caeb..a41fd3f 100644 --- a/backend/src/options_api/main.py +++ b/backend/src/options_api/main.py @@ -32,6 +32,7 @@ HypotheticalRiskView, MarketOddsView, Moneyness, + PhysicalModel, PredictiveOddsView, TickerSearchResponse, normalize_ticker, @@ -43,6 +44,7 @@ from options_api.prospective_panel import ProspectivePanel from options_api.service import OptionChainService from options_api.universe import TickerUniverse +from options_api.version import router as version_router from options_api.watchlist import WatchlistService, router as watchlist_router from stocksweeper.config import Settings, load_settings from stocksweeper.forecast.labels import collect_matured_labels @@ -226,6 +228,7 @@ async def _load_page( ticker: str, load: Callable[..., Awaitable[CoveredCallPage | CashSecuredPutPage]], moneyness: Moneyness | None, + forecast_model: PhysicalModel, ) -> CoveredCallPage | CashSecuredPutPage: _check_origin(request) normalized = await _known_ticker(request, ticker) @@ -264,6 +267,22 @@ async def _load_page( contract.predictive_odds = PredictiveOddsView( status="unavailable", reason="contract_terms_ambiguous" ) + contract.physical_models = [ + PredictiveOddsView( + method=method, status="unavailable", reason="contract_terms_ambiguous" + ) + for method in ( + "lognormal_ewma", "empirical_scaled", "student_t_ewma", + "gjr_garch_t", "intraday_shadow", + ) + ] + contract.market_models = [ + MarketOddsView( + method=method, status="unavailable", + reason="Contract terms cannot be verified", + ) + for method in ("regimelib", "constrained_call_curve") + ] contract.hypothetical_risk = HypotheticalRiskView( reason="Contract terms cannot be verified" ) @@ -281,6 +300,8 @@ async def _load_page( strike=Decimal(contract.strike_exact), displayed_chain_fetched_at=page.chain_fetched_at, displayed_chain_source=page.chain_source, + physical_shadow=request.app.state.physical_shadow, + forecast_model=forecast_model, ) if result.issuance is not None: issuances.append(result.issuance) @@ -292,6 +313,18 @@ async def _load_page( ) ) contract.predictive_odds = result.predictive + contract.physical_models = list(result.physical_models) + contract.market_models = ( + list(result.market_models) + if market_snapshot_matches + else [ + MarketOddsView( + method=method, status="pending", + reason="Refreshing odds for displayed quotes", + ) + for method in ("regimelib", "constrained_call_curve") + ] + ) contract.hypothetical_risk = result.risk contract.greeks_rate_pct_tenths = result.greeks_rate_pct_tenths contract.greeks_rate_as_of_session = result.greeks_rate_as_of_session @@ -310,7 +343,10 @@ async def _load_page( LOG.exception("forecast issuance ledger unavailable for %s", normalized) for expiration in page.expirations: for contract in expiration.contracts: - if contract.predictive_odds.status == "available": + if contract.predictive_odds.status in {"available", "pending"} or any( + model.status in {"available", "pending"} + for model in contract.physical_models + ): contract.predictive_odds = contract.predictive_odds.model_copy( update={ "status": "unavailable", @@ -323,8 +359,20 @@ async def _load_page( contract.hypothetical_risk = HypotheticalRiskView( reason="Forecast evidence unavailable" ) + contract.physical_models = [ + model.model_copy( + update={ + "status": "unavailable", + "reason": "Forecast evidence unavailable", + "itm_pct_tenths": None, "otm_pct_tenths": None, + "atm_pct_tenths": None, + } + ) + for model in contract.physical_models + ] else: request.app.state.physical_shadow.submit(issuances) + page.model_evidence = predictive.evidence_index() return page except NasdaqError as exc: raise _http_nasdaq_error(exc) from exc @@ -337,8 +385,9 @@ async def get_covered_calls( request: Request, ticker: Annotated[str, Path(min_length=1, max_length=8)], moneyness: Annotated[Moneyness | None, Query()] = None, + forecast_model: Annotated[PhysicalModel, Query()] = "lognormal_ewma", ) -> CoveredCallPage: - return await _load_page(request, ticker, load_covered_calls, moneyness) + return await _load_page(request, ticker, load_covered_calls, moneyness, forecast_model) @router.get("/api/cash-secured-puts/{ticker}", response_model=CashSecuredPutPage) @@ -346,8 +395,9 @@ async def get_cash_secured_puts( request: Request, ticker: Annotated[str, Path(min_length=1, max_length=8)], moneyness: Annotated[Moneyness | None, Query()] = None, + forecast_model: Annotated[PhysicalModel, Query()] = "lognormal_ewma", ) -> CashSecuredPutPage: - return await _load_page(request, ticker, load_cash_secured_puts, moneyness) + return await _load_page(request, ticker, load_cash_secured_puts, moneyness, forecast_model) def create_app( @@ -384,7 +434,6 @@ async def lifespan(app: FastAPI) -> AsyncIterator[None]: refresh_enabled=predictive_refresh, ) app.state.promotion_registry = PromotionRegistry(settings.resolved_data_dir()) - app.state.physical_shadow = PhysicalShadowCapture(app.state.predictive_odds) app.state.market_odds = MarketWatchOdds( app.state.service, client, @@ -395,6 +444,9 @@ async def lifespan(app: FastAPI) -> AsyncIterator[None]: for record in app.state.watchlist.store.list() ), ) + app.state.physical_shadow = PhysicalShadowCapture( + app.state.predictive_odds, app.state.market_odds + ) app.state.prospective_panel = ProspectivePanel( settings.resolved_data_dir(), app.state.service, @@ -557,6 +609,7 @@ async def capture_panel() -> None: ) app.add_middleware(LocalHostMiddleware) app.include_router(router) + app.include_router(version_router, dependencies=[Depends(_check_origin)]) app.include_router(watchlist_router, dependencies=[Depends(_check_origin)]) return app diff --git a/backend/src/options_api/market_curve_shadow.py b/backend/src/options_api/market_curve_shadow.py index 0bb420b..7c5e819 100644 --- a/backend/src/options_api/market_curve_shadow.py +++ b/backend/src/options_api/market_curve_shadow.py @@ -6,7 +6,7 @@ import time from collections import Counter, defaultdict from collections.abc import Callable, Sequence -from dataclasses import dataclass +from dataclasses import dataclass, field from datetime import date, datetime from decimal import Decimal @@ -44,6 +44,8 @@ class CurveShadowResult: paired_shadow_inside: int = 0 paired_benchmark_inside: int = 0 held_out_cohort: str = "curve_only" + expiry_reasons: dict[str, str] = field(default_factory=dict) + contract_reasons: dict[tuple[str, Decimal], str] = field(default_factory=dict) def _fit( @@ -148,6 +150,8 @@ def calculate_curve_shadow( return CurveShadowResult({}, 0, 0, 0, {"invalid_spot": 1}) by_expiry: dict[str, list[_Quote]] = defaultdict(list) rejected: Counter[str] = Counter() + expiry_reasons: dict[str, str] = {} + contract_reasons: dict[tuple[str, Decimal], str] = {} for row in rows: if time.monotonic() >= deadline: rejected["shadow_budget_exceeded"] += 1 @@ -160,13 +164,16 @@ def calculate_curve_shadow( rate = rate_for_expiry(expiry) except (ArithmeticError, TypeError, ValueError): rejected["invalid_expiration_or_rate"] += 1 + contract_reasons[(row.expiration, row.strike)] = "invalid_expiration_or_rate" continue if years <= 0 or rate is None or not math.isfinite(rate) or not 0 <= rate <= 0.25: rejected["invalid_expiration_or_rate"] += 1 + contract_reasons[(row.expiration, row.strike)] = "invalid_expiration_or_rate" continue quote = _valid_quote(row, years, rate, spot_float) if quote is None: rejected["invalid_quote_or_terms"] += 1 + contract_reasons[(row.expiration, row.strike)] = "invalid_quote_or_terms" continue by_expiry[row.expiration].append(quote) odds: dict[tuple[str, Decimal], OddsEstimate] = {} @@ -190,13 +197,16 @@ def calculate_curve_shadow( for expiry, quotes in by_expiry.items(): if time.monotonic() >= deadline: rejected["shadow_budget_exceeded"] += len(quotes) - break + expiry_reasons[expiry] = "shadow_budget_exceeded" + continue quotes.sort(key=lambda q: q.strike) if len({q.strike for q in quotes}) != len(quotes): rejected["duplicate_strike"] += len(quotes) + expiry_reasons[expiry] = "duplicate_strike" continue if len(quotes) < 5 or len(quotes) > _MAX_STRIKES: rejected["sparse_or_large_strip"] += len(quotes) + expiry_reasons[expiry] = "sparse_or_large_strip" continue fitted = _fit(quotes, deadline=deadline) # The published benchmark already held these keys out of its fit. @@ -237,6 +247,7 @@ def calculate_curve_shadow( held_shadow[key] = target.bid <= estimate <= target.ask if fitted is None: rejected["infeasible_curve"] += len(quotes) + expiry_reasons[expiry] = "infeasible_curve" continue perturbed_fits: dict[tuple[int, int], np.ndarray | None] = {} for index in range(1, len(quotes) - 1): @@ -309,4 +320,6 @@ def calculate_curve_shadow( paired_shadow_inside=sum(held_shadow[key] for key in paired), paired_benchmark_inside=sum(held_benchmark[key] for key in paired), held_out_cohort=cohort, + expiry_reasons=expiry_reasons, + contract_reasons=contract_reasons, ) diff --git a/backend/src/options_api/market_watch.py b/backend/src/options_api/market_watch.py index a8572c4..5e21283 100644 --- a/backend/src/options_api/market_watch.py +++ b/backend/src/options_api/market_watch.py @@ -249,6 +249,7 @@ def __init__( self._last_good_session: date | None = None self._cache: dict[str, _Snapshot] = {} self._curve_shadow: dict[str, dict[str, object]] = {} + self._curve_results: dict[str, tuple[_Snapshot, CurveShadowResult]] = {} self._tasks: dict[str, asyncio.Task[None]] = {} self._shadow_tasks: set[asyncio.Task[None]] = set() self._pending: dict[str, None] = {} @@ -437,11 +438,13 @@ def _remember(self, ticker: str, snapshot: _Snapshot) -> None: snapshot.refreshed_at = _as_utc(self.clock()) self._cache.pop(ticker, None) self._curve_shadow.pop(ticker, None) + self._curve_results.pop(ticker, None) self._cache[ticker] = snapshot while len(self._cache) > _MAX_CACHED_TICKERS: oldest = next(iter(self._cache)) self._cache.pop(oldest) self._chain_refresh_attempts.pop(oldest, None) + self._curve_results.pop(oldest, None) self._persist_current_watches(ticker, snapshot) def curve_shadow_report(self, ticker: str) -> dict[str, object] | None: @@ -498,6 +501,7 @@ def _record_curve_shadow( "by_expiry_moneyness": bands, } self._curve_shadow[ticker] = report + self._curve_results[ticker] = snapshot, shadow if self._data_path is not None and snapshot.source is not None: with connect(self._data_path) as connection: connection.execute( @@ -572,6 +576,77 @@ def lookup( **common, ) + def lookup_curve( + self, ticker: str, side: str, expiry: str, strike: Decimal, root: str | None = None + ) -> MarketOddsView: + """Read only the result of the bounded background curve fit.""" + common = {"method": "constrained_call_curve", "model_version": _CURVE_VERSION} + if root is not None and root != ticker: + return MarketOddsView( + status="unavailable", reason="Contract terms cannot be verified", **common + ) + snapshot = self._cache.get(ticker) + if snapshot is None or not self._valid(snapshot, _as_utc(self.clock())): + return MarketOddsView( + status="pending", reason="Refreshing option quote snapshot", **common + ) + common.update( + source=snapshot.source, + fetched_at=snapshot.fetched_at, + session_date=snapshot.session_date, + ) + if snapshot.error or expiry in snapshot.reasons: + return MarketOddsView( + status="unavailable", reason=snapshot.error or snapshot.reasons[expiry], **common + ) + if (expiry, strike) not in snapshot.valid_contracts: + return MarketOddsView( + status="unavailable", reason="Contract terms cannot be verified", **common + ) + saved = self._curve_results.get(ticker) + if saved is None or saved[0] is not snapshot: + return MarketOddsView(status="pending", reason="Fitting option quote curve", **common) + result = saved[1] + report = self._curve_shadow.get(ticker) + evidence = ( + { + "held_out_inside": result.held_out_inside, + "held_out_count": result.held_out_count, + "paired_held_out_count": result.paired_held_out_count, + "paired_shadow_inside": result.paired_shadow_inside, + "paired_benchmark_inside": result.paired_benchmark_inside, + "one_tick_stable": None, + "bid_ask_fit": None, + "refresh_ms": report.get("live_refresh_ms") if report else None, + "fit_ms": result.elapsed_ms, + "rejection_reasons": result.rejection_reasons, + } + ) + common["model_evidence"] = evidence + estimate = result.odds.get((expiry, strike)) + call_itm = _probability_tenths(estimate) + if call_itm is None: + reason = ( + estimate.reason if estimate is not None else + result.contract_reasons.get((expiry, strike)) + or result.expiry_reasons.get(expiry) + or next((name for name in ("curve_fit_failed", "invalid_spot") + if result.rejection_reasons.get(name)), None) + or "curve_strike_not_supported" + ) + evidence["one_tick_stable"] = False if reason == "one_tick_unstable" else None + evidence["bid_ask_fit"] = False if reason == "infeasible_curve" else None + return MarketOddsView(status="unavailable", reason=reason, **common) + low, high, support = _quote_support(estimate, side) + itm = call_itm if side == "call" else 1000 - call_itm + evidence["one_tick_stable"] = True + evidence["bid_ask_fit"] = True + return MarketOddsView( + status="available", itm_pct_tenths=itm, otm_pct_tenths=1000 - itm, + bound_low_pct_tenths=low, bound_high_pct_tenths=high, + quote_support_score=support, **common, + ) + def lookup_last_good( self, ticker: str, side: str, expiry: str, strike: Decimal, root: str | None = None ) -> MarketOddsView | None: @@ -795,6 +870,12 @@ async def _run_curve_shadow( self._record_curve_shadow(ticker, snapshot, shadow, benchmark_ms, live_refresh_ms) except Exception: LOG.exception("market curve shadow failed for %s", ticker) + if self._cache.get(ticker) is snapshot: + self._record_curve_shadow( + ticker, snapshot, + CurveShadowResult({}, 0, 0, 0, {"curve_fit_failed": 1}), + benchmark_ms, live_refresh_ms, + ) async def _completed_session_close(self, ticker: str, now: datetime) -> Decimal | None: session = latest_completed_session(now) diff --git a/backend/src/options_api/models.py b/backend/src/options_api/models.py index d58cab8..0aeca26 100644 --- a/backend/src/options_api/models.py +++ b/backend/src/options_api/models.py @@ -14,6 +14,9 @@ Side = Literal["call", "put"] GreeksSource = Literal["bid", "mid"] MarketSource = Literal["nasdaq", "yahoo"] +PhysicalModel = Literal[ + "lognormal_ewma", "empirical_scaled", "student_t_ewma", "gjr_garch_t", "intraday_shadow" +] def normalize_ticker(raw: str) -> str | None: @@ -108,6 +111,7 @@ class TickerSearchResponse(BaseModel): class MarketOddsView(BaseModel): + method: Literal["regimelib", "constrained_call_curve"] | None = None status: Literal["pending", "available", "unavailable"] = "pending" itm_pct_tenths: int | None = None otm_pct_tenths: int | None = None @@ -119,12 +123,14 @@ class MarketOddsView(BaseModel): bound_low_pct_tenths: int | None = None bound_high_pct_tenths: int | None = None quote_support_score: int | None = None + model_evidence: dict[str, object] | None = None class PredictiveValidationEvidence(BaseModel): """Comparable calibration from independent, prospective as-issued forecasts.""" source: Literal["prospective_as_issued"] + option_side: Side model_version: str horizon_band: str moneyness_band: str @@ -138,7 +144,7 @@ class PredictiveOddsView(BaseModel): """Physical expiry-close forecast, distinct from risk-neutral option odds.""" status: Literal["pending", "available", "unavailable"] = "pending" - method: str | None = None + method: PhysicalModel | None = None reason: str | None = None itm_pct_tenths: int | None = None otm_pct_tenths: int | None = None @@ -151,12 +157,15 @@ class PredictiveOddsView(BaseModel): price_basis: Literal["completed_close", "validated_underlying_quote"] | None = None price_as_of: datetime | None = None validation_evidence: PredictiveValidationEvidence | None = None + evidence_key: str | None = None + model_evidence: dict[str, object] | None = None class HypotheticalRiskView(BaseModel): """One-contract hold-to-expiry payoff from a dated, coherent entry quote.""" status: Literal["available", "unavailable"] = "unavailable" + forecast_method: PhysicalModel | None = None reason: str | None = None assumed_spot_cents: int | None = None assumed_bid_cents: int | None = None @@ -209,6 +218,8 @@ class CoveredCallContract(BaseModel): greeks_rate_as_of_session: date | None = None market_odds: MarketOddsView = Field(default_factory=MarketOddsView) predictive_odds: PredictiveOddsView = Field(default_factory=PredictiveOddsView) + physical_models: list[PredictiveOddsView] = Field(default_factory=list) + market_models: list[MarketOddsView] = Field(default_factory=list) hypothetical_risk: HypotheticalRiskView = Field(default_factory=HypotheticalRiskView) @@ -241,6 +252,7 @@ class _ChainPageBase(BaseModel): history_from_cache: bool risk_free_rate_pct_tenths: int | None lows: PeriodLows + model_evidence: dict[str, object] = Field(default_factory=dict) class CoveredCallPage(_ChainPageBase): @@ -285,6 +297,8 @@ class CashSecuredPutContract(BaseModel): greeks_rate_as_of_session: date | None = None market_odds: MarketOddsView = Field(default_factory=MarketOddsView) predictive_odds: PredictiveOddsView = Field(default_factory=PredictiveOddsView) + physical_models: list[PredictiveOddsView] = Field(default_factory=list) + market_models: list[MarketOddsView] = Field(default_factory=list) hypothetical_risk: HypotheticalRiskView = Field(default_factory=HypotheticalRiskView) diff --git a/backend/src/options_api/physical_shadow_capture.py b/backend/src/options_api/physical_shadow_capture.py index 477601f..dc4ccb7 100644 --- a/backend/src/options_api/physical_shadow_capture.py +++ b/backend/src/options_api/physical_shadow_capture.py @@ -1,4 +1,4 @@ -"""Bounded, as-issued capture of research forecasts; never supplies live odds.""" +"""Bounded, as-issued capture and cached results for model comparison.""" from __future__ import annotations @@ -9,58 +9,123 @@ from dataclasses import replace from datetime import UTC, datetime, timedelta from decimal import Decimal +from time import monotonic from options_api.market_calendar import session_close +from options_api.intraday_shadow import forecast_intraday_shadow +from options_api.market_watch import MarketWatchOdds, UnderlyingQuote from options_api.predictive_watch import PredictiveWatchOdds from stocksweeper.forecast.ledger import ForecastIssuance -from stocksweeper.forecast.physical_contest import PhysicalShadowForecaster +from stocksweeper.forecast.physical_contest import PhysicalShadowForecaster, ShadowForecast from stocksweeper.forecast.predictive import PredictiveDistribution LOG = logging.getLogger(__name__) _MAX_BATCH_CONTRACTS = 512 _MAX_PENDING_BATCHES = 4 +_MAX_CACHED_GROUPS = 512 +_RETRY_SECONDS = 30 Entry = tuple[ForecastIssuance, PredictiveDistribution | None] class PhysicalShadowCapture: - def __init__(self, predictive: PredictiveWatchOdds) -> None: + def __init__( + self, predictive: PredictiveWatchOdds, market: MarketWatchOdds | None = None + ) -> None: self.predictive = predictive + self.market = market self.forecaster = PhysicalShadowForecaster(predictive.forecaster) self._semaphore = asyncio.Semaphore(1) self._tasks: set[asyncio.Task[None]] = set() self._fit_tasks: set[asyncio.Task[None]] = set() - self._recent: OrderedDict[str, None] = OrderedDict() + self._recent: OrderedDict[str, tuple[float, str]] = OrderedDict() + self._live: OrderedDict[ + tuple[str, object, object, object, str], + tuple[dict[str, ShadowForecast], UnderlyingQuote | None], + ] = OrderedDict() self._closed = False + def candidate( + self, + base: PredictiveDistribution, + method: str, + expiry: object, + contract_since: object, + quote: UnderlyingQuote | None = None, + ) -> ShadowForecast: + if method == "intraday_shadow" and quote is None: + return ShadowForecast(None, "underlying_quote_unavailable", 0, 0) + if base.status != "available" or base.data_hash is None: + return ShadowForecast(None, base.reason or "completed_close_forecast_unavailable", 0, 0) + if base.horizon_sessions > 25 and method != "lognormal_ewma": + return ShadowForecast(None, "candidate_horizon_unsupported", 0, 0) + if method == base.method and method != "intraday_shadow": + return ShadowForecast(base, None, 0, 0) + key = base.ticker, base.as_of, expiry, contract_since, base.data_hash + cached = self._live.get(key) + if cached is not None: + results, saved_quote = cached + if method != "intraday_shadow" or saved_quote == quote: + return results.get(method, ShadowForecast(None, "candidate_not_prepared", 0, 0)) + return ShadowForecast(None, "candidate_not_prepared", 0, 0) + def submit(self, entries: list[Entry]) -> None: if self._closed or not entries: return + quotes = ( + {issue.ticker: self.market.underlying_quote(issue.ticker) for issue, _ in entries} + if self.market is not None + else {} + ) fingerprint = hashlib.sha256( "|".join( sorted( f"{issue.contract_key}:{issue.input_session}:{issue.data_hash}:" - f"{issue.terms_note}:{issue.contract_since}" + f"{issue.terms_note}:{issue.contract_since}:" + f"{quotes.get(issue.ticker)!r}" for issue, _ in entries ) ).encode() ).hexdigest() - if fingerprint in self._recent: - return - self._recent[fingerprint] = None + now = monotonic() + previous = self._recent.get(fingerprint) + if previous is not None: + attempted_at, state = previous + if state == "running" or (state == "retry" and now - attempted_at < _RETRY_SECONDS): + return + if state == "complete" and all( + not issue.data_hash or ( + issue.ticker, issue.input_session, issue.expiration, + issue.contract_since, issue.data_hash, + ) in self._live + for issue, _ in entries + ): + return + self._recent[fingerprint] = (now, "running") + self._recent.move_to_end(fingerprint) while len(self._recent) > 128: self._recent.popitem(last=False) if len(self._fit_tasks) >= _MAX_PENDING_BATCHES: # Backpressure is cheaper than an unbounded research queue. - task = asyncio.create_task(self._record_capacity_async(entries)) + self._recent[fingerprint] = (now, "retry") + task = asyncio.create_task(self._record_capacity_async(entries, quotes)) self._tasks.add(task) task.add_done_callback(self._tasks.discard) return - task = asyncio.create_task(self._capture(entries)) + task = asyncio.create_task(self._capture(entries, quotes)) self._fit_tasks.add(task) task.add_done_callback(self._fit_tasks.discard) self._tasks.add(task) task.add_done_callback(self._tasks.discard) + task.add_done_callback(lambda done, key=fingerprint: self._capture_done(key, done)) + + def _capture_done(self, fingerprint: str, task: asyncio.Task[bool]) -> None: + try: + complete = not task.cancelled() and task.result() is True + except Exception: + complete = False + self._recent[fingerprint] = (monotonic(), "complete" if complete else "retry") + self._recent.move_to_end(fingerprint) @staticmethod def _capacity_entries(entries: list[Entry]) -> list[Entry]: @@ -88,25 +153,68 @@ def _capacity_entries(entries: list[Entry]) -> list[Entry]: def _record_capacity(self, entries: list[Entry]) -> None: self.predictive.ledger.record_batch(self._capacity_entries(entries)) - async def _record_capacity_async(self, entries: list[Entry]) -> None: + def _cache_capacity( + self, + entries: list[Entry], + quotes: dict[str, UnderlyingQuote | None], + target: dict | None = None, + ) -> None: + target = self._live if target is None else target + for issue, distribution in sorted(entries, key=lambda pair: pair[1] is None): + if not issue.data_hash: + continue + key = ( + issue.ticker, issue.input_session, issue.expiration, + issue.contract_since, issue.data_hash, + ) + if key in target or key in self._live: + continue + rejected = { + method: ShadowForecast(None, "shadow_capacity_exceeded", 0, 0) + for method in ( + "lognormal_ewma", "empirical_scaled", "student_t_ewma", + "gjr_garch_t", "intraday_shadow", + ) + } + if distribution is not None and distribution.method == "lognormal_ewma": + rejected["lognormal_ewma"] = ShadowForecast(distribution, None, 0, 0) + target[key] = rejected, quotes.get(issue.ticker) + + async def _record_capacity_async( + self, entries: list[Entry], quotes: dict[str, UnderlyingQuote | None] + ) -> None: try: await asyncio.to_thread(self._record_capacity, entries) except Exception: LOG.exception("physical shadow capacity recording failed") + else: + self._cache_capacity(entries, quotes) + while len(self._live) > _MAX_CACHED_GROUPS: + self._live.popitem(last=False) - async def _capture(self, entries: list[Entry]) -> None: + async def _capture( + self, entries: list[Entry], quotes: dict[str, UnderlyingQuote | None] + ) -> bool: async with self._semaphore: try: - await asyncio.to_thread(self._capture_sync, entries) + return await asyncio.to_thread(self._capture_sync, entries, quotes) except Exception: LOG.exception("physical forecast shadow capture failed") + return False - def _capture_sync(self, entries: list[Entry]) -> None: + def _capture_sync( + self, entries: list[Entry], quotes: dict[str, UnderlyingQuote | None] | None = None + ) -> bool: + quotes = quotes or {} selected = sorted( entries, key=lambda pair: hashlib.sha256(pair[0].contract_key.encode()).digest() )[:_MAX_BATCH_CONTRACTS] selected_keys = {issue.contract_key for issue, _ in selected} groups: dict[tuple[str, object, object, object], list[ForecastIssuance]] = {} + live_updates: dict[ + tuple[str, object, object, object, str], + tuple[dict[str, ShadowForecast], UnderlyingQuote | None], + ] = {} for issue, _ in selected: key = (issue.ticker, issue.input_session, issue.expiration, issue.contract_since) groups.setdefault(key, []).append(issue) @@ -135,6 +243,29 @@ def _capture_sync(self, entries: list[Entry]) -> None: except Exception: LOG.exception("physical shadow fit failed for %s", ticker) fit_failed = True + if ( + candidates is not None + and (baseline := candidates["lognormal_ewma"].distribution) is not None + and baseline.data_hash == reference.data_hash + and baseline.as_of == reference.input_session + ): + quote = quotes.get(ticker) + if quote is not None: + try: + intraday = forecast_intraday_shadow( + self.predictive.forecaster, baseline, quote + ) + candidates["intraday_shadow"] = ShadowForecast( + intraday.distribution, intraday.reason, 0, 0 + ) + except (OSError, ValueError, OverflowError): + candidates["intraday_shadow"] = ShadowForecast( + None, "intraday_model_failed", 0, 0 + ) + live_updates[ + (ticker, input_session, expiry, contract_since, reference.data_hash) + ] = candidates, quote + group_live: dict[str, ShadowForecast] = {} for issue in contracts: baseline = candidates.get("lognormal_ewma") if candidates is not None else None volatility = ( @@ -210,12 +341,40 @@ def _capture_sync(self, entries: list[Entry]) -> None: distribution, ) ) - recorded.extend( - self._capacity_entries( - [entry for entry in entries if entry[0].contract_key not in selected_keys] - ) - ) + if issue is reference: + group_live[method] = ShadowForecast( + distribution, + None if available else reason, + candidate.prepare_ms if candidate is not None else 0, + candidate.lookup_ms if candidate is not None else 0, + ) + if reference.data_hash and ( + ticker, input_session, expiry, contract_since, reference.data_hash + ) not in live_updates: + group_live["intraday_shadow"] = ShadowForecast( + None, + group_live["lognormal_ewma"].reason, + 0, + 0, + ) + live_updates[ + (ticker, input_session, expiry, contract_since, reference.data_hash) + ] = group_live, quotes.get(ticker) + unselected = [entry for entry in entries if entry[0].contract_key not in selected_keys] + recorded.extend(self._capacity_entries(unselected)) self.predictive.ledger.record_batch(recorded) + self._cache_capacity(unselected, quotes, live_updates) + self._live.update(live_updates) + while len(self._live) > _MAX_CACHED_GROUPS: + self._live.popitem(last=False) + retryable = { + "shadow_fit_failed", "input_provenance_unverified", "shadow_capacity_exceeded" + } + return not any( + forecast.reason in retryable + for forecasts, _ in live_updates.values() + for forecast in forecasts.values() + ) async def close(self) -> None: self._closed = True diff --git a/backend/src/options_api/predictive_watch.py b/backend/src/options_api/predictive_watch.py index b5d0e52..95828c4 100644 --- a/backend/src/options_api/predictive_watch.py +++ b/backend/src/options_api/predictive_watch.py @@ -17,6 +17,7 @@ normalize_ticker, ) from stocksweeper.forecast.calibration import build_calibration, horizon_band, moneyness_band +from stocksweeper.forecast.evidence_reports import build_model_evidence from stocksweeper.forecast.ledger import ForecastLedger from stocksweeper.forecast.predictive import PredictiveDistribution, PredictiveForecaster @@ -54,8 +55,10 @@ def __init__( self._semaphore = asyncio.Semaphore(2) self._refresh_locks: dict[str, asyncio.Lock] = {} self._champion_attempts: dict[tuple[str, date, str], datetime] = {} - self._calibration: dict[tuple[str, str, str, str], dict[str, object]] = {} + self._calibration: dict[tuple[str, str, str, str, Side], dict[str, object]] = {} self._calibration_day: date | None = None + self._model_evidence: dict[tuple[str, str], dict[str, object]] = {} + self._model_evidence_day: date | None = None self._calibration_task: asyncio.Task[None] | None = None self._closed = False @@ -121,7 +124,10 @@ def schedule_calibration(self) -> None: """Refresh displayed reliability once per UTC day without blocking pages.""" if ( self._closed - or self._now().date() == self._calibration_day + or ( + self._now().date() == self._calibration_day + and self._now().date() == self._model_evidence_day + ) or (self._calibration_task is not None and not self._calibration_task.done()) ): return @@ -129,15 +135,38 @@ def schedule_calibration(self) -> None: async def _refresh_calibration(self) -> None: now = self._now() - try: - summary = await asyncio.to_thread(build_calibration, self.ledger, now) - except asyncio.CancelledError: - raise - except Exception: - LOG.warning("forecast calibration refresh failed", exc_info=True) - else: - self._calibration = summary - self._calibration_day = now.date() + if self._calibration_day != now.date(): + try: + summary = await asyncio.to_thread(build_calibration, self.ledger, now) + except asyncio.CancelledError: + raise + except Exception: + LOG.warning("forecast calibration refresh failed", exc_info=True) + else: + self._calibration = summary + self._calibration_day = now.date() + if self._model_evidence_day != now.date(): + try: + reports = await asyncio.to_thread( + build_model_evidence, self.data_dir, self.ledger, now + ) + except asyncio.CancelledError: + raise + except Exception: + LOG.warning("model evidence refresh failed", exc_info=True) + else: + self._model_evidence = reports + self._model_evidence_day = now.date() + + def evidence_for(self, method: str, horizon_sessions: int) -> dict[str, object] | None: + band = horizon_band(horizon_sessions) + return self._model_evidence.get((method, band)) if band else None + + def evidence_index(self) -> dict[str, object]: + return { + f"{method}:{band}": report + for (method, band), report in self._model_evidence.items() + } async def _refresh(self, ticker: str) -> None: lock = self._refresh_locks.setdefault(ticker, asyncio.Lock()) @@ -231,7 +260,7 @@ def distribution( ): self._schedule_champion(ticker, completed, champion) return cached - result = self.forecaster.forecast( + result = getattr(self.forecaster, "forecast_baseline", self.forecaster.forecast)( ticker, now, expiry, @@ -264,6 +293,22 @@ def lookup( contract_since=contract_since, standard_terms=standard_terms, ) + return self.view_for_distribution( + distribution, side, strike, standard_terms=standard_terms, ticker=ticker + ), distribution + + def view_for_distribution( + self, + distribution: PredictiveDistribution, + side: Side, + strike: Decimal, + *, + standard_terms: bool = True, + ticker: str | None = None, + price_basis: str = "completed_close", + price_as_of: datetime | None = None, + ) -> PredictiveOddsView: + ticker = ticker or distribution.ticker common = { "method": distribution.method, "as_of_session": distribution.as_of, @@ -271,11 +316,16 @@ def lookup( "model_version": distribution.model_version, "support": distribution.support or None, "data_hash": distribution.data_hash, - "price_basis": "completed_close" if distribution.status == "available" else None, + "price_basis": price_basis if distribution.status == "available" else None, "price_as_of": ( - session_close(distribution.as_of) + price_as_of or session_close(distribution.as_of) if distribution.status == "available" else None ), + "evidence_key": ( + f"{distribution.method}:{band}" + if distribution.method and (band := horizon_band(distribution.horizon_sessions)) + else None + ), } if distribution.status != "available": refreshing = standard_terms and (ticker in self._tasks or ticker in self._pending) @@ -285,17 +335,17 @@ def lookup( "Refreshing completed price history" if refreshing else distribution.reason ), **common, - ), distribution + ) call = distribution.probability("call", strike) put = distribution.probability("put", strike) if call is None or put is None: return PredictiveOddsView( status="unavailable", reason="model_probability_invalid", **common - ), distribution + ) if not 0 <= call <= 1 or not 0 <= put <= 1 or call + put > 1 + 1e-9: return PredictiveOddsView( status="unavailable", reason="model_probability_invalid", **common - ), distribution + ) call_tenths = int( (Decimal(str(call)) * 1000).to_integral_value(rounding=ROUND_HALF_UP) ) @@ -316,7 +366,7 @@ def lookup( ) evidence = ( self._calibration.get( - (distribution.model_version, "completed_close", band, money) + (distribution.model_version, price_basis, band, money, side) ) if distribution.model_version and band and money else None @@ -330,7 +380,7 @@ def lookup( PredictiveValidationEvidence.model_validate(evidence) if evidence else None ), **common, - ), distribution + ) async def close(self) -> None: self._closed = True diff --git a/backend/src/options_api/version.py b/backend/src/options_api/version.py new file mode 100644 index 0000000..7324325 --- /dev/null +++ b/backend/src/options_api/version.py @@ -0,0 +1,100 @@ +"""Read-only running and remote revision status for the local UI.""" + +from __future__ import annotations + +import asyncio +import os +import re +import subprocess +import threading +import time +from datetime import UTC, datetime +from pathlib import Path +from typing import Annotated, Literal + +from fastapi import APIRouter, Query +from pydantic import BaseModel + +ROOT = Path(__file__).resolve().parents[3] +CANONICAL_ORIGINS = { + "git@github.com:hypertrial/hyperoptions.git", + "ssh://git@github.com/hypertrial/hyperoptions.git", + "https://github.com/hypertrial/hyperoptions.git", + "https://github.com/hypertrial/hyperoptions", +} +SHA = re.compile(r"^[0-9a-f]{40}$") +POLL_SECONDS = 60 +SSH_COMMAND = "ssh -oBatchMode=yes -oConnectTimeout=4 -oConnectionAttempts=1" +router = APIRouter() + + +class VersionStatus(BaseModel): + running_sha: str | None + branch: str | None + remote_sha: str | None + status: Literal["current", "update_available", "offline", "unverified_checkout"] + checked_at: datetime + frontend_matches: bool | None + + +def _git(*args: str, timeout: int = 5) -> str | None: + try: + result = subprocess.run( + ["git", *args], + cwd=ROOT, + env={**os.environ, "GIT_TERMINAL_PROMPT": "0", "GIT_SSH_COMMAND": SSH_COMMAND}, + capture_output=True, + text=True, + timeout=timeout, + check=False, + ) + except (OSError, subprocess.TimeoutExpired): + return None + return result.stdout.strip() if result.returncode == 0 else None + + +_running_sha = os.getenv("OPTIONS_APP_SHA") or _git("rev-parse", "HEAD") +if _running_sha is not None and SHA.fullmatch(_running_sha) is None: + _running_sha = None +_branch = _git("symbolic-ref", "--short", "HEAD") +_cache_lock = threading.Lock() +_cached_at = 0.0 +_cached_status: tuple[str | None, str, datetime] | None = None + + +def get_version_status(frontend_sha: str | None = None) -> VersionStatus: + global _cached_at, _cached_status + with _cache_lock: + if _cached_status is None or time.monotonic() - _cached_at >= POLL_SECONDS: + checked_at = datetime.now(UTC) + origin = _git("remote", "get-url", "origin") + if _running_sha is None or _branch != "main" or origin not in CANONICAL_ORIGINS: + remote_sha, status = None, "unverified_checkout" + else: + response = _git("ls-remote", "origin", "refs/heads/main", timeout=8) + remote_sha = response.split("\t", 1)[0] if response else None + if remote_sha is None or SHA.fullmatch(remote_sha) is None: + remote_sha, status = None, "offline" + elif remote_sha == _running_sha: + status = "current" + else: + status = "update_available" + _cached_status = remote_sha, status, checked_at + _cached_at = time.monotonic() + remote_sha, status, checked_at = _cached_status + + return VersionStatus( + running_sha=_running_sha, + branch=_branch, + remote_sha=remote_sha, + status=status, + checked_at=checked_at, + frontend_matches=(_running_sha == frontend_sha) if frontend_sha is not None else None, + ) + + +@router.get("/api/version", response_model=VersionStatus) +async def version( + frontend_sha: Annotated[str | None, Query(pattern=r"^[0-9a-f]{40}$")] = None, +) -> VersionStatus: + return await asyncio.to_thread(get_version_status, frontend_sha) diff --git a/backend/src/options_api/watchlist.py b/backend/src/options_api/watchlist.py index 9025dfd..d684d83 100644 --- a/backend/src/options_api/watchlist.py +++ b/backend/src/options_api/watchlist.py @@ -10,13 +10,13 @@ from datetime import UTC, date, datetime, timedelta from decimal import Decimal from pathlib import Path -from typing import Literal +from typing import Annotated, Literal -from fastapi import APIRouter, HTTPException, Request +from fastapi import APIRouter, HTTPException, Query, Request from pydantic import BaseModel, ConfigDict, Field from options_api.contract_identity import make_watch_key, parse_watch_key, strike_exact -from options_api.live_quant import quant_for_contract +from options_api.live_quant import _MODEL_VERSIONS, quant_for_contract from options_api.market_calendar import ( expiry_session_completed, first_session_after_completed, @@ -27,9 +27,11 @@ HypotheticalRiskView, MarketOddsView, OptionQuote, + PhysicalModel, PredictiveOddsView, normalize_ticker, ) +from options_api.market_watch import _CURVE_VERSION, _MODEL_VERSION from options_api.nasdaq import NasdaqError from options_api.outcomes import ( TERMS_NOTE, @@ -76,6 +78,8 @@ class WatchItem(BaseModel): market_odds: MarketOddsView = Field(default_factory=MarketOddsView) last_available_market_odds: MarketOddsView | None = None predictive_odds: PredictiveOddsView = Field(default_factory=PredictiveOddsView) + physical_models: list[PredictiveOddsView] = Field(default_factory=list) + market_models: list[MarketOddsView] = Field(default_factory=list) hypothetical_risk: HypotheticalRiskView = Field(default_factory=HypotheticalRiskView) outcome: OutcomeView @@ -83,12 +87,14 @@ class WatchItem(BaseModel): class WatchListResponse(BaseModel): items: list[WatchItem] active_job: Job | None = None + model_evidence: dict[str, object] = Field(default_factory=dict) class WatchCreateResponse(BaseModel): item: WatchItem created: bool job: Job | None + model_evidence: dict[str, object] = Field(default_factory=dict) class WatchRefreshResponse(BaseModel): @@ -468,7 +474,10 @@ def _queue(request: Request, *, force: bool = False, retry_pending: bool = False @router.get("/api/watchlist", response_model=WatchListResponse) -async def get_watchlist(request: Request) -> WatchListResponse: +async def get_watchlist( + request: Request, + forecast_model: Annotated[PhysicalModel, Query()] = "lognormal_ewma", +) -> WatchListResponse: now = _now(request) items = request.app.state.watchlist.items(as_of=now) odds = request.app.state.market_odds @@ -483,9 +492,28 @@ async def get_watchlist(request: Request) -> WatchListResponse: status="unavailable", reason="Expiry session completed; see outcome" ) item.predictive_odds = PredictiveOddsView( - status="unavailable", reason="expiry_completed" + method=forecast_model, status="unavailable", reason="expiry_completed" + ) + item.physical_models = [ + PredictiveOddsView( + method=method, model_version=version, + status="unavailable", reason="expiry_completed", + ) + for method, version in _MODEL_VERSIONS.items() + ] + item.market_models = [ + MarketOddsView( + method=method, model_version=version, + status="unavailable", reason="expiry_completed", + ) + for method, version in ( + ("regimelib", _MODEL_VERSION), + ("constrained_call_curve", _CURVE_VERSION), + ) + ] + item.hypothetical_risk = HypotheticalRiskView( + reason="Expiry session completed", forecast_method=forecast_model ) - item.hypothetical_risk = HypotheticalRiskView(reason="Expiry session completed") continue result = quant_for_contract( odds, @@ -498,12 +526,16 @@ async def get_watchlist(request: Request) -> WatchListResponse: contract_since=first_session_after_completed(item.created_at), watched=True, terms_note=item.terms_note, + physical_shadow=getattr(request.app.state, "physical_shadow", None), + forecast_model=forecast_model, ) if result.issuance is not None: issuances.append(result.issuance) item.market_odds = result.market item.last_available_market_odds = result.last_available_market item.predictive_odds = result.predictive + item.physical_models = list(result.physical_models) + item.market_models = list(result.market_models) item.hypothetical_risk = result.risk if issuances: try: @@ -511,7 +543,9 @@ async def get_watchlist(request: Request) -> WatchListResponse: except Exception: LOG.exception("forecast issuance ledger unavailable for watchlist") for item in items: - if item.predictive_odds.status == "available": + if item.predictive_odds.status in {"available", "pending"} or any( + model.status in {"available", "pending"} for model in item.physical_models + ): item.predictive_odds = item.predictive_odds.model_copy( update={ "status": "unavailable", @@ -524,16 +558,28 @@ async def get_watchlist(request: Request) -> WatchListResponse: item.hypothetical_risk = HypotheticalRiskView( reason="Forecast evidence unavailable" ) + item.physical_models = [ + model.model_copy(update={ + "status": "unavailable", "reason": "Forecast evidence unavailable", + "itm_pct_tenths": None, "otm_pct_tenths": None, + "atm_pct_tenths": None, + }) for model in item.physical_models + ] else: request.app.state.physical_shadow.submit(issuances) return WatchListResponse( items=items, active_job=request.app.state.jobs.active("watch_refresh"), + model_evidence=request.app.state.predictive_odds.evidence_index(), ) @router.post("/api/watchlist", response_model=WatchCreateResponse) -async def add_watch(request: Request, body: WatchCreate) -> WatchCreateResponse: +async def add_watch( + request: Request, + body: WatchCreate, + forecast_model: Annotated[PhysicalModel, Query()] = "lognormal_ewma", +) -> WatchCreateResponse: parsed = parse_watch_key(body.watch_key) if parsed is None: raise HTTPException(status_code=400, detail="Invalid watch key") @@ -585,6 +631,8 @@ async def add_watch(request: Request, body: WatchCreate) -> WatchCreateResponse: contract_since=first_session_after_completed(record.created_at), watched=True, terms_note=item.terms_note, + physical_shadow=getattr(request.app.state, "physical_shadow", None), + forecast_model=forecast_model, ) ledger_failed = False if result.issuance is not None: @@ -603,13 +651,23 @@ async def add_watch(request: Request, body: WatchCreate) -> WatchCreateResponse: status="unavailable", reason="Forecast evidence unavailable" ) item.hypothetical_risk = HypotheticalRiskView(reason="Forecast evidence unavailable") + item.physical_models = [ + model.model_copy(update={ + "status": "unavailable", "reason": "Forecast evidence unavailable", + "itm_pct_tenths": None, "otm_pct_tenths": None, + "atm_pct_tenths": None, + }) for model in result.physical_models + ] else: item.predictive_odds = result.predictive item.hypothetical_risk = result.risk + item.physical_models = list(result.physical_models) + item.market_models = list(result.market_models) return WatchCreateResponse( item=item, created=created, job=job, + model_evidence=request.app.state.predictive_odds.evidence_index(), ) diff --git a/backend/src/stocksweeper/forecast/calibration.py b/backend/src/stocksweeper/forecast/calibration.py index 7899b08..7d37e53 100644 --- a/backend/src/stocksweeper/forecast/calibration.py +++ b/backend/src/stocksweeper/forecast/calibration.py @@ -16,7 +16,7 @@ from stocksweeper.forecast.calendar import SessionCalendar from stocksweeper.forecast.ledger import ForecastLedger -CalibrationKey = tuple[str, str, str, str] +CalibrationKey = tuple[str, str, str, str, str] _BANDS = {"1": (1, 1), "2-5": (2, 5), "6-25": (6, 25), "26-252": (26, 252)} _LOOKBACK = timedelta(days=4 * 366) @@ -164,7 +164,9 @@ def horizon(origin: date, expiration: date) -> int: ) if row["data_hash"] != vintage[unit][2]: continue - key = (row["model_version"], row["price_basis"], band, moneyness) + # Calls and puts at the same strike have complementary labels. Mixing + # them can make a poorly calibrated model appear perfectly calibrated. + key = (row["model_version"], row["price_basis"], band, moneyness, row["side"]) grouped[key][row["input_session"]][unit[:3]].append( ( float(row["itm_probability"]), @@ -194,6 +196,7 @@ def horizon(origin: date, expiration: date) -> int: "model_version": key[0], "horizon_band": key[2], "moneyness_band": key[3], + "option_side": key[4], "independent_units": len(units), "predicted_itm_pct_tenths": _tenths(mean(unit[1] for unit in units)), "observed_itm_pct_tenths": _tenths(mean(unit[2] for unit in units)), diff --git a/backend/src/stocksweeper/forecast/evidence_reports.py b/backend/src/stocksweeper/forecast/evidence_reports.py new file mode 100644 index 0000000..40cdf34 --- /dev/null +++ b/backend/src/stocksweeper/forecast/evidence_reports.py @@ -0,0 +1,385 @@ +"""Dated, descriptive model evidence for the local comparison UI. + +These reports never select a champion. Formal promotion still requires the +separate, predeclared prospective holdout in ``promotion.py``. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import tempfile +from dataclasses import replace +from datetime import UTC, date, datetime, timedelta +from decimal import Decimal +from pathlib import Path +from typing import Any + +from options_api.market_calendar import session_close +from options_api.intraday_shadow import _VERSION as INTRADAY_VERSION +from stocksweeper.forecast.calendar import SessionCalendar +from stocksweeper.forecast.intraday_evidence import evaluate as evaluate_intraday +from stocksweeper.forecast.ledger import ForecastLedger +from stocksweeper.forecast.physical_contest import ( + EMPIRICAL_SHADOW_VERSION, + GJR_VERSION, + STUDENT_VERSION, +) +from stocksweeper.forecast.physical_evaluation import ContestRow, evaluate_band +from stocksweeper.forecast.predictive import BASELINE_VERSION + +CANDIDATE_VERSIONS = { + "empirical_scaled": EMPIRICAL_SHADOW_VERSION, + "student_t_ewma": STUDENT_VERSION, + "gjr_garch_t": GJR_VERSION, +} +BANDS = {"1": range(1, 2), "2-5": range(2, 6), "6-25": range(6, 26)} + + +def ledger_band_rows( + ledger: ForecastLedger, + calendar: SessionCalendar, + provenance: str, + candidate: str, + holdout_start: date | None, + period: str, + band: str, + as_of: datetime | None = None, + since: date | None = None, +) -> list[ContestRow]: + """Adapt first-party issuance rows to the paired evaluator's exact grain.""" + horizons = BANDS[band] + start = max( + (day for day in (since, holdout_start if period == "holdout" else None) if day), + default=None, + ) + issues = ledger.iter_evaluation_rows( + provenance=provenance, + since=start, + expiry_before=holdout_start if period == "screen" else None, + methods=("lognormal_ewma", candidate), + horizon_range=(horizons.start, horizons.stop - 1), + with_crps=True, + issued_before=as_of, + label_as_of=as_of, + ) + rows: list[ContestRow] = [] + for item in issues: + origin = item["input_session"] + method = item["method"] + if origin is None or method is None: + continue + label_valid = ( + item["label_status"] == "valid" + and item["label_checked_at"] is not None + and item["label_checked_at"] >= session_close(item["expiry_session"]) + and item["label_checked_at"] > item["issued_at"] + and item["issued_at"] < session_close(item["expiry_session"]) + ) + spot = Decimal(item["spot_exact"]) if item["spot_exact"] else None + relative = abs(float(Decimal(item["strike_exact"]) / spot - 1)) if spot else None + moneyness = ( + "unknown" if relative is None else + "near_atm" if relative <= 0.05 else + "moderate" if relative <= 0.15 else "tail" + ) + rows.append( + ContestRow( + ticker=item["ticker"], origin=origin, + expiry_session=item["expiry_session"], + horizon=calendar.horizon(origin, item["expiration"]), + strike=item["strike_exact"], side=item["side"], method=method, + probability=(item["itm_probability"] if item["status"] == "available" else None), + observed_itm=(bool(item["observed_itm"]) if label_valid else None), + provenance=provenance, moneyness=moneyness, + volatility_regime=item["volatility_regime"] or "unknown", + event_status=item["known_event_status"] or "unknown", + reason=item["unavailable_reason"] or item["label_reason"], + input_vintage=item["data_hash"], issued_at=item["issued_at"], + issuance_key=item["idempotency_key"], contract_id=item["contract_key"], + crps=(item["crps"] if label_valid else None), + prepare_ms=item["prepare_ms"], lookup_ms=item["lookup_ms"], + ) + ) + return rows + + +def ledger_contest( + ledger: ForecastLedger, + calendar: SessionCalendar, + provenance: str, + candidate: str, + holdout_start: date | None, + period: str, + as_of: datetime | None = None, + since: date | None = None, +) -> dict[str, Any]: + start = max( + (day for day in (since, holdout_start if period == "holdout" else None) if day), + default=None, + ) + return { + "source": "append-only forecast ledger", + "provenance": provenance, + "skipped_attempts": ledger.evaluation_skipped_attempts(provenance), + "prospective_panel_coverage": ( + ledger.panel_coverage( + since=start, + before=holdout_start if period == "screen" else None, + as_of=as_of, + ) if provenance == "as_issued" else None + ), + "bands": { + band: evaluate_band( + ledger_band_rows( + ledger, calendar, provenance, candidate, holdout_start, period, band, + as_of=as_of, + since=since, + ), + candidate, band, holdout_start=holdout_start, period=period, calendar=calendar, + ) + for band in BANDS + }, + } + + +def _digest(value: object) -> str: + return hashlib.sha256( + json.dumps(value, sort_keys=True, separators=(",", ":"), allow_nan=False).encode() + ).hexdigest() + + +def save_replay_report(data_dir: Path, candidate: str, report: dict[str, Any]) -> Path: + """Store explicitly retrospective evidence outside Git, atomically.""" + if candidate not in CANDIDATE_VERSIONS or report.get("provenance") != "immutable_replay": + raise ValueError("invalid replay candidate or provenance") + path = data_dir / "forecast" / "evidence" / f"replay-{candidate}.json" + payload = { + "schema_version": 2, + "candidate": candidate, + "candidate_version": CANDIDATE_VERSIONS[candidate], + "baseline_version": BASELINE_VERSION, + "generated_at": datetime.now(UTC).isoformat(), + "report": report, + "report_hash": _digest(report), + } + path.parent.mkdir(parents=True, exist_ok=True) + fd, temporary = tempfile.mkstemp(prefix=".replay-", dir=path.parent) + try: + with os.fdopen(fd, "w") as stream: + json.dump(payload, stream, sort_keys=True, allow_nan=False) + stream.flush() + os.fsync(stream.fileno()) + os.replace(temporary, path) + finally: + if os.path.exists(temporary): + os.unlink(temporary) + return path + + +def load_replay_report(data_dir: Path, candidate: str) -> dict[str, Any] | None: + if candidate not in CANDIDATE_VERSIONS: + return None + path = data_dir / "forecast" / "evidence" / f"replay-{candidate}.json" + try: + payload = json.loads(path.read_text()) + report = payload["report"] + if ( + payload.get("schema_version") != 2 + or payload["candidate"] != candidate + or payload["candidate_version"] != CANDIDATE_VERSIONS[candidate] + or payload["baseline_version"] != BASELINE_VERSION + or report["provenance"] != "immutable_replay" + or payload["report_hash"] != _digest(report) + or not isinstance(report["bands"], dict) + ): + return None + return payload + except (OSError, KeyError, TypeError, ValueError): + return None + + +def _summary( + band_report: dict[str, Any], *, provenance: str, generated_at: str, + report_hash: str, model_version: str, reference_only: bool = False, + audit_session: str | None = None, audit_frozen_at: str | None = None, + evidence_window_start: date | None = None, +) -> dict[str, Any]: + dates = band_report["independent_date_blocks"] + brier = dict(band_report["brier"]) + log_loss = dict(band_report["log_loss"]) + if dates < 20: + brier["bootstrap_95"] = None + log_loss["bootstrap_95"] = None + if reference_only: + brier["paired_delta"] = None + brier["bootstrap_95"] = None + log_loss["paired_delta"] = None + log_loss["bootstrap_95"] = None + return { + "provenance": provenance, + "generated_at": generated_at, + "report_hash": report_hash, + "model_version": model_version, + "input_version": ( + "frozen-yahoo-snapshot-v1" if provenance == "immutable_replay" + else "forecast-ledger-v1" + ), + "audit_session": audit_session, + "audit_frozen_at": audit_frozen_at, + "evidence_window_start": ( + evidence_window_start.isoformat() if evidence_window_start else None + ), + "tickers": band_report["tickers"], + "independent_date_blocks": dates, + "ticker_origin_horizon_units": band_report["ticker_origin_horizon_units"], + "contract_forecasts_available": band_report["contract_forecasts_available"], + "contract_cells_attempted": band_report["contract_cells_attempted"], + "coverage": ( + band_report["contract_forecasts_available"] / band_report["contract_cells_attempted"] + if band_report["contract_cells_attempted"] else None + ), + "coverage_basis": "recorded_contract_cells_with_baseline_issuance", + "replay_scheduled_units": band_report.get("replay_scheduled_units"), + "replay_baseline_available_units": band_report.get("replay_baseline_available_units"), + "replay_fit_coverage": ( + band_report["replay_baseline_available_units"] / band_report["replay_scheduled_units"] + if band_report.get("replay_scheduled_units") else None + ), + "replay_rejection_reasons": band_report.get("replay_rejection_reasons"), + "rejection_reasons": band_report["rejection_reasons"], + "brier": brier, + "log_loss": log_loss, + "crps": band_report["crps"], + "calibration_by_side": band_report["calibration_by_side"], + "calibration_count_basis": "independent_ticker_origin_horizon_side_bin", + "subgroups": band_report["subgroups"], + "latency_ms": band_report["latency_ms"], + "significance": ( + "not_applicable" if reference_only else + "exploratory_interval" if dates >= 20 else "not_estimable" + ), + } + + +def build_model_evidence( + data_dir: Path, ledger: ForecastLedger, as_of: datetime, +) -> dict[tuple[str, str], dict[str, Any]]: + """Build once in a background task; HTTP lookups read the resulting map.""" + calendar = SessionCalendar() + # ponytail: cap the daily ledger scan at three years; use offline grouped reports if it grows. + since = as_of.date() - timedelta(days=1096) + result: dict[tuple[str, str], dict[str, Any]] = {} + for band in BANDS: + baseline_rows = ledger_band_rows( + ledger, calendar, "as_issued", "lognormal_ewma", None, "all", band, + as_of=as_of, since=since, + ) + reference = evaluate_band( + [*baseline_rows, *(replace(row, method="baseline_reference") for row in baseline_rows)], + "baseline_reference", band, calendar=calendar, bootstrap_samples=100, + ) + result[("lognormal_ewma", band)] = { + "prospective": _summary( + reference, provenance="as_issued", generated_at=as_of.isoformat(), + report_hash=_digest(reference), model_version=BASELINE_VERSION, + reference_only=True, + evidence_window_start=since, + ), + "retrospective": None, + } + for candidate, version in CANDIDATE_VERSIONS.items(): + prospective = ledger_contest( + ledger, calendar, "as_issued", candidate, None, "all", as_of=as_of, + since=since, + ) + replay = load_replay_report(data_dir, candidate) + prospective_hash = _digest(prospective) + for band in BANDS: + result[(candidate, band)] = { + "prospective": _summary( + prospective["bands"][band], provenance="as_issued", + generated_at=as_of.isoformat(), report_hash=prospective_hash, + model_version=version, + evidence_window_start=since, + ), + "retrospective": ( + _summary( + replay["report"]["bands"][band], provenance="immutable_replay", + generated_at=replay["generated_at"], report_hash=replay["report_hash"], + model_version=version, + audit_session=replay["report"].get("audit_session"), + audit_frozen_at=replay["report"].get("audit_frozen_at"), + ) if replay is not None and band in replay["report"]["bands"] else None + ), + } + for band, horizons in BANDS.items(): + intraday_rows = list(ledger.iter_evaluation_rows( + provenance="as_issued", + methods=("lognormal_ewma", "quote_reanchored_comparator", "intraday_shadow"), + horizon_range=(horizons.start, horizons.stop - 1), + since=since, + issued_before=as_of, + label_as_of=as_of, + )) + report = evaluate_intraday(intraday_rows, as_of=as_of) + summary = report["overall"] + blocks = summary["scored_date_blocks"] + intervals = summary["paired_calendar_date_bootstrap_95"] if blocks >= 20 else None + result[("intraday_shadow", band)] = { + "prospective": { + "provenance": "as_issued_matched_intraday_windows", + "generated_at": as_of.isoformat(), + "report_hash": _digest(report), + "model_version": INTRADAY_VERSION, + "input_version": "forecast-ledger-quote-snapshot-v1", + "evidence_window_start": since.isoformat(), + "tickers": summary["scored_tickers"], + "independent_date_blocks": blocks, + "ticker_origin_horizon_units": summary[ + "scored_ticker_date_window_expiry_units" + ], + "contract_forecasts_available": summary[ + "forecast_available_cells" + ]["intraday_shadow"], + "contract_cells_attempted": summary["recorded_contract_windows"], + "coverage": ( + summary["forecast_available_cells"]["intraday_shadow"] + / summary["recorded_contract_windows"] + if summary["recorded_contract_windows"] else None + ), + "coverage_basis": "recorded_intraday_contract_windows", + "coverage_limit": report["denominator_limit"], + "rejection_reasons": summary["rejection_reasons"], + "brier": { + "baseline": summary["mean"]["brier"]["dated_close"], + "candidate": summary["mean"]["brier"]["intraday_shadow"], + "paired_delta": summary[ + "paired_shadow_minus_reference" + ]["brier"]["dated_close"], + "bootstrap_95": intervals["brier"]["dated_close"] if intervals else None, + }, + "log_loss": { + "baseline": summary["mean"]["log_loss"]["dated_close"], + "candidate": summary["mean"]["log_loss"]["intraday_shadow"], + "paired_delta": summary[ + "paired_shadow_minus_reference" + ]["log_loss"]["dated_close"], + "bootstrap_95": intervals["log_loss"]["dated_close"] if intervals else None, + }, + "quote_reanchored_comparator": { + metric: summary["mean"][metric]["quote_reanchored_comparator"] + for metric in ("brier", "log_loss") + }, + "calibration_by_side": summary["calibration_by_side"], + "calibration_count_basis": "independent_ticker_date_window_expiry_side_bin", + "latency_ms": summary["latency_ms"].get("intraday_shadow", {}), + "by_window": report["by_window"], + "by_expiry": report["by_expiry"], + "by_calendar": report["by_calendar"], + "significance": "exploratory_interval" if intervals else "not_estimable", + }, + "retrospective": None, + } + return result diff --git a/backend/src/stocksweeper/forecast/intraday_evidence.py b/backend/src/stocksweeper/forecast/intraday_evidence.py new file mode 100644 index 0000000..0bb0e2f --- /dev/null +++ b/backend/src/stocksweeper/forecast/intraday_evidence.py @@ -0,0 +1,388 @@ +"""Matched-window prospective evaluation of intraday forecasts.""" + +from __future__ import annotations + +from collections import Counter, defaultdict +from datetime import UTC, date, datetime, time, timedelta +from statistics import mean +from zoneinfo import ZoneInfo + +import numpy as np + +from options_api.market_calendar import _calendar, session_close +from stocksweeper.forecast.physical_evaluation import _quantile, _score + +_NY = ZoneInfo("America/New_York") +_WINDOWS = ("10:00", "13:00", "15:30") +_MODELS = ("dated_close", "quote_reanchored_comparator", "intraday_shadow") +_CHALLENGERS = _MODELS[1:] + + +def _cell_key(row: dict[str, object]) -> tuple[object, ...]: + return ( + row["contract_key"], + row["root"], + row["side"], + row["expiration"], + row["expiry_session"], + row["strike_exact"], + row["terms_note"], + row["contract_since"], + row["issued_at"].astimezone(_NY).date(), + row["snapshot_window"], + ) + + +def _vintage(row: dict[str, object]) -> tuple[object, ...]: + # A refetch can change retrieval time without changing the model input. + return row["input_session"], row["data_hash"] + + +def _order(row: dict[str, object]) -> tuple[datetime, str]: + return row["issued_at"], row["idempotency_key"] + + +def _calendar_stratum(day: date) -> str: + calendar = _calendar() + if not calendar.is_session(day.isoformat()): + return "holiday_or_non_session" + session = calendar.date_to_session(day.isoformat(), direction="none") + duration = calendar.session_close(session) - calendar.session_open(session) + return "early_close" if duration < timedelta(hours=6, minutes=30) else "regular_session" + + +def _summary(cells: list[dict[str, object]], *, bootstrap: bool = False) -> dict[str, object]: + available: Counter[str] = Counter() + issued: Counter[str] = Counter() + reasons: Counter[str] = Counter() + latencies: dict[str, list[float]] = defaultdict(list) + quote_ages: list[float] = [] + target_lags: list[float] = [] + units: dict[tuple[object, ...], list[dict[str, tuple[float, float]]]] = defaultdict(list) + for cell in cells: + attempts = cell["attempts"] + for model, attempt in attempts.items(): + if attempt is None: + continue + issued[model] += 1 + if attempt["status"] == "available": + available[model] += 1 + latency = attempt["lookup_ms"] + if ( + model in _CHALLENGERS + and latency is not None + and np.isfinite(latency) + and latency >= 0 + ): + latencies[model].append(float(latency)) + reason = cell["reason"] + if reason is not None: + reasons[reason] += 1 + if cell["scores"] is not None: + units[(cell["ticker"], cell["day"], cell["window"], cell["expiry_session"])].append( + cell["scores"] + ) + if cell["quote_age_ms"] is not None: + quote_ages.append(cell["quote_age_ms"]) + if cell["target_lag_ms"] is not None: + target_lags.append(cell["target_lag_ms"]) + latest_expiry_by_day: dict[date, date] = {} + for _, day, _, expiry in units: + latest_expiry_by_day[day] = max(expiry, latest_expiry_by_day.get(day, expiry)) + independent_days: set[date] = set() + next_allowed: date | None = None + for day, latest_expiry in sorted(latest_expiry_by_day.items()): + if next_allowed is None or day >= next_allowed: + independent_days.add(day) + next_allowed = latest_expiry + timedelta(days=1) + independent_units = { + key: contracts for key, contracts in units.items() if key[1] in independent_days + } + calibration_bins = {side: [[] for _ in range(10)] for side in ("call", "put")} + calibration_units: dict[ + tuple[str, date, str, date, str, int], list[tuple[float, float]] + ] = defaultdict(list) + for cell in cells: + if cell["scores"] is None or cell["day"] not in independent_days: + continue + issued_shadow = cell["attempts"]["intraday_shadow"] + probability = float(issued_shadow["itm_probability"]) + calibration_units[ + cell["ticker"], cell["day"], cell["window"], cell["expiry_session"], + cell["side"], min(9, int(probability * 10)), + ].append( + (probability, float(issued_shadow["observed_itm"])) + ) + for (_, _, _, _, side, bin_index), values in calibration_units.items(): + calibration_bins[side][bin_index].append( + (mean(p for p, _ in values), mean(y for _, y in values)) + ) + unit_scores = [ + { + model: ( + mean(contract[model][0] for contract in contracts), + mean(contract[model][1] for contract in contracts), + ) + for model in _MODELS + } + for contracts in independent_units.values() + ] + means = { + metric: { + model: mean(score[model][index] for score in unit_scores) if unit_scores else None + for model in _MODELS + } + for index, metric in enumerate(("brier", "log_loss")) + } + deltas = { + metric: { + reference: ( + means[metric]["intraday_shadow"] - means[metric][reference] if unit_scores else None + ) + for reference in _MODELS[:2] + } + for metric in ("brier", "log_loss") + } + intervals = None + if bootstrap and len(independent_days) >= 2: + by_day: dict[date, list[dict[str, tuple[float, float]]]] = defaultdict(list) + for key, contracts in independent_units.items(): + by_day[key[1]].append( + { + model: ( + mean(contract[model][0] for contract in contracts), + mean(contract[model][1] for contract in contracts), + ) + for model in _MODELS + } + ) + days = sorted(by_day) + counts = np.asarray([len(by_day[day]) for day in days]) + sampled = np.random.default_rng(20260927).integers(0, len(days), size=(2000, len(days))) + sampled_counts = counts[sampled].sum(axis=1) + intervals = {} + for index, metric in enumerate(("brier", "log_loss")): + intervals[metric] = {} + for reference in _MODELS[:2]: + sums = np.asarray( + [ + sum( + score["intraday_shadow"][index] - score[reference][index] + for score in by_day[day] + ) + for day in days + ] + ) + values = sums[sampled].sum(axis=1) / sampled_counts + intervals[metric][reference] = [ + _quantile(values.tolist(), 0.025), + _quantile(values.tolist(), 0.975), + ] + return { + "recorded_contract_windows": len(cells), + "recorded_contracts": len({cell["contract_key"] for cell in cells}), + "recorded_tickers": len({cell["ticker"] for cell in cells}), + "recorded_dates": len({cell["day"] for cell in cells}), + "forecast_present_cells": {model: issued[model] for model in _MODELS}, + "forecast_available_cells": {model: available[model] for model in _MODELS}, + "triple_available_contract_windows": sum( + all( + attempt is not None and attempt["status"] == "available" + for attempt in cell["attempts"].values() + ) + for cell in cells + ), + "scored_contract_windows": sum(cell["scores"] is not None for cell in cells), + "scored_ticker_date_window_expiry_units": len(independent_units), + "scored_tickers": len({key[0] for key in independent_units}), + "scored_date_blocks": len(independent_days), + "scored_dates_before_overlap_purge": len(latest_expiry_by_day), + "mean": means, + "paired_shadow_minus_reference": deltas, + "paired_calendar_date_bootstrap_95": intervals, + "calibration_by_side": { + side: [ + { + "lower": index / 10, + "upper": (index + 1) / 10, + "count": len(values), + "forecast_mean": mean(p for p, _ in values) if values else None, + "observed_rate": mean(y for _, y in values) if values else None, + } + for index, values in enumerate(bins) + ] + for side, bins in calibration_bins.items() + }, + "rejection_reasons": dict(sorted(reasons.items())), + "latency_ms": { + model: { + "p50": _quantile(latencies[model], 0.5), + "p95": _quantile(latencies[model], 0.95), + } + for model in _CHALLENGERS + }, + "quote_age_ms": {"p50": _quantile(quote_ages, 0.5), "p95": _quantile(quote_ages, 0.95)}, + "target_to_issue_lag_ms": { + "p50": _quantile(target_lags, 0.5), + "p95": _quantile(target_lags, 0.95), + }, + } + + +def evaluate(rows: list[dict[str, object]], *, as_of: datetime | None = None) -> dict[str, object]: + """Use first prospective attempt per contract/window; never select by outcome.""" + now = as_of or datetime.now(UTC) + if now.tzinfo is None: + raise ValueError("as_of must be timezone-aware") + primary: dict[tuple[object, ...], list[dict[str, object]]] = defaultdict(list) + candidates: dict[tuple[object, ...], dict[str, list[dict[str, object]]]] = defaultdict( + lambda: defaultdict(list) + ) + for row in rows: + if row["provenance"] != "as_issued": + continue + if row["method"] in _CHALLENGERS and row["snapshot_window"] in _WINDOWS: + candidates[_cell_key(row)][row["method"]].append(row) + elif ( + row["method"] == "lognormal_ewma" + and row["price_basis"] == "completed_close" + and row["snapshot_window"] is None + ): + key = _cell_key({**row, "snapshot_window": None})[:8] + primary[key].append(row) + cells: list[dict[str, object]] = [] + duplicate_attempts = 0 + for key, attempts in sorted(candidates.items(), key=lambda pair: tuple(map(str, pair[0]))): + contract = key[:8] + day, window = key[-2:] + target = datetime.combine(day, time.fromisoformat(window), _NY).astimezone(UTC) + selected = { + model: min(attempts[model], key=_order) if attempts[model] else None + for model in _CHALLENGERS + } + duplicate_attempts += sum(max(0, len(attempts[model]) - 1) for model in _CHALLENGERS) + first = min((row for row in selected.values() if row is not None), key=_order) + last = max((row for row in selected.values() if row is not None), key=_order) + # Primary issues are idempotent across unchanged data; a window refresh + # may retain the earlier as-issued close forecast rather than insert one. + same_contract = [row for row in primary[contract] if row["issued_at"] <= first["issued_at"]] + same_vintage = [row for row in same_contract if _vintage(row) == _vintage(first)] + baseline = max(same_vintage, key=_order) if same_vintage else None + trio = {"dated_close": baseline, **selected} + reason = None + if len(selected) != 2 or any(row is None for row in selected.values()): + reason = "missing_challenger_attempt" + elif selected[_CHALLENGERS[0]]["issued_at"] != selected[_CHALLENGERS[1]]["issued_at"]: + reason = "challenger_capture_time_mismatch" + elif _vintage(selected[_CHALLENGERS[0]]) != _vintage(selected[_CHALLENGERS[1]]): + reason = "challenger_input_vintage_mismatch" + elif baseline is None: + reason = "baseline_input_vintage_mismatch" if same_contract else "baseline_not_issued" + elif any(row["status"] != "available" for row in trio.values()): + reason = next( + row["unavailable_reason"] or f"{model}_unavailable" + for model, row in trio.items() + if row["status"] != "available" + ) + elif any( + row["itm_probability"] is None or not 0 <= row["itm_probability"] <= 1 + for row in trio.values() + ): + reason = "invalid_probability" + elif ( + selected[_CHALLENGERS[0]]["quote_digest"] is None + or selected[_CHALLENGERS[0]]["quote_digest"] + != selected[_CHALLENGERS[1]]["quote_digest"] + ): + reason = "quote_vintage_missing_or_mismatch" + elif any(row["price_basis"] != "underlying_quote" for row in selected.values()): + reason = "quote_price_basis_mismatch" + elif ( + _calendar_stratum(day) == "holiday_or_non_session" + or target >= session_close(day) + or any( + not target <= row["issued_at"] < target + timedelta(minutes=5) + for row in selected.values() + ) + ): + reason = "snapshot_outside_regular_session" + elif any(row["label_status"] != "valid" for row in trio.values()): + reason = next( + row["label_reason"] or row["label_status"] or "label_missing" + for row in trio.values() + if row["label_status"] != "valid" + ) + elif ( + None in {row["observed_itm"] for row in trio.values()} + or len({row["observed_itm"] for row in trio.values()}) != 1 + ): + reason = "conflicting_exact_labels" + elif any(row["issued_at"] >= session_close(row["expiry_session"]) for row in trio.values()): + reason = "issued_after_expiry_close" + elif ( + first["expiry_session"] != last["expiry_session"] + or session_close(first["expiry_session"]) > now + or any( + row["label_checked_at"] is None + or row["label_checked_at"] < session_close(row["expiry_session"]) + or row["label_checked_at"] <= row["issued_at"] + or row["label_checked_at"] > now + for row in trio.values() + ) + ): + reason = "label_not_mature_at_scoring" + scores = None + if reason is None: + observed = bool(first["observed_itm"]) + scores = { + model: _score(float(row["itm_probability"]), observed) + for model, row in trio.items() + } + quote_time = first["quote_time"] + cells.append( + { + "contract_key": first["contract_key"], + "ticker": first["ticker"], + "side": first["side"], + "day": day, + "window": window, + "expiry_session": first["expiry_session"], + "calendar": _calendar_stratum(day), + "expiry": "same_day" if first["expiry_session"] == day else "future", + "attempts": trio, + "reason": reason, + "scores": scores, + "quote_age_ms": ( + (first["issued_at"] - quote_time).total_seconds() * 1000 + if quote_time is not None and first["issued_at"] >= quote_time + else None + ), + "target_lag_ms": ( + (first["issued_at"] - target).total_seconds() * 1000 + if first["issued_at"] >= target + else None + ), + } + ) + return { + "source": "append-only forecast ledger", + "provenance": "as_issued", + "scope": "recorded watchlist intraday contract windows only", + "denominator_limit": "Missed windows are absent from the issuance ledger.", + "metric": "binary ITM at exact expiry-session close; equality is ATM", + "selection": "First challenger attempt, preceding matching close forecast", + "duplicate_challenger_attempts_excluded": duplicate_attempts, + "overall": _summary(cells, bootstrap=True), + "by_window": { + window: _summary([cell for cell in cells if cell["window"] == window]) + for window in _WINDOWS + }, + "by_expiry": { + expiry: _summary([cell for cell in cells if cell["expiry"] == expiry]) + for expiry in ("same_day", "future") + }, + "by_calendar": { + calendar: _summary([cell for cell in cells if cell["calendar"] == calendar]) + for calendar in ("regular_session", "early_close", "holiday_or_non_session") + }, + } diff --git a/backend/src/stocksweeper/forecast/physical_evaluation.py b/backend/src/stocksweeper/forecast/physical_evaluation.py index fbd2fbd..4b9a79f 100644 --- a/backend/src/stocksweeper/forecast/physical_evaluation.py +++ b/backend/src/stocksweeper/forecast/physical_evaluation.py @@ -214,7 +214,9 @@ def evaluate_band( ) units: list[dict] = [] subgroup_values: dict[tuple[str, str, str, date, int], list[float]] = defaultdict(list) - calibration_bins: list[list[tuple[float, bool]]] = [[] for _ in range(10)] + calibration_bins: dict[str, list[list[tuple[float, float]]]] = { + side: [[] for _ in range(10)] for side in ("call", "put") + } for (ticker, origin, horizon), pairs in unit_pairs.items(): baseline_scores = [_score(base.probability, base.observed_itm) for base, _ in pairs] candidate_scores = [ @@ -246,10 +248,11 @@ def evaluate_band( "crps_candidate": mean(b for _, b in crps_values) if crps_values else None, } ) + unit_calibration: dict[tuple[str, int], list[tuple[float, float]]] = defaultdict(list) for base, challenger in pairs: - calibration_bins[min(9, int(challenger.probability * 10))].append( - (challenger.probability, challenger.observed_itm) - ) + unit_calibration[ + challenger.side, min(9, int(challenger.probability * 10)) + ].append((challenger.probability, float(challenger.observed_itm))) for category, value in ( ("horizon", str(horizon)), ("moneyness", challenger.moneyness), @@ -261,6 +264,10 @@ def evaluate_band( subgroup_values[(category, value, ticker, origin, horizon)].append( challenger_brier - base_brier ) + for (side, bin_index), values in unit_calibration.items(): + calibration_bins[side][bin_index].append( + (mean(p for p, _ in values), mean(y for _, y in values)) + ) by_date: dict[date, list[dict]] = defaultdict(list) for unit in units: by_date[unit["origin"]].append(unit) @@ -318,16 +325,19 @@ def evaluate_band( ), "availability": baseline_available == candidate_available, } - calibration = [ - { - "lower": index / 10, - "upper": (index + 1) / 10, - "count": len(values), - "forecast_mean": mean(p for p, _ in values) if values else None, - "observed_rate": mean(float(y) for _, y in values) if values else None, - } - for index, values in enumerate(calibration_bins) - ] + def calibration(bins: list[list[tuple[float, float]]]) -> list[dict[str, float | int | None]]: + return [ + { + "lower": index / 10, + "upper": (index + 1) / 10, + "count": len(values), + "forecast_mean": mean(p for p, _ in values) if values else None, + "observed_rate": mean(y for _, y in values) if values else None, + } + for index, values in enumerate(bins) + ] + + calibration_by_side = {side: calibration(bins) for side, bins in calibration_bins.items()} crps_units = [unit for unit in units if unit["crps_candidate"] is not None] return { "candidate": candidate, @@ -364,7 +374,7 @@ def evaluate_band( if crps_units else None, }, - "calibration": calibration, + "calibration_by_side": calibration_by_side, "subgroups": subgroups, "latency_ms": { label: {"p50": _quantile(values, 0.5), "p95": _quantile(values, 0.95)} diff --git a/backend/src/stocksweeper/forecast/predictive.py b/backend/src/stocksweeper/forecast/predictive.py index 52efdc3..f69f097 100644 --- a/backend/src/stocksweeper/forecast/predictive.py +++ b/backend/src/stocksweeper/forecast/predictive.py @@ -635,6 +635,7 @@ def forecast( *, contract_since: date | None = None, standard_terms: bool = True, + force_baseline: bool = False, ) -> PredictiveDistribution: if as_of.tzinfo is None: raise ValueError("as_of must have a timezone") @@ -715,7 +716,7 @@ def unavailable(reason: str) -> PredictiveDistribution: clean, spot, volatility, digest = inputs evidence = SelectionEvidence(rejection_reason="horizon_above_empirical_limit") empirical = None - if horizon <= _EMPIRICAL_MAX_HORIZON: + if horizon <= _EMPIRICAL_MAX_HORIZON and not force_baseline: candidate_key = (ticker, completed, horizon) candidate = self._candidates.get(candidate_key) if candidate is None: @@ -759,10 +760,24 @@ def unavailable(reason: str) -> PredictiveDistribution: weights=weights, selection=evidence, ) - return self._select_promoted(result) + return result if force_baseline else self._select_promoted(result) except (OSError, ValueError, OverflowError, pl.exceptions.PolarsError): return unavailable("market_data_invalid") + def forecast_baseline( + self, + ticker: str, + as_of: datetime, + expiry: date, + *, + contract_since: date | None = None, + standard_terms: bool = True, + ) -> PredictiveDistribution: + return self.forecast( + ticker, as_of, expiry, contract_since=contract_since, + standard_terms=standard_terms, force_baseline=True, + ) + def _select_promoted( self, frozen: PredictiveDistribution, diff --git a/backend/tests/test_dev_update.py b/backend/tests/test_dev_update.py new file mode 100644 index 0000000..ea93273 --- /dev/null +++ b/backend/tests/test_dev_update.py @@ -0,0 +1,98 @@ +from __future__ import annotations + +import importlib.util +import subprocess +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +SPEC = importlib.util.spec_from_file_location("dev_update", ROOT / "scripts" / "dev_update.py") +assert SPEC and SPEC.loader +dev_update = importlib.util.module_from_spec(SPEC) +SPEC.loader.exec_module(dev_update) + + +def git(repo: Path, *args: str) -> str: + result = subprocess.run( + ["git", *args], cwd=repo, capture_output=True, text=True, check=True + ) + return result.stdout.strip() + + +def repo_pair(tmp_path: Path, monkeypatch) -> tuple[Path, Path, Path]: + remote = tmp_path / "remote.git" + remote.mkdir() + git(remote, "init", "--bare", "-q") + local = tmp_path / "local" + local.mkdir() + git(local, "init", "-q", "-b", "main") + git(local, "config", "user.name", "Test") + git(local, "config", "user.email", "test@example.com") + (local / "README.md").write_text("base\n") + git(local, "add", ".") + git(local, "commit", "-qm", "base") + git(local, "remote", "add", "origin", str(remote)) + git(local, "push", "-q", "-u", "origin", "main") + git(remote, "symbolic-ref", "HEAD", "refs/heads/main") + other = tmp_path / "other" + git(tmp_path, "clone", "-q", str(remote), str(other)) + git(other, "config", "user.name", "Test") + git(other, "config", "user.email", "test@example.com") + monkeypatch.setattr(dev_update, "CANONICAL_ORIGINS", {str(remote)}) + return local, other, remote + + +def test_dev_update_fast_forwards_and_refuses_dirty_or_feature_branch( + tmp_path, monkeypatch +) -> None: + local, other, _ = repo_pair(tmp_path, monkeypatch) + (other / "README.md").write_text("new\n") + git(other, "commit", "-qam", "new") + git(other, "push", "-q", "origin", "main") + assert dev_update.update(local) == "updated" + assert git(local, "rev-parse", "HEAD") == git(other, "rev-parse", "HEAD") + assert dev_update.update(local) == "current" + + (local / "untracked").write_text("unsafe") + with pytest.raises(RuntimeError, match="commit or stash"): + dev_update.update(local) + (local / "untracked").unlink() + git(local, "checkout", "-qb", "codex/feature") + with pytest.raises(RuntimeError, match="checkout main"): + dev_update.update(local) + + +def test_dev_update_refuses_divergence_and_allows_offline_clean_main(tmp_path, monkeypatch) -> None: + local, other, remote = repo_pair(tmp_path, monkeypatch) + (local / "README.md").write_text("local\n") + git(local, "commit", "-qam", "local") + (other / "README.md").write_text("remote\n") + git(other, "commit", "-qam", "remote") + git(other, "push", "-q", "origin", "main") + with pytest.raises(RuntimeError, match="diverged"): + dev_update.update(local) + + remote.rename(tmp_path / "remote-offline.git") + with pytest.raises(RuntimeError, match="diverged"): + dev_update.update(local) + git(local, "reset", "--hard", "HEAD~1") + assert dev_update.update(local) == "offline" + + +def test_dev_update_timeout_is_unverified_but_wrong_origin_is_rejected( + tmp_path, monkeypatch +) -> None: + local, _, remote = repo_pair(tmp_path, monkeypatch) + real_git = dev_update.git + + def timed_out(root, *args, **kwargs): + if args[0] == "fetch": + raise subprocess.TimeoutExpired(["git", "fetch"], 15) + return real_git(root, *args, **kwargs) + + monkeypatch.setattr(dev_update, "git", timed_out) + assert dev_update.update(local) == "offline" + git(local, "remote", "set-url", "origin", str(remote) + "-unexpected") + with pytest.raises(RuntimeError, match="canonical"): + dev_update.update(local) diff --git a/backend/tests/test_evidence_reports.py b/backend/tests/test_evidence_reports.py new file mode 100644 index 0000000..4477f8a --- /dev/null +++ b/backend/tests/test_evidence_reports.py @@ -0,0 +1,80 @@ +"""Dated comparison evidence stays descriptive when outcomes are scarce.""" + +from __future__ import annotations + +import json +from datetime import UTC, date, datetime + +from stocksweeper.forecast.evidence_reports import ( + build_model_evidence, + ledger_band_rows, + load_replay_report, + save_replay_report, +) +from stocksweeper.forecast.ledger import ForecastLedger +from stocksweeper.forecast.calendar import SessionCalendar +from stocksweeper.forecast.physical_evaluation import evaluate_band + + +def test_unlabelled_ledger_reports_zero_and_no_significance(tmp_path): + evidence = build_model_evidence( + tmp_path, ForecastLedger(tmp_path), datetime(2026, 9, 27, tzinfo=UTC) + ) + assert len(evidence) == 15 + for (method, band), result in evidence.items(): + assert method and band + prospective = result["prospective"] + assert prospective["provenance"].startswith("as_issued") + assert prospective["ticker_origin_horizon_units"] == 0 + assert prospective["brier"]["candidate"] is None + assert prospective["brier"]["bootstrap_95"] is None + assert prospective["calibration_by_side"]["call"][0]["count"] == 0 + assert prospective["calibration_by_side"]["put"][0]["count"] == 0 + assert result["retrospective"] is None + + +def test_replay_report_is_explicitly_current_vintage_and_detects_tampering(tmp_path): + band = evaluate_band([], "student_t_ewma", "1", bootstrap_samples=100) + report = {"provenance": "immutable_replay", "bands": {"1": band}} + path = save_replay_report(tmp_path, "student_t_ewma", report) + loaded = load_replay_report(tmp_path, "student_t_ewma") + assert loaded["report"] == report + assert loaded["report_hash"] + assert loaded["schema_version"] == 2 + + payload = json.loads(path.read_text()) + payload["report"]["bands"]["1"]["tickers"] = 99 + path.write_text(json.dumps(payload)) + assert load_replay_report(tmp_path, "student_t_ewma") is None + + +def test_ledger_evidence_requires_mature_label_and_as_of_snapshot(): + issued = datetime(2026, 9, 25, 22, tzinfo=UTC) + cutoff = datetime(2026, 9, 28, 22, tzinfo=UTC) + kwargs_seen = {} + + class Ledger: + def iter_evaluation_rows(self, **kwargs): + kwargs_seen.update(kwargs) + yield { + "ticker": "TEST", "input_session": date(2026, 9, 25), + "expiration": date(2026, 9, 28), "expiry_session": date(2026, 9, 28), + "method": "student_t_ewma", "label_status": "valid", + "label_checked_at": datetime(2026, 9, 28, 19, tzinfo=UTC), + "issued_at": issued, "spot_exact": "99", "strike_exact": "100", + "side": "call", "status": "available", "itm_probability": 0.6, + "observed_itm": True, "volatility_regime": "unknown", + "known_event_status": "unknown", "data_hash": "a" * 64, + "idempotency_key": "one", "contract_key": "one", + "unavailable_reason": None, "label_reason": None, "crps": None, + "prepare_ms": None, "lookup_ms": None, + } + + rows = ledger_band_rows( + Ledger(), SessionCalendar(), "as_issued", "student_t_ewma", None, "all", "1", + as_of=cutoff, + ) + assert len(rows) == 1 + assert rows[0].observed_itm is None + assert kwargs_seen["issued_before"] == cutoff + assert kwargs_seen["label_as_of"] == cutoff diff --git a/backend/tests/test_evidence_selection_regressions.py b/backend/tests/test_evidence_selection_regressions.py new file mode 100644 index 0000000..ff3599e --- /dev/null +++ b/backend/tests/test_evidence_selection_regressions.py @@ -0,0 +1,243 @@ +"""Independence and provenance boundaries for displayed model comparisons.""" + +from __future__ import annotations + +from dataclasses import replace +from datetime import UTC, date, datetime, timedelta +from decimal import Decimal +from types import SimpleNamespace + +import pytest + +from options_api.contract_identity import make_watch_key +from options_api.intraday_capture import capture_intraday_window +from options_api.predictive_watch import PredictiveWatchOdds +from options_api.watchlist import OutcomeView, WatchItem, get_watchlist +from stocksweeper.forecast.calendar import SessionCalendar +from stocksweeper.forecast.calibration import build_calibration +from stocksweeper.forecast.evidence_reports import build_model_evidence +from stocksweeper.forecast.intraday_evidence import evaluate as evaluate_intraday +from stocksweeper.forecast.ledger import ForecastLedger +from stocksweeper.forecast.physical_evaluation import ContestRow, evaluate_band +from stocksweeper.forecast.predictive import BASELINE_VERSION, PredictiveForecaster + +from .test_intraday_capture import _fixture, _future_session +from .test_intraday_prospective_evaluation import _rows +from .test_predictive_watch import _CalibrationLedger, _calibration_rows +from .test_promotion import _Prices, _activate, _bars + + +def test_forced_baseline_and_watch_ignore_prepared_promoted_champion( + tmp_path, monkeypatch: pytest.MonkeyPatch +) -> None: + from stocksweeper.forecast.physical_contest import STUDENT_VERSION, ShadowForecast + from stocksweeper.forecast.physical_contest import PhysicalShadowForecaster + from stocksweeper.forecast.promotion import PromotionRegistry + + completed = date(2026, 9, 24) + now = datetime(2026, 9, 24, 23, tzinfo=UTC) + expiry = date(2026, 9, 25) + forecaster = PredictiveForecaster(tmp_path, _Prices(_bars())) + forecaster.prepare("AAPL", completed) + _activate(PromotionRegistry(tmp_path), monkeypatch) + + def promoted(_shadow, frozen, _method, *, clean): + assert clean is not None + return ShadowForecast( + replace(frozen, method="student_t_ewma", model_version=STUDENT_VERSION), + None, 0, 0, + ) + + monkeypatch.setattr(PhysicalShadowForecaster, "cached_candidate", promoted) + forecaster.prepare("AAPL", completed) + assert forecaster.forecast("AAPL", now, expiry).method == "student_t_ewma" + baseline = forecaster.forecast("AAPL", now, expiry, force_baseline=True) + assert (baseline.method, baseline.model_version) == ("lognormal_ewma", BASELINE_VERSION) + assert forecaster.forecast_baseline("AAPL", now, expiry) == baseline + + watch = PredictiveWatchOdds(tmp_path, lambda: now, forecaster, refresh_enabled=False) + view, issued = watch.lookup("AAPL", "call", expiry, Decimal("100")) + assert issued == baseline + assert (view.method, view.evidence_key) == ("lognormal_ewma", "lognormal_ewma:1") + + +def test_intraday_capture_keeps_ewma_as_its_dated_close_reference( + tmp_path, monkeypatch: pytest.MonkeyPatch +) -> None: + ledger, forecaster, market, (issue, _), _, now = _fixture(tmp_path, _future_session()) + monkeypatch.setattr( + forecaster, "_select_promoted", + lambda frozen: replace(frozen, method="student_t_ewma"), + ) + assert forecaster.forecast("TEST", now, issue.expiration).method == "student_t_ewma" + watch = PredictiveWatchOdds(tmp_path, lambda: now, forecaster, refresh_enabled=False) + baseline = watch.distribution("TEST", issue.expiration, contract_since=issue.input_session) + assert baseline.method == "lognormal_ewma" + call = baseline.probability("call", Decimal(issue.strike_exact)) + put = baseline.probability("put", Decimal(issue.strike_exact)) + assert call is not None and put is not None + issue = replace( + issue, method=baseline.method, model_version=baseline.model_version, + itm_probability=call, otm_probability=put, atm_probability=1 - call - put, + ) + assert ledger.record_batch([(issue, baseline)]) == 1 + assert capture_intraday_window( + ledger, forecaster, market, [(issue, baseline)], now=now, window="10:00" + ) == 2 + assert {row["method"] for row in ledger.evaluation_rows()} == { + "lognormal_ewma", "quote_reanchored_comparator", "intraday_shadow", + } + + +def test_same_side_strikes_do_not_inflate_calibration_threshold() -> None: + rows = _calibration_rows() + removed = rows.pop(0) + repeated = dict(rows[1]) # The next call shares a ticker, origin, and horizon. + strike = Decimal("101") + repeated["strike_exact"] = "101.000" + repeated["contract_key"] = make_watch_key( + repeated["ticker"], repeated["ticker"], "call", + repeated["expiration"].isoformat(), strike, + ) + repeated["idempotency_key"] = f"{repeated['contract_key']}:{repeated['input_session']}" + rows.append(repeated) + assert removed["side"] == repeated["side"] == "call" + assert sum(row["side"] == "call" for row in rows) == 500 + + evidence = build_calibration( + _CalibrationLedger(rows), datetime(2025, 5, 1, tzinfo=UTC) + ) + assert ("test-v1", "completed_close", "1", "near ATM", "call") not in evidence + assert evidence[("test-v1", "completed_close", "1", "near ATM", "put")][ + "independent_units" + ] == 500 + + +def test_physical_calibration_bin_counts_independent_unit_once() -> None: + origin = date(2026, 9, 28) + expiry = date(2026, 9, 29) + rows = [ + ContestRow( + ticker="TEST", origin=origin, expiry_session=expiry, horizon=1, + strike=strike, side="call", method=method, probability=probability, + observed_itm=True, provenance="as_issued", input_vintage="same-bars", + ) + for strike in ("100", "101") + for method, probability in (("lognormal_ewma", 0.5), ("student_t_ewma", 0.6)) + ] + report = evaluate_band(rows, "student_t_ewma", "1", bootstrap_samples=100) + assert report["contract_forecasts_available"] == 2 + assert report["ticker_origin_horizon_units"] == 1 + assert report["calibration_by_side"]["call"][6] == { + "lower": 0.6, "upper": 0.7, "count": 1, + "forecast_mean": 0.6, "observed_rate": 1.0, + } + + +def test_overlapping_expiries_do_not_create_intraday_significance(tmp_path) -> None: + calendar = SessionCalendar() + days = calendar.sessions(date(2026, 9, 28), date(2026, 10, 27))[:20] + rows = [ + row + for day in days + for row in _rows(day=day, expiry=calendar.offset(day, 4)) + ] + as_of = datetime.combine(max(row["expiration"] for row in rows), datetime.min.time(), UTC) + as_of += timedelta(hours=23) + summary = evaluate_intraday(rows, as_of=as_of)["overall"] + assert summary["scored_contract_windows"] == 20 + assert summary["scored_dates_before_overlap_purge"] == 20 + assert 1 < summary["scored_date_blocks"] < 20 + + class Ledger: + def iter_evaluation_rows(self, **kwargs): + if "intraday_shadow" not in kwargs["methods"]: + return iter(()) + lower, upper = kwargs["horizon_range"] + return ( + row for row in rows + if row["method"] in kwargs["methods"] + and lower <= calendar.horizon(row["input_session"], row["expiration"]) <= upper + ) + + def evaluation_skipped_attempts(self, _provenance): + return 0 + + def panel_coverage(self, **_kwargs): + return {} + + prospective = build_model_evidence(tmp_path, Ledger(), as_of)[ + ("intraday_shadow", "2-5") + ]["prospective"] + assert prospective["independent_date_blocks"] == summary["scored_date_blocks"] + assert prospective["significance"] == "not_estimable" + assert prospective["brier"]["bootstrap_95"] is None + + +@pytest.mark.asyncio +async def test_expired_watch_keeps_five_physical_two_market_and_selected_risk() -> None: + expiry = date(2026, 9, 25) + report = {"prospective": {"ticker_origin_horizon_units": 0}} + item = WatchItem( + id="a" * 32, ticker="TEST", root="TEST", side="call", + expiration=expiry, strike_exact="100.000", terms_note="standard 100-share terms", + created_at=datetime(2026, 9, 24, tzinfo=UTC), + outcome=OutcomeView(status="pending"), + ) + state = SimpleNamespace( + clock=lambda: datetime(2026, 9, 28, 21, tzinfo=UTC), + watchlist=SimpleNamespace(items=lambda *, as_of: [item]), + market_odds=SimpleNamespace(schedule=lambda _tickers: None), + predictive_odds=SimpleNamespace( + schedule=lambda _tickers: None, + evidence_index=lambda: {"student_t_ewma:1": report}, + ), + jobs=SimpleNamespace(active=lambda _kind: None), + ) + response = await get_watchlist( + SimpleNamespace(app=SimpleNamespace(state=state)), forecast_model="student_t_ewma" + ) + body = response.model_dump(mode="json") + saved = body["items"][0] + assert body["model_evidence"] == {"student_t_ewma:1": report} + assert [model["method"] for model in saved["physical_models"]] == [ + "lognormal_ewma", "empirical_scaled", "student_t_ewma", "gjr_garch_t", + "intraday_shadow", + ] + assert [model["method"] for model in saved["market_models"]] == [ + "regimelib", "constrained_call_curve", + ] + assert all(model["status"] == "unavailable" for model in saved["physical_models"]) + assert all(model["status"] == "unavailable" for model in saved["market_models"]) + assert all(model["model_evidence"] is None for model in saved["physical_models"]) + assert saved["predictive_odds"]["method"] == "student_t_ewma" + assert saved["hypothetical_risk"]["forecast_method"] == "student_t_ewma" + + +def test_shared_evidence_key_and_zero_sample_semantics(tmp_path) -> None: + watch = PredictiveWatchOdds( + tmp_path, lambda: datetime(2026, 9, 27, tzinfo=UTC), refresh_enabled=False + ) + evidence = build_model_evidence( + tmp_path, ForecastLedger(tmp_path), datetime(2026, 9, 27, tzinfo=UTC) + ) + watch._model_evidence = evidence + baseline = watch.forecaster.forecast_baseline( + "TEST", datetime(2026, 9, 27, tzinfo=UTC), date(2026, 9, 29) + ) + # An unavailable forecast has no usable horizon key; an available model does. + available = replace( + baseline, status="available", reason=None, method="lognormal_ewma", + as_of=date(2026, 9, 25), expiry_session=date(2026, 9, 29), horizon_sessions=2, + spot=100.0, daily_volatility=0.02, model_version=BASELINE_VERSION, + terminal_prices=(90.0, 110.0), weights=(0.5, 0.5), + ) + view = watch.view_for_distribution(available, "call", Decimal("100")) + key = view.evidence_key + assert key == "lognormal_ewma:2-5" + assert view.model_evidence is None + assert watch.evidence_index()[key] == evidence[("lognormal_ewma", "2-5")] + assert watch.evidence_index()[key]["prospective"]["ticker_origin_horizon_units"] == 0 + assert watch.evidence_index()["intraday_shadow:2-5"]["prospective"][ + "significance" + ] == "not_estimable" diff --git a/backend/tests/test_golden_pages.py b/backend/tests/test_golden_pages.py index e0d4ae2..681ab91 100644 --- a/backend/tests/test_golden_pages.py +++ b/backend/tests/test_golden_pages.py @@ -30,9 +30,11 @@ def _pages(assemble) -> dict[str, object]: for page in pages.values(): assert page.pop("chain_source") == "nasdaq" assert page.pop("chain_fetched_at") == page["fetched_at"] + assert page.pop("model_evidence") == {} for group in page["expirations"]: for contract in group["contracts"]: assert contract.pop("market_odds") == { + "method": None, "model_evidence": None, "status": "pending", "itm_pct_tenths": None, "otm_pct_tenths": None, "reason": None, "source": None, "fetched_at": None, "session_date": None, "model_version": None, @@ -40,6 +42,8 @@ def _pages(assemble) -> dict[str, object]: "quote_support_score": None, } assert contract.pop("predictive_odds")["status"] == "pending" + assert contract.pop("physical_models") == [] + assert contract.pop("market_models") == [] assert contract.pop("hypothetical_risk")["status"] == "unavailable" assert contract.pop("greeks_rate_pct_tenths") is None assert contract.pop("greeks_rate_as_of_session") is None diff --git a/backend/tests/test_intraday_prospective_evaluation.py b/backend/tests/test_intraday_prospective_evaluation.py index a5bec84..1e9764f 100644 --- a/backend/tests/test_intraday_prospective_evaluation.py +++ b/backend/tests/test_intraday_prospective_evaluation.py @@ -8,6 +8,7 @@ from scripts.evaluate_intraday_prospective import evaluate from options_api.market_calendar import session_on_or_before +from stocksweeper.forecast.calendar import SessionCalendar _NY = ZoneInfo("America/New_York") _DAY = date(2026, 9, 28) @@ -121,6 +122,8 @@ def test_scores_exact_triplets_once_and_averages_correlated_contracts() -> None: assert overall["paired_shadow_minus_reference"]["brier"]["dated_close"] == pytest.approx(-0.12) assert overall["paired_calendar_date_bootstrap_95"] is None assert overall["latency_ms"]["intraday_shadow"] == {"p50": 18.0, "p95": 18.0} + assert overall["calibration_by_side"]["call"][8]["observed_rate"] == 1 + assert overall["calibration_by_side"]["put"][2]["observed_rate"] == 0 assert report["by_window"]["10:00"]["scored_contract_windows"] == 2 assert report["by_window"]["13:00"]["recorded_contract_windows"] == 0 @@ -162,6 +165,29 @@ def test_idempotent_close_forecast_issued_before_window_still_pairs() -> None: assert report["overall"]["rejection_reasons"] == {} +def test_intraday_dated_close_comparator_is_ewma_even_with_other_physical_issues() -> None: + rows = _rows("MIXED") + other_model = rows[0].copy() + other_model.update( + idempotency_key="MIXED-gjr", method="gjr_garch_t", itm_probability=0.1, + issued_at=rows[0]["issued_at"] + timedelta(milliseconds=500), + ) + report = evaluate([*rows, other_model], as_of=_as_of(_EXPIRY)) + assert report["overall"]["mean"]["brier"]["dated_close"] == pytest.approx(0.16) + + +def test_repeated_forecasts_of_one_expiry_do_not_create_independent_date_blocks() -> None: + expiry = date(2026, 10, 30) + days = SessionCalendar().sessions(date(2026, 9, 28), date(2026, 10, 27))[:20] + rows = [item for day in days for item in _rows(day=day, expiry=expiry)] + report = evaluate(rows, as_of=_as_of(expiry))["overall"] + assert report["scored_contract_windows"] == 20 + assert report["scored_dates_before_overlap_purge"] == 20 + assert report["scored_date_blocks"] == 1 + assert report["scored_ticker_date_window_expiry_units"] == 1 + assert report["paired_calendar_date_bootstrap_95"] is None + + def test_same_day_early_close_and_holiday_strata_do_not_invent_scores() -> None: same = _rows("SAME", expiry=_DAY) same_report = evaluate(same, as_of=_as_of(_DAY)) diff --git a/backend/tests/test_live_quant.py b/backend/tests/test_live_quant.py index bfa5bdd..95eea72 100644 --- a/backend/tests/test_live_quant.py +++ b/backend/tests/test_live_quant.py @@ -104,6 +104,9 @@ def lookup(self, _ticker: str, side: str, _expiry: date, _strike: Decimal, **kwa def schedule(self, tickers: object) -> None: self.scheduled = list(tickers) + def evidence_index(self) -> dict[str, object]: + return {} + def _quote( *, diff --git a/backend/tests/test_market_curve_shadow.py b/backend/tests/test_market_curve_shadow.py index 83c1f50..69a8bfa 100644 --- a/backend/tests/test_market_curve_shadow.py +++ b/backend/tests/test_market_curve_shadow.py @@ -72,6 +72,7 @@ def test_shadow_curve_rejects_sparse_or_inconsistent_strip() -> None: assert sparse.odds.get((EXPIRY.isoformat(), Decimal(100)), None) is None or ( sparse.odds[(EXPIRY.isoformat(), Decimal(100))].call_itm_probability is None ) + assert sparse.expiry_reasons[EXPIRY.isoformat()] == "sparse_or_large_strip" rows = _quotes(range(93, 108)) rows[7] = rows[7].model_copy(update={"call_bid": Decimal("50"), "call_ask": Decimal("50.04")}) contradictory = calculate_curve_shadow( diff --git a/backend/tests/test_model_comparison.py b/backend/tests/test_model_comparison.py new file mode 100644 index 0000000..2528166 --- /dev/null +++ b/backend/tests/test_model_comparison.py @@ -0,0 +1,110 @@ +"""Live model selection reads only matching prepared snapshots.""" + +from __future__ import annotations + +from dataclasses import replace +from datetime import UTC, date, datetime +from decimal import Decimal + +from options_api.market_curve_shadow import CurveShadowResult +from options_api.market_odds import OddsEstimate +from options_api.market_watch import MarketWatchOdds, _Snapshot +from options_api.physical_shadow_capture import PhysicalShadowCapture +from options_api.predictive_watch import PredictiveWatchOdds +from stocksweeper.forecast.physical_contest import ShadowForecast +from stocksweeper.forecast.predictive import PredictiveDistribution + + +def test_candidate_cache_rejects_revised_input_and_unsupported_horizon(tmp_path) -> None: + predictive = PredictiveWatchOdds(tmp_path, lambda: datetime.now(UTC), refresh_enabled=False) + capture = PhysicalShadowCapture(predictive) + origin = date(2026, 9, 25) + expiry = date(2026, 10, 2) + baseline = PredictiveDistribution( + ticker="IREN", status="available", reason=None, + method="lognormal_ewma", as_of=origin, expiry_session=expiry, + horizon_sessions=5, spot=100.0, daily_volatility=0.02, + model_version="baseline-v1", support=60, data_hash="original", + terminal_prices=(90.0, 110.0), weights=(0.5, 0.5), + ) + challenger = replace(baseline, method="student_t_ewma", model_version="student-v1") + capture._live[("IREN", origin, expiry, origin, "original")] = ( + {"student_t_ewma": ShadowForecast(challenger, None, 0, 0)}, None, + ) + + assert capture.candidate(baseline, "student_t_ewma", expiry, origin).distribution == challenger + revised = replace(baseline, data_hash="revised") + assert capture.candidate(revised, "student_t_ewma", expiry, origin).reason == ( + "candidate_not_prepared" + ) + assert capture.candidate(revised, "lognormal_ewma", expiry, origin).distribution == revised + long_horizon = replace(baseline, horizon_sessions=26) + assert capture.candidate(long_horizon, "student_t_ewma", expiry, origin).reason == ( + "candidate_horizon_unsupported" + ) + + +def test_curve_result_is_hidden_after_quote_snapshot_rollover() -> None: + now = datetime(2026, 9, 25, 18, tzinfo=UTC) + expiry = "2026-10-02" + strike = Decimal("100") + first = _Snapshot( + now, date(2026, 9, 25), "nasdaq", {}, {}, + valid_contracts={(expiry, strike)}, + ) + odds = MarketWatchOdds.__new__(MarketWatchOdds) + odds.clock = lambda: now + odds._valid = lambda _snapshot, _now: True + odds._cache = {"IREN": first} + odds._curve_shadow = {} + odds._curve_results = { + "IREN": ( + first, + CurveShadowResult( + {(expiry, strike): OddsEstimate(0.4, bounds=(0.3, 0.5))}, + 1, 1, 5.0, {}, + ), + ) + } + available = odds.lookup_curve("IREN", "call", expiry, strike, "IREN") + assert available.status == "available" + assert (available.itm_pct_tenths, available.otm_pct_tenths) == (400, 600) + + odds._cache["IREN"] = replace(first) + stale = odds.lookup_curve("IREN", "call", expiry, strike, "IREN") + assert stale.status == "pending" + assert stale.itm_pct_tenths is None + + +def test_curve_result_preserves_sparse_and_failed_fit_reasons() -> None: + now = datetime(2026, 9, 25, 18, tzinfo=UTC) + expiry = "2026-10-02" + strike = Decimal("100") + snapshot = _Snapshot( + now, date(2026, 9, 25), "nasdaq", {}, {}, + valid_contracts={(expiry, strike)}, + ) + odds = MarketWatchOdds.__new__(MarketWatchOdds) + odds.clock = lambda: now + odds._valid = lambda _snapshot, _now: True + odds._cache = {"IREN": snapshot} + odds._curve_shadow = {} + odds._curve_results = { + "IREN": ( + snapshot, + CurveShadowResult( + {}, 0, 0, 1.0, {"sparse_or_large_strip": 3}, + expiry_reasons={expiry: "sparse_or_large_strip"}, + ), + ), + } + sparse = odds.lookup_curve("IREN", "call", expiry, strike, "IREN") + assert sparse.status == "unavailable" + assert sparse.reason == "sparse_or_large_strip" + assert sparse.model_evidence["bid_ask_fit"] is None + + odds._curve_results["IREN"] = ( + snapshot, CurveShadowResult({}, 0, 0, 0, {"curve_fit_failed": 1}) + ) + failed = odds.lookup_curve("IREN", "call", expiry, strike, "IREN") + assert failed.reason == "curve_fit_failed" diff --git a/backend/tests/test_physical_capture.py b/backend/tests/test_physical_capture.py index 6d63bf1..7da1b06 100644 --- a/backend/tests/test_physical_capture.py +++ b/backend/tests/test_physical_capture.py @@ -123,6 +123,21 @@ def forecast(*args, **kwargs): assert empirical["atm_probability"] == pytest.approx(0.5) +def test_failed_shadow_issuance_cannot_enter_live_cache(tmp_path) -> None: + capture = _capture(tmp_path) + capture.forecaster = SimpleNamespace( + forecast_candidates=lambda *_args, **_kwargs: _candidates() + ) + + def failed_record(_entries): + raise OSError("ledger unavailable") + + capture.predictive.ledger.record_batch = failed_record + with pytest.raises(OSError, match="ledger unavailable"): + capture._capture_sync([(_issue(), _distribution())]) + assert not capture._live + + def test_as_issued_report_includes_persisted_preparation_and_lookup_latency(tmp_path) -> None: capture = _capture(tmp_path) candidates = _candidates() @@ -176,6 +191,71 @@ def test_one_challenger_rejection_does_not_erase_available_baseline(tmp_path) -> assert rows["gjr_garch_t"]["unavailable_reason"] == "gjr_nonconverged" +def test_shadow_failure_cannot_mask_valid_live_method(tmp_path) -> None: + capture = _capture(tmp_path) + base = _distribution("empirical_scaled") + key = base.ticker, base.as_of, EXPIRY, INPUT, base.data_hash + capture._live[key] = ( + {"empirical_scaled": ShadowForecast(None, "market_data_missing", 0, 0)}, + None, + ) + + result = capture.candidate(base, "empirical_scaled", EXPIRY, INPUT) + assert result.distribution is base + assert result.reason is None + assert capture.candidate(base, "student_t_ewma", EXPIRY, INPUT).reason == ( + "candidate_not_prepared" + ) + + +@pytest.mark.parametrize("changed", ("hash", "session", "expiry", "contract_since")) +def test_stale_shadow_candidate_is_not_reused_for_new_input(tmp_path, changed: str) -> None: + capture = _capture(tmp_path) + old = _distribution("empirical_scaled") + key = old.ticker, old.as_of, EXPIRY, INPUT, old.data_hash + capture._live[key] = ( + {"student_t_ewma": ShadowForecast(_distribution("student_t_ewma"), None, 0, 0)}, + None, + ) + + current = old + expiry = EXPIRY + contract_since = INPUT + if changed == "hash": + current = replace(old, data_hash="b" * 64) + elif changed == "session": + current = replace(old, as_of=date(2026, 9, 26)) + elif changed == "expiry": + expiry = date(2026, 10, 9) + current = replace(old, expiry_session=expiry) + else: + contract_since = date(2026, 9, 24) + + assert ( + capture.candidate(current, "empirical_scaled", expiry, contract_since).distribution + is current + ) + shadow = capture.candidate(current, "student_t_ewma", expiry, contract_since) + assert shadow.distribution is None + assert shadow.reason == "candidate_not_prepared" + + +@pytest.mark.parametrize( + ("base", "reason"), + ( + (replace(_distribution(), status="unavailable", reason="source_outage"), "source_outage"), + (replace(_distribution(), data_hash=None), "completed_close_forecast_unavailable"), + ), +) +def test_invalid_base_cannot_bypass_availability_guard( + tmp_path, base: PredictiveDistribution, reason: str +) -> None: + capture = _capture(tmp_path) + result = capture.candidate(base, base.method, EXPIRY, INPUT) + assert result.distribution is None + assert result.reason == reason + + def test_unavailable_first_contract_does_not_suppress_valid_peer(tmp_path) -> None: capture = _capture(tmp_path) capture.forecaster = SimpleNamespace( @@ -213,6 +293,9 @@ def test_manifest_mismatch_records_four_unavailable_attempts_without_fitting(tmp assert all(row["status"] == "unavailable" for row in rows) assert all(row["unavailable_reason"] == "input_provenance_unverified" for row in rows) assert all(row["itm_probability"] is None and row["distribution_hash"] is None for row in rows) + assert capture.candidate( + _distribution("lognormal_ewma"), "student_t_ewma", EXPIRY, INPUT + ).reason == "input_provenance_unverified" def test_changed_input_vintage_is_rejected_even_if_fit_succeeds(tmp_path) -> None: @@ -241,6 +324,16 @@ def broken_fit(*_args, **_kwargs): assert {row["method"] for row in rows} == METHODS assert all(row["status"] == "unavailable" for row in rows) assert all(row["unavailable_reason"] == "shadow_fit_failed" for row in rows) + assert capture.candidate( + _distribution("lognormal_ewma"), "student_t_ewma", EXPIRY, INPUT + ).reason == "shadow_fit_failed" + capture.forecaster = SimpleNamespace( + forecast_candidates=lambda *_args, **_kwargs: _candidates() + ) + capture._capture_sync([(_issue(), _distribution())]) + assert capture.candidate( + _distribution("lognormal_ewma"), "student_t_ewma", EXPIRY, INPUT + ).distribution is not None def test_restart_does_not_duplicate_the_same_shadow_issuances(tmp_path) -> None: @@ -265,12 +358,11 @@ def test_contracts_over_fit_capacity_leave_recorded_unavailable_attempts( capture.forecaster = SimpleNamespace( forecast_candidates=lambda *_args, **_kwargs: _candidates() ) - capture._capture_sync( - [ - (_issue("100.000"), _distribution()), - (_issue("110.000"), _distribution()), - ] - ) + contracts = [ + (_issue("100.000"), _distribution()), + (replace(_issue("110.000"), contract_since=date(2026, 9, 24)), _distribution()), + ] + capture._capture_sync(contracts) rows = ForecastLedger(tmp_path).evaluation_rows() assert len(rows) == 8 @@ -282,6 +374,13 @@ def test_contracts_over_fit_capacity_leave_recorded_unavailable_attempts( assert ( len([row for row in rows if row["unavailable_reason"] == "shadow_capacity_exceeded"]) == 4 ) + skipped = max( + (issue for issue, _ in contracts), + key=lambda issue: hashlib.sha256(issue.contract_key.encode()).digest(), + ) + assert capture.candidate( + _distribution("lognormal_ewma"), "student_t_ewma", EXPIRY, skipped.contract_since + ).reason == "shadow_capacity_exceeded" @pytest.mark.asyncio @@ -292,7 +391,7 @@ async def test_full_pending_queue_retains_failed_candidate_attempts( capture = _capture(tmp_path) gate = asyncio.Event() - async def blocked(_entries): + async def blocked(_entries, _quotes): await gate.wait() capture._capture = blocked @@ -320,3 +419,80 @@ def record_off_loop(entries): assert {row["method"] for row in rejected} == METHODS assert all(row["status"] == "unavailable" for row in rejected) assert all(row["unavailable_reason"] == "shadow_capacity_exceeded" for row in rejected) + assert capture.candidate( + _distribution("lognormal_ewma"), "student_t_ewma", EXPIRY, INPUT + ).reason == "shadow_capacity_exceeded" + + +@pytest.mark.asyncio +async def test_capacity_rejection_can_refit_same_snapshot_after_bounded_retry( + tmp_path, monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.setattr("options_api.physical_shadow_capture._MAX_PENDING_BATCHES", 1) + clock = [0.0] + monkeypatch.setattr("options_api.physical_shadow_capture.monotonic", lambda: clock[0]) + capture = _capture(tmp_path) + calls: list[str] = [] + + async def record_capacity(_entries, _quotes): + calls.append("capacity") + + async def fit(_entries, _quotes): + calls.append("fit") + + capture._record_capacity_async = record_capacity + capture._capture = fit + placeholder = asyncio.create_task(asyncio.sleep(0)) + capture._fit_tasks.add(placeholder) + entries = [(_issue(), _distribution())] + capture.submit(entries) + await asyncio.sleep(0) + capture._fit_tasks.discard(placeholder) + capture.submit(entries) + await asyncio.sleep(0) + assert calls == ["capacity"] + clock[0] = 31.0 + capture.submit(entries) + await capture.close() + assert calls == ["capacity", "fit"] + + +@pytest.mark.asyncio +async def test_failed_submit_retries_then_success_dedupes_same_snapshot( + tmp_path, monkeypatch: pytest.MonkeyPatch +) -> None: + clock = [0.0] + monkeypatch.setattr("options_api.physical_shadow_capture.monotonic", lambda: clock[0]) + capture = _capture(tmp_path) + calls = [0] + + def fit(*_args, **_kwargs): + calls[0] += 1 + if calls[0] == 1: + raise RuntimeError("temporary optimizer failure") + return _candidates() + + capture.forecaster = SimpleNamespace(forecast_candidates=fit) + entries = [(_issue(), _distribution("lognormal_ewma"))] + capture.submit(entries) + await asyncio.gather(*capture._tasks) + await asyncio.sleep(0) + assert calls[0] == 1 + assert capture.candidate( + _distribution("lognormal_ewma"), "student_t_ewma", EXPIRY, INPUT + ).reason == "shadow_fit_failed" + + clock[0] = 31.0 + capture.submit(entries) + await asyncio.gather(*capture._tasks) + await asyncio.sleep(0) + assert calls[0] == 2 + assert capture.candidate( + _distribution("lognormal_ewma"), "student_t_ewma", EXPIRY, INPUT + ).distribution is not None + + clock[0] = 62.0 + capture.submit(entries) + await asyncio.sleep(0) + assert calls[0] == 2 + await capture.close() diff --git a/backend/tests/test_physical_contest.py b/backend/tests/test_physical_contest.py index 10c1415..e6306db 100644 --- a/backend/tests/test_physical_contest.py +++ b/backend/tests/test_physical_contest.py @@ -13,6 +13,7 @@ import pytest from stocksweeper.forecast.calendar import SessionCalendar +from stocksweeper.forecast.evidence_reports import ledger_contest from stocksweeper.forecast.audit import AuditCohort, AuditMember from stocksweeper.forecast.physical_contest import ( PhysicalShadowForecaster, @@ -213,6 +214,12 @@ def test_paired_band_gate_uses_ticker_origin_units_and_as_issued_evidence(): assert report["brier"]["bootstrap_95"][1] < 0 assert report["crps"]["scored_units"] == 500 assert report["promotion_eligible"] is True + call_bins = report["calibration_by_side"]["call"] + put_bins = report["calibration_by_side"]["put"] + assert sum(item["count"] for item in call_bins) == 500 + assert sum(item["count"] for item in put_bins) == 500 + assert next(item for item in call_bins if item["count"])["observed_rate"] == 1 + assert next(item for item in put_bins if item["count"])["observed_rate"] == 0 replay = evaluate_band( _paired_rows("immutable_replay"), "student_t_ewma", @@ -319,11 +326,6 @@ def fake_candidates(self, ticker, when, expiry): def test_ledger_contest_uses_exact_joined_label_and_counts_failed_attempts(): - script = Path(__file__).parents[1] / "scripts" / "evaluate_predictive.py" - spec = spec_from_file_location("evaluate_predictive_ledger_script", script) - assert spec is not None and spec.loader is not None - evaluator = module_from_spec(spec) - spec.loader.exec_module(evaluator) issued = datetime(2026, 9, 25, 22, tzinfo=UTC) common = { "ticker": "AAPL", @@ -380,7 +382,7 @@ def panel_coverage(self, **_kwargs): def evaluation_skipped_attempts(self, provenance): return {} - report = evaluator._ledger_contest( + report = ledger_contest( Ledger(), SessionCalendar(), "as_issued", "student_t_ewma", None, "all" ) first = report["bands"]["1"] @@ -410,7 +412,7 @@ def evaluation_rows(self, *, provenance): }, ] - late_report = evaluator._ledger_contest( + late_report = ledger_contest( LateLedger(), SessionCalendar(), "as_issued", "student_t_ewma", None, "all" ) assert late_report["bands"]["1"]["ticker_origin_horizon_units"] == 0 diff --git a/backend/tests/test_predictive_watch.py b/backend/tests/test_predictive_watch.py index 46a96d6..8295ca2 100644 --- a/backend/tests/test_predictive_watch.py +++ b/backend/tests/test_predictive_watch.py @@ -309,7 +309,7 @@ def _calibration_rows() -> list[dict[str, object]]: return rows -def test_calibration_requires_500_units_and_averages_shared_call_put_close() -> None: +def test_calibration_requires_500_units_and_keeps_call_put_separate() -> None: rows = _calibration_rows() as_of = datetime(2025, 5, 1, tzinfo=UTC) ledger = _CalibrationLedger(rows[:-2]) @@ -318,11 +318,14 @@ def test_calibration_requires_500_units_and_averages_shared_call_put_close() -> ledger.rows = rows summary = build_calibration(ledger, as_of) - evidence = summary[("test-v1", "completed_close", "1", "near ATM")] - assert evidence["independent_units"] == 500 - assert evidence["predicted_itm_pct_tenths"] == 500 - assert evidence["observed_itm_pct_tenths"] == 500 - assert evidence["through_session"] == rows[-1]["expiry_session"] + call = summary[("test-v1", "completed_close", "1", "near ATM", "call")] + put = summary[("test-v1", "completed_close", "1", "near ATM", "put")] + assert call["independent_units"] == put["independent_units"] == 500 + assert call["predicted_itm_pct_tenths"] == 600 + assert put["predicted_itm_pct_tenths"] == 400 + assert call["observed_itm_pct_tenths"] == 1000 + assert put["observed_itm_pct_tenths"] == 0 + assert call["through_session"] == put["through_session"] == rows[-1]["expiry_session"] assert ledger.reads == 2 revised = dict(rows[0]) diff --git a/backend/tests/test_quant_api.py b/backend/tests/test_quant_api.py index 994b889..f90bf9b 100644 --- a/backend/tests/test_quant_api.py +++ b/backend/tests/test_quant_api.py @@ -89,14 +89,23 @@ async def load(*_args, **_kwargs): distribution, ), ) - response = client.get("/api/covered-calls/IREN?moneyness=all") + response = client.get( + "/api/covered-calls/IREN?moneyness=all&forecast_model=empirical_scaled" + ) + default_response = client.get("/api/covered-calls/IREN?moneyness=all") + invalid_response = client.get("/api/covered-calls/IREN?forecast_model=unknown") page = assemble_covered_calls(chain, info, history, today, now, "all") def failed_evidence(_entries): raise OSError("evidence disk unavailable") monkeypatch.setattr(app.state.predictive_odds.ledger, "record_batch", failed_evidence) - degraded = client.get("/api/covered-calls/IREN?moneyness=all") + degraded = client.get( + "/api/covered-calls/IREN?moneyness=all&forecast_model=empirical_scaled" + ) + selected_unavailable = client.get( + "/api/covered-calls/IREN?moneyness=all&forecast_model=intraday_shadow" + ) assert response.status_code == 200 body = response.json() @@ -108,12 +117,29 @@ def failed_evidence(_entries): ) assert row["market_odds"]["status"] == "unavailable" assert row["predictive_odds"]["status"] == "available" + assert row["predictive_odds"]["method"] == "empirical_scaled" + assert [view["method"] for view in row["physical_models"]] == [ + "lognormal_ewma", "empirical_scaled", "student_t_ewma", + "gjr_garch_t", "intraday_shadow", + ] + assert [view["method"] for view in row["market_models"]] == [ + "regimelib", "constrained_call_curve", + ] + assert row["hypothetical_risk"]["forecast_method"] == "empirical_scaled" assert row["predictive_odds"]["model_version"] == "test-v1" assert row["hypothetical_risk"]["status"] == "available" assert row["hypothetical_risk"]["assumed_spot_cents"] == 5010 assert row["hypothetical_risk"]["assumed_bid_cents"] == int(quoted.call_bid * 100) assert row["hypothetical_risk"]["quote_session"] == today.isoformat() assert body["chain_fetched_at"] == now.isoformat().replace("+00:00", "Z") + default_row = next( + contract for group in default_response.json()["expirations"] + for contract in group["contracts"] if contract["watch_key"] is not None + ) + assert default_row["predictive_odds"]["method"] == "lognormal_ewma" + assert default_row["predictive_odds"]["status"] == "pending" + assert default_row["hypothetical_risk"]["status"] == "unavailable" + assert invalid_response.status_code == 422 assert degraded.status_code == 200 degraded_row = next( contract @@ -125,3 +151,11 @@ def failed_evidence(_entries): assert degraded_row["predictive_odds"]["status"] == "unavailable" assert degraded_row["predictive_odds"]["reason"] == "Forecast evidence unavailable" assert degraded_row["hypothetical_risk"]["status"] == "unavailable" + unavailable_row = next( + contract for group in selected_unavailable.json()["expirations"] + for contract in group["contracts"] if contract["watch_key"] is not None + ) + assert unavailable_row["predictive_odds"]["status"] == "unavailable" + assert all( + view["status"] == "unavailable" for view in unavailable_row["physical_models"] + ) diff --git a/backend/tests/test_repository_workflows.py b/backend/tests/test_repository_workflows.py index 7d7e78d..b83905a 100644 --- a/backend/tests/test_repository_workflows.py +++ b/backend/tests/test_repository_workflows.py @@ -27,12 +27,30 @@ def _executable(path: Path, content: str) -> None: path.chmod(0o755) +def _fake_dev_git(fake_bin: Path) -> None: + _executable( + fake_bin / "git", + "#!/bin/sh\n" + '[ "$1" != -C ] || shift 2\n' + 'case "$1 $2" in\n' + ' "rev-parse --show-toplevel") pwd ;;\n' + ' "rev-parse HEAD"|"rev-parse refs/remotes/origin/main") ' + 'printf "%040d\\n" 1 ;;\n' + ' "symbolic-ref --short") printf "main\\n" ;;\n' + ' "remote get-url") printf "https://github.com/hypertrial/hyperoptions.git\\n" ;;\n' + ' "status --porcelain"|"fetch --no-tags"|"merge-base --is-ancestor") exit 0 ;;\n' + ' *) exit 1 ;;\n' + 'esac\n', + ) + + def test_dev_repairs_an_existing_incomplete_node_modules(tmp_path: Path) -> None: shutil.copytree(ROOT / "scripts", tmp_path / "scripts") (tmp_path / "backend").mkdir() (tmp_path / "frontend" / "node_modules").mkdir(parents=True) fake_bin = tmp_path / "bin" fake_bin.mkdir() + _fake_dev_git(fake_bin) log = tmp_path / "calls.log" _executable( fake_bin / "npm", @@ -61,8 +79,8 @@ def test_dev_repairs_an_existing_incomplete_node_modules(tmp_path: Path) -> None ) calls = log.read_text().splitlines() - install = f"{tmp_path / 'frontend'}|npm install" - backend_sync = f"{tmp_path / 'backend'}|uv sync --group dev --group research" + install = f"{tmp_path / 'frontend'}|npm ci" + backend_sync = f"{tmp_path / 'backend'}|uv sync --frozen --group dev --group research" backend_start = ( f"{tmp_path / 'backend'}|uv run --group research uvicorn options_api.main:app " "--reload --no-access-log --host 127.0.0.1 --port 8000" @@ -78,6 +96,7 @@ def test_dev_cleanup_does_not_kill_a_later_port_owner(tmp_path: Path) -> None: (tmp_path / "frontend").mkdir() fake_bin = tmp_path / "bin" fake_bin.mkdir() + _fake_dev_git(fake_bin) lsof_count = tmp_path / "lsof-count" unrelated = subprocess.Popen(["sleep", "30"]) try: @@ -126,6 +145,7 @@ def test_dev_cleanup_signals_owned_descendants(tmp_path: Path) -> None: (tmp_path / "frontend").mkdir() fake_bin = tmp_path / "bin" fake_bin.mkdir() + _fake_dev_git(fake_bin) term_log = tmp_path / "terminated.log" child = tmp_path / "child.py" child.write_text( @@ -156,7 +176,7 @@ def test_dev_cleanup_signals_owned_descendants(tmp_path: Path) -> None: ) _executable( fake_bin / "npm", - '#!/bin/sh\n[ "${1:-}" = install ] && exit 0\nexec "$TEST_SERVICE" frontend\n', + '#!/bin/sh\n[ "${1:-}" = ci ] && exit 0\nexec "$TEST_SERVICE" frontend\n', ) _executable(fake_bin / "curl", "#!/bin/sh\nexit 0\n") _executable(fake_bin / "lsof", "#!/bin/sh\nexit 1\n") @@ -209,6 +229,7 @@ def test_dev_cleanup_stops_a_launcher_before_its_process_group_exists_even_if_te (tmp_path / "frontend").mkdir() fake_bin = tmp_path / "bin" fake_bin.mkdir() + _fake_dev_git(fake_bin) launcher_pid = tmp_path / "launcher.pid" term_log = tmp_path / "terminated.log" launcher = tmp_path / "launcher.py" @@ -231,6 +252,7 @@ def test_dev_cleanup_stops_a_launcher_before_its_process_group_exists_even_if_te _executable( fake_bin / "python3", "#!/bin/sh\n" + '[ "$1" != "$TEST_DEV_UPDATE" ] || exec "$TEST_REAL_PYTHON" "$@"\n' 'exec "$TEST_REAL_PYTHON" "$TEST_LAUNCHER" ' '"$TEST_LAUNCHER_PID" "$TEST_TERM_LOG"\n', ) @@ -239,6 +261,7 @@ def test_dev_cleanup_stops_a_launcher_before_its_process_group_exists_even_if_te "PATH": f"{fake_bin}:{os.environ['PATH']}", "TEST_LAUNCHER": str(launcher), "TEST_LAUNCHER_PID": str(launcher_pid), + "TEST_DEV_UPDATE": str(tmp_path / "scripts" / "dev_update.py"), "TEST_REAL_PYTHON": sys.executable, "TEST_TERM_LOG": str(term_log), } diff --git a/backend/tests/test_version.py b/backend/tests/test_version.py new file mode 100644 index 0000000..e1d336d --- /dev/null +++ b/backend/tests/test_version.py @@ -0,0 +1,72 @@ +from __future__ import annotations + +from fastapi import FastAPI +from fastapi.testclient import TestClient + +from options_api import version + + +def test_version_reports_remote_change_and_frontend_mismatch(monkeypatch) -> None: + running = "a" * 40 + remote = "b" * 40 + calls: list[tuple[str, ...]] = [] + + def fake_git(*args: str, timeout: int = 5) -> str | None: + calls.append(args) + if args[:2] == ("remote", "get-url"): + return "https://github.com/hypertrial/hyperoptions.git" + if args[0] == "ls-remote": + return f"{remote}\trefs/heads/main" + raise AssertionError(args) + + monkeypatch.setattr(version, "_running_sha", running) + monkeypatch.setattr(version, "_branch", "main") + monkeypatch.setattr(version, "_cached_status", None) + monkeypatch.setattr(version, "_git", fake_git) + first = version.get_version_status("c" * 40) + second = version.get_version_status(running) + assert first.status == "update_available" + assert first.remote_sha == remote + assert first.frontend_matches is False + assert second.frontend_matches is True + assert calls.count(("ls-remote", "origin", "refs/heads/main")) == 1 + + +def test_version_preserves_offline_and_unverified_distinction(monkeypatch) -> None: + monkeypatch.setattr(version, "_running_sha", "a" * 40) + monkeypatch.setattr(version, "_branch", "main") + monkeypatch.setattr(version, "_cached_status", None) + monkeypatch.setattr( + version, + "_git", + lambda *args, **kwargs: ( + "https://github.com/hypertrial/hyperoptions.git" + if args[:2] == ("remote", "get-url") + else None + ), + ) + assert version.get_version_status().status == "offline" + + monkeypatch.setattr(version, "_cached_status", None) + monkeypatch.setattr(version, "_branch", "codex/feature") + assert version.get_version_status().status == "unverified_checkout" + + +def test_version_route_validates_frontend_sha(monkeypatch) -> None: + monkeypatch.setattr(version, "_running_sha", "a" * 40) + monkeypatch.setattr(version, "_branch", "main") + monkeypatch.setattr(version, "_cached_status", None) + monkeypatch.setattr( + version, + "_git", + lambda *args, **kwargs: ( + "https://github.com/hypertrial/hyperoptions.git" + if args[:2] == ("remote", "get-url") + else f"{'a' * 40}\trefs/heads/main" + ), + ) + app = FastAPI() + app.include_router(version.router) + client = TestClient(app) + assert client.get(f"/api/version?frontend_sha={'a' * 40}").json()["status"] == "current" + assert client.get("/api/version?frontend_sha=bad").status_code == 422 diff --git a/backend/tests/test_watchlist.py b/backend/tests/test_watchlist.py index 83de6e3..a1e6755 100644 --- a/backend/tests/test_watchlist.py +++ b/backend/tests/test_watchlist.py @@ -113,7 +113,7 @@ def prepared(_ticker, _expiry, *, contract_since, standard_terms): monkeypatch.setattr(predictive, "distribution", prepared) monkeypatch.setattr(app.state.physical_shadow, "submit", lambda _entries: None) - response = client.get("/api/watchlist") + response = client.get("/api/watchlist?forecast_model=empirical_scaled") assert response.status_code == 200 by_root = {item["root"]: item for item in response.json()["items"]} assert by_root["IREN"]["predictive_odds"]["status"] == "available" @@ -126,7 +126,7 @@ def prepared(_ticker, _expiry, *, contract_since, standard_terms): "IREN1": "unavailable", } predictive._pending["IREN"] = None - refreshing = client.get("/api/watchlist") + refreshing = client.get("/api/watchlist?forecast_model=empirical_scaled") assert refreshing.status_code == 200 by_root = {item["root"]: item for item in refreshing.json()["items"]} assert by_root["IREN"]["predictive_odds"]["status"] == "available" diff --git a/backend/uv.lock b/backend/uv.lock index 4d1f73c..abc4b60 100644 --- a/backend/uv.lock +++ b/backend/uv.lock @@ -492,6 +492,7 @@ name = "options-api" version = "0.1.0" source = { editable = "." } dependencies = [ + { name = "arch" }, { name = "duckdb" }, { name = "exchange-calendars" }, { name = "fastapi" }, @@ -514,12 +515,11 @@ dev = [ { name = "pytest-asyncio" }, { name = "ruff" }, ] -research = [ - { name = "arch" }, -] +research = [] [package.metadata] requires-dist = [ + { name = "arch", specifier = ">=8,<9" }, { name = "duckdb", specifier = ">=1.2" }, { name = "exchange-calendars", specifier = ">=4.11,<5" }, { name = "fastapi", specifier = ">=0.115" }, @@ -542,7 +542,7 @@ dev = [ { name = "pytest-asyncio", specifier = ">=0.25" }, { name = "ruff", specifier = ">=0.15" }, ] -research = [{ name = "arch", specifier = ">=8,<9" }] +research = [] [[package]] name = "packaging" diff --git a/frontend/e2e/chain.spec.ts b/frontend/e2e/chain.spec.ts index 14ac669..41954ed 100644 --- a/frontend/e2e/chain.spec.ts +++ b/frontend/e2e/chain.spec.ts @@ -9,7 +9,7 @@ function watchNasdaq(page: import("@playwright/test").Page) { } async function chooseTicker(page: import("@playwright/test").Page, symbol: string) { - const picker = page.getByRole("combobox") + const picker = page.getByRole("combobox", { name: "Ticker" }) await picker.click() await picker.fill(symbol) const option = page.getByRole("option", { name: new RegExp(symbol) }) @@ -188,6 +188,39 @@ test("keeps unavailable odds reasons accessible without filling every compact ro await expect(row.locator(".mobile-row-details")).toContainText("A coherent underlying bid and ask is unavailable") }) +test("compares every model on a narrow screen without nesting the disclosure trigger", async ({ page }) => { + await page.setViewportSize({ width: 390, height: 844 }) + await page.route("**/api/covered-calls/IREN**", async (route) => { + const response = await route.fetch() + const chain = await response.json() + const contract = chain.expirations[0].contracts[0] + contract.predictive_odds = { status: "available", method: "student_t_ewma", itm_pct_tenths: 611, otm_pct_tenths: 389, atm_pct_tenths: 0, evidence_key: "student_t_ewma:2-5" } + contract.physical_models = [ + { status: "available", method: "lognormal_ewma", itm_pct_tenths: 600, otm_pct_tenths: 400, atm_pct_tenths: 0 }, + { status: "available", method: "empirical_scaled", itm_pct_tenths: 620, otm_pct_tenths: 380, atm_pct_tenths: 0 }, + { status: "available", method: "student_t_ewma", itm_pct_tenths: 611, otm_pct_tenths: 389, atm_pct_tenths: 0, evidence_key: "student_t_ewma:2-5" }, + { status: "pending", method: "gjr_garch_t", reason: "candidate_not_prepared" }, + { status: "unavailable", method: "intraday_shadow", reason: "stale_quote" }, + ] + contract.market_models = [ + { status: "available", method: "regimelib", itm_pct_tenths: 580, otm_pct_tenths: 420 }, + { status: "unavailable", method: "constrained_call_curve", reason: "sparse_strikes" }, + ] + chain.model_evidence = { "student_t_ewma:2-5": { prospective: { generated_at: "2026-09-27T12:00:00Z", model_version: "student-v1", input_version: "forecast-ledger-v1", tickers: 0, independent_date_blocks: 0, ticker_origin_horizon_units: 0, contract_forecasts_available: 0, contract_cells_attempted: 0, brier: { baseline: null, candidate: null }, log_loss: { baseline: null, candidate: null }, calibration_by_side: { call: [], put: [] }, latency_ms: {} }, retrospective: null } } + await route.fulfill({ response, json: chain }) + }) + await page.goto("/") + await page.getByRole("combobox", { name: "Stock forecast model" }).selectOption("student_t_ewma") + const first = page.locator(".mobile-option-row").first() + await expect(first.locator(".odds-physical")).toContainText("Student-t EWMA") + expect(await first.evaluate((row) => row.querySelector(".mobile-row-summary")!.contains(row.querySelector(".mobile-model-compare details")))).toBe(false) + await first.getByText("Compare models").click() + await expect(first).toContainText("GJR-GARCH Student-t") + await expect(first).toContainText("N=0") + await expect(first).toContainText("sparse_strikes") + expect(await page.evaluate(() => document.documentElement.scrollWidth)).toBeLessThanOrEqual(390) +}) + for (const viewport of [ { width: 1440, height: 900 }, { width: 1024, height: 768 }, diff --git a/frontend/e2e/watchlist.spec.ts b/frontend/e2e/watchlist.spec.ts index 16f47ea..ccfee24 100644 --- a/frontend/e2e/watchlist.spec.ts +++ b/frontend/e2e/watchlist.spec.ts @@ -84,7 +84,7 @@ test("watches a desktop chain contract and restores its URL after visiting the w await page.goto("/?t=IREN&side=call&m=itm&cols=strike_cents") await expect(page.getByRole("columnheader", { name: "Watch" })).toBeVisible() - await expect(page.getByRole("columnheader", { name: "Expiry odds" })).toBeVisible() + await expect(page.getByRole("columnheader", { name: /Expiry odds · EWMA lognormal/ })).toBeVisible() await expect(page.locator(".odds-market").first()).toContainText("62.0% ITM") await expect(page.locator(".odds-market").first()).toContainText("38.0% OTM") const watch = page.getByRole("button", { name: /Watch IREN 2026-09-18 .* strike/ }).first() @@ -205,6 +205,23 @@ test("shows a separate forecast, prior market context, and dated hypothetical ri await expect.poll(() => page.evaluate(() => document.documentElement.scrollWidth <= innerWidth)).toBe(true) }) +test("shows shared model evidence in the mobile watchlist comparison", async ({ page }) => { + await page.setViewportSize({ width: 320, height: 800 }) + await page.route("**/api/watchlist", async (route) => { + const forecast = { status: "available", method: "lognormal_ewma", itm_pct_tenths: 520, otm_pct_tenths: 480, atm_pct_tenths: 0, evidence_key: "lognormal_ewma:2-5" } + await route.fulfill({ json: { + items: [{ ...item, predictive_odds: forecast, physical_models: [forecast, { status: "pending", method: "student_t_ewma", reason: "candidate_not_prepared" }], market_models: [{ status: "unavailable", method: "constrained_call_curve", reason: "sparse_strikes" }] }], + model_evidence: { "lognormal_ewma:2-5": { prospective: { generated_at: "2026-09-27T12:00:00Z", model_version: "ewma-v1", input_version: "forecast-ledger-v1", tickers: 0, independent_date_blocks: 0, ticker_origin_horizon_units: 0, contract_forecasts_available: 0, contract_cells_attempted: 0, brier: { baseline: null, candidate: null }, log_loss: { baseline: null, candidate: null }, calibration_by_side: { call: [], put: [] }, latency_ms: {} }, retrospective: null } }, + } }) + }) + await page.goto("/watchlist") + const odds = page.getByRole("region", { name: "Odds estimates" }) + await odds.getByText("Compare models").click() + await expect(odds).toContainText("N=0") + await expect(odds).toContainText("sparse_strikes") + expect(await page.evaluate(() => document.documentElement.scrollWidth)).toBeLessThanOrEqual(320) +}) + test("shows a complete populated watch card at 100% desktop zoom", async ({ page }) => { await page.setViewportSize({ width: 1440, height: 800 }) await page.route("**/api/watchlist", async (route) => { diff --git a/frontend/openapi.json b/frontend/openapi.json index ec59d38..b91841e 100644 --- a/frontend/openapi.json +++ b/frontend/openapi.json @@ -112,6 +112,23 @@ ], "title": "Moneyness" } + }, + { + "name": "forecast_model", + "in": "query", + "required": false, + "schema": { + "enum": [ + "lognormal_ewma", + "empirical_scaled", + "student_t_ewma", + "gjr_garch_t", + "intraday_shadow" + ], + "type": "string", + "default": "lognormal_ewma", + "title": "Forecast Model" + } } ], "responses": { @@ -174,6 +191,23 @@ ], "title": "Moneyness" } + }, + { + "name": "forecast_model", + "in": "query", + "required": false, + "schema": { + "enum": [ + "lognormal_ewma", + "empirical_scaled", + "student_t_ewma", + "gjr_garch_t", + "intraday_shadow" + ], + "type": "string", + "default": "lognormal_ewma", + "title": "Forecast Model" + } } ], "responses": { @@ -200,6 +234,53 @@ } } }, + "/api/version": { + "get": { + "summary": "Version", + "operationId": "version_api_version_get", + "parameters": [ + { + "name": "frontend_sha", + "in": "query", + "required": false, + "schema": { + "anyOf": [ + { + "type": "string", + "pattern": "^[0-9a-f]{40}$" + }, + { + "type": "null" + } + ], + "title": "Frontend Sha" + } + } + ], + "responses": { + "200": { + "description": "Successful Response", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/VersionStatus" + } + } + } + }, + "422": { + "description": "Validation Error", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + } + } + } + } + }, "/api/jobs/{job_id}": { "get": { "summary": "Get Job", @@ -243,6 +324,25 @@ "get": { "summary": "Get Watchlist", "operationId": "get_watchlist_api_watchlist_get", + "parameters": [ + { + "name": "forecast_model", + "in": "query", + "required": false, + "schema": { + "enum": [ + "lognormal_ewma", + "empirical_scaled", + "student_t_ewma", + "gjr_garch_t", + "intraday_shadow" + ], + "type": "string", + "default": "lognormal_ewma", + "title": "Forecast Model" + } + } + ], "responses": { "200": { "description": "Successful Response", @@ -253,21 +353,50 @@ } } } + }, + "422": { + "description": "Validation Error", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + } } } }, "post": { "summary": "Add Watch", "operationId": "add_watch_api_watchlist_post", + "parameters": [ + { + "name": "forecast_model", + "in": "query", + "required": false, + "schema": { + "enum": [ + "lognormal_ewma", + "empirical_scaled", + "student_t_ewma", + "gjr_garch_t", + "intraday_shadow" + ], + "type": "string", + "default": "lognormal_ewma", + "title": "Forecast Model" + } + } + ], "requestBody": { + "required": true, "content": { "application/json": { "schema": { "$ref": "#/components/schemas/WatchCreate" } } - }, - "required": true + } }, "responses": { "200": { @@ -697,6 +826,20 @@ "predictive_odds": { "$ref": "#/components/schemas/PredictiveOddsView" }, + "physical_models": { + "items": { + "$ref": "#/components/schemas/PredictiveOddsView" + }, + "type": "array", + "title": "Physical Models" + }, + "market_models": { + "items": { + "$ref": "#/components/schemas/MarketOddsView" + }, + "type": "array", + "title": "Market Models" + }, "hypothetical_risk": { "$ref": "#/components/schemas/HypotheticalRiskView" } @@ -930,6 +1073,11 @@ "lows": { "$ref": "#/components/schemas/PeriodLows" }, + "model_evidence": { + "additionalProperties": true, + "type": "object", + "title": "Model Evidence" + }, "expirations": { "items": { "$ref": "#/components/schemas/CashSecuredPutExpiration" @@ -1337,6 +1485,20 @@ "predictive_odds": { "$ref": "#/components/schemas/PredictiveOddsView" }, + "physical_models": { + "items": { + "$ref": "#/components/schemas/PredictiveOddsView" + }, + "type": "array", + "title": "Physical Models" + }, + "market_models": { + "items": { + "$ref": "#/components/schemas/MarketOddsView" + }, + "type": "array", + "title": "Market Models" + }, "hypothetical_risk": { "$ref": "#/components/schemas/HypotheticalRiskView" } @@ -1572,6 +1734,11 @@ "lows": { "$ref": "#/components/schemas/PeriodLows" }, + "model_evidence": { + "additionalProperties": true, + "type": "object", + "title": "Model Evidence" + }, "expirations": { "items": { "$ref": "#/components/schemas/CoveredCallExpiration" @@ -1641,6 +1808,24 @@ "title": "Status", "default": "unavailable" }, + "forecast_method": { + "anyOf": [ + { + "type": "string", + "enum": [ + "lognormal_ewma", + "empirical_scaled", + "student_t_ewma", + "gjr_garch_t", + "intraday_shadow" + ] + }, + { + "type": "null" + } + ], + "title": "Forecast Method" + }, "reason": { "anyOf": [ { @@ -1828,6 +2013,21 @@ }, "MarketOddsView": { "properties": { + "method": { + "anyOf": [ + { + "type": "string", + "enum": [ + "regimelib", + "constrained_call_curve" + ] + }, + { + "type": "null" + } + ], + "title": "Method" + }, "status": { "type": "string", "enum": [ @@ -1953,6 +2153,18 @@ } ], "title": "Quote Support Score" + }, + "model_evidence": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "title": "Model Evidence" } }, "type": "object", @@ -2130,7 +2342,14 @@ "method": { "anyOf": [ { - "type": "string" + "type": "string", + "enum": [ + "lognormal_ewma", + "empirical_scaled", + "student_t_ewma", + "gjr_garch_t", + "intraday_shadow" + ] }, { "type": "null" @@ -2275,6 +2494,29 @@ "type": "null" } ] + }, + "evidence_key": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Evidence Key" + }, + "model_evidence": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "title": "Model Evidence" } }, "type": "object", @@ -2288,6 +2530,14 @@ "const": "prospective_as_issued", "title": "Source" }, + "option_side": { + "type": "string", + "enum": [ + "call", + "put" + ], + "title": "Option Side" + }, "model_version": { "type": "string", "title": "Model Version" @@ -2326,6 +2576,7 @@ "type": "object", "required": [ "source", + "option_side", "model_version", "horizon_band", "moneyness_band", @@ -2444,6 +2695,79 @@ ], "title": "ValidationError" }, + "VersionStatus": { + "properties": { + "running_sha": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Running Sha" + }, + "branch": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Branch" + }, + "remote_sha": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Remote Sha" + }, + "status": { + "type": "string", + "enum": [ + "current", + "update_available", + "offline", + "unverified_checkout" + ], + "title": "Status" + }, + "checked_at": { + "type": "string", + "format": "date-time", + "title": "Checked At" + }, + "frontend_matches": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "title": "Frontend Matches" + } + }, + "type": "object", + "required": [ + "running_sha", + "branch", + "remote_sha", + "status", + "checked_at", + "frontend_matches" + ], + "title": "VersionStatus" + }, "WatchCreate": { "properties": { "watch_key": { @@ -2478,6 +2802,11 @@ "type": "null" } ] + }, + "model_evidence": { + "additionalProperties": true, + "type": "object", + "title": "Model Evidence" } }, "type": "object", @@ -2544,6 +2873,20 @@ "predictive_odds": { "$ref": "#/components/schemas/PredictiveOddsView" }, + "physical_models": { + "items": { + "$ref": "#/components/schemas/PredictiveOddsView" + }, + "type": "array", + "title": "Physical Models" + }, + "market_models": { + "items": { + "$ref": "#/components/schemas/MarketOddsView" + }, + "type": "array", + "title": "Market Models" + }, "hypothetical_risk": { "$ref": "#/components/schemas/HypotheticalRiskView" }, @@ -2583,6 +2926,11 @@ "type": "null" } ] + }, + "model_evidence": { + "additionalProperties": true, + "type": "object", + "title": "Model Evidence" } }, "type": "object", diff --git a/frontend/src/App.routes.browser.test.tsx b/frontend/src/App.routes.browser.test.tsx index 26917f6..44331b7 100644 --- a/frontend/src/App.routes.browser.test.tsx +++ b/frontend/src/App.routes.browser.test.tsx @@ -4,11 +4,13 @@ import { cleanup, fireEvent, render, screen, waitFor } from "@testing-library/re import { afterEach, expect, it, vi } from "vitest" import App from "./App" +import { samplePage } from "./testFixtures" afterEach(() => { cleanup() vi.unstubAllGlobals() window.sessionStorage.clear() + window.localStorage.clear() }) it("keeps the live chain query on the option chain link", () => { @@ -73,3 +75,23 @@ it("offers navigation when a route does not exist", () => { expect(main.querySelector('a[href="/watchlist"]')?.textContent).toContain("watchlist") expect(document.title).toBe("Page not found · HyperOptions") }) + +it("persists one model choice across chain and watchlist reads", async () => { + window.history.replaceState(null, "", "/") + const fetchMock = vi.fn(async (input: RequestInfo | URL) => new Response(JSON.stringify( + String(input).startsWith("/api/watchlist") ? { items: [] } : samplePage(), + ))) + vi.stubGlobal("fetch", fetchMock) + const mounted = render() + const picker = screen.getByRole("combobox", { name: "Stock forecast model" }) as HTMLSelectElement + expect(picker.value).toBe("lognormal_ewma") + fireEvent.change(picker, { target: { value: "student_t_ewma" } }) + await waitFor(() => expect(fetchMock.mock.calls.some(([url]) => String(url).includes("forecast_model=student_t_ewma"))).toBe(true)) + expect(localStorage.getItem("hyperoptions.forecastModel")).toBe("student_t_ewma") + fireEvent.click(screen.getByRole("link", { name: "Watchlist" })) + await waitFor(() => expect(fetchMock.mock.calls.some(([url]) => String(url) === "/api/watchlist?forecast_model=student_t_ewma")).toBe(true)) + expect((screen.getByRole("combobox", { name: "Stock forecast model" }) as HTMLSelectElement).value).toBe("student_t_ewma") + mounted.unmount() + render() + expect((screen.getByRole("combobox", { name: "Stock forecast model" }) as HTMLSelectElement).value).toBe("student_t_ewma") +}) diff --git a/frontend/src/App.tsx b/frontend/src/App.tsx index 100ab0f..82c5f45 100644 --- a/frontend/src/App.tsx +++ b/frontend/src/App.tsx @@ -2,6 +2,8 @@ import { lazy, Suspense, useEffect, useState } from "react" import { BrowserRouter, Link, MemoryRouter, Navigate, Route, Routes, useLocation } from "react-router-dom" import ItmChain from "./ItmChain" +import { DEFAULT_FORECAST_MODEL, FORECAST_MODEL_KEY, PHYSICAL_MODEL_NAMES, PHYSICAL_MODELS, physicalModel, type PhysicalModel } from "./forecastModels" +import VersionBanner from "./VersionBanner" import "./index.css" const Watchlist = lazy(() => import("./watchlist/Watchlist")) @@ -9,6 +11,13 @@ const Watchlist = lazy(() => import("./watchlist/Watchlist")) function Workspace() { const location = useLocation() const [lastChainUrl, setLastChainUrl] = useState("/") + const [forecastModel, setForecastModel] = useState(() => { + try { return physicalModel(localStorage.getItem(FORECAST_MODEL_KEY)) ?? DEFAULT_FORECAST_MODEL } + catch { return DEFAULT_FORECAST_MODEL } + }) + useEffect(() => { + try { localStorage.setItem(FORECAST_MODEL_KEY, forecastModel) } catch { /* Selection still works for this session. */ } + }, [forecastModel]) const watchlist = location.pathname === "/watchlist" useEffect(() => { const page = location.pathname === "/" ? "Option chain" : watchlist ? "Watchlist" : "Page not found" @@ -21,14 +30,25 @@ function Workspace() { return (
Skip to main content - +
+ + + +
- } /> + } /> } /> - Loading watchlist…

}>} /> + Loading watchlist…

}>} />

Page not found

This address does not match a workstation page.

Open the option chain or go to the watchlist.

} />
diff --git a/frontend/src/ExpiryTable.tsx b/frontend/src/ExpiryTable.tsx index adc7500..49c7f62 100644 --- a/frontend/src/ExpiryTable.tsx +++ b/frontend/src/ExpiryTable.tsx @@ -11,6 +11,8 @@ import { integer, moneyStrike } from "./format" import { heatmapHue, heatmapStop, type MetricRange } from "./heatmap" import { oddsAvailable, oddsLabel, oddsMessage, predictiveAvailable } from "./marketOdds" import OddsValues from "./OddsValues" +import ModelComparison from "./ModelComparison" +import { physicalModelName, type PhysicalModel } from "./forecastModels" import type { Side } from "./types" import { useMediaQuery } from "./useMediaQuery" import { PREDICTIVE_ODDS_SORT_ID, type SortState, type VisibleGroup } from "./viewModel" @@ -52,6 +54,8 @@ type Props = { copiedKey: string | null sort: SortState density: Density + forecastModel: PhysicalModel + modelEvidence?: Record | null onToggleExpiration: (expiration: string) => void onSort: (id: string) => void onCopy: (row: SizedContract, group: VisibleGroup["group"]) => void @@ -68,6 +72,8 @@ function DesktopResults({ copiedKey, sort, density, + forecastModel, + modelEvidence, onSort, onCopy, watchStates, @@ -81,6 +87,8 @@ function DesktopResults({ copiedKey: string | null sort: SortState density: Density + forecastModel: PhysicalModel + modelEvidence?: Record | null onSort: (id: string) => void onCopy: (row: SizedContract, group: VisibleGroup["group"]) => void watchStates: Readonly> @@ -123,9 +131,9 @@ function DesktopResults({ {active ? (sort.dir === "asc" ?