diff --git a/README.md b/README.md index aeece46..1417ae7 100644 --- a/README.md +++ b/README.md @@ -98,9 +98,9 @@ A ticker with no options shows `Options are not available for {ticker}`. A price Use **Watch** on a desktop or mobile chain row, then open `/watchlist`. Watches track contracts, not positions or trades; the app does not store holdings, premiums, or assignment decisions. Any supported Nasdaq-listed ticker may be watched. The server checks each submitted watch key against the current chain and Nasdaq universe and deduplicates the exact ticker, option root, call/put side, expiration, and strike. Only ordinary contracts whose parsed root and terms match the chain row are eligible. Adjusted or ambiguous series have a disabled Watch control with a reason; saved watches show **“Assuming standard 100-share terms.”** -Before expiry, the chain and watchlist lead with a dated **stock forecast** showing ITM and OTM odds for the real-world expiry-session close. The event is `close > strike` for a call and `close < strike` for a put. Equality remains a separate outcome in the model and API, so the two displayed percentages can total less than 100%. Completed Yahoo `Close` history gives the dated physical forecast. The default is a zero-drift lognormal distribution with 60-session EWMA volatility. Compare models shows separate EWMA, volatility-scaled empirical, Student-t EWMA, GJR-GARCH Student-t, and intraday results. A user-selected physical model drives the compact odds and hypothetical risk; no model is automatically chosen from past scores. The empirical, Student-t, and GJR methods support 1–25 sessions, while EWMA supports up to one year. The latest completed split-safe bar and verified cache are required. Each result shows its availability, input and fit diagnostics, and separately measured accuracy when matured outcomes exist. The comparison reports independent empirical history blocks, a maximum 95% Monte Carlo error for simulated odds, and the spread between comparable completed-close methods. Those describe sampling or method sensitivity, not predictive accuracy. Missing evidence is N/A, not forecast confidence. +Before expiry, the chain and watchlist lead with a dated **stock forecast** showing ITM and OTM odds for the real-world expiry-session close. The event is `close > strike` for a call and `close < strike` for a put. Equality remains a separate outcome in the model and API, so the two displayed percentages can total less than 100%. Completed Yahoo `Close` history gives the dated physical forecast. The default is a zero-drift lognormal distribution with 60-session EWMA volatility. Compare models also shows volatility-scaled empirical, Student-t EWMA, GJR-GARCH Student-t, daily-OHLC range/HAR proxy, skewed-t EWMA, EGARCH skewed-t, two-regime switching variance, pooled NGBoost, earnings-jump, IV-informed, and intraday results. A user-selected physical model drives the compact odds and hypothetical risk; no model is automatically chosen from past scores. The new historical methods support 1–25 sessions, while EWMA supports up to one year. The latest completed split-safe bar and verified cache are required. Rights-dependent methods remain visibly unavailable until their input provenance qualifies. Each result shows its availability, input and fit diagnostics, and separately measured accuracy when matured outcomes exist. The comparison reports independent empirical history blocks, a maximum 95% Monte Carlo error for simulated odds, and the spread between comparable completed-close methods. Those describe sampling or method sensitivity, not predictive accuracy. Missing evidence is N/A, not forecast confidence. -**Market-implied odds** are displayed separately as risk-neutral option-price context, not as a substitute for the physical forecast. The `regimelib` estimator fits one two-state distribution to validated call quotes using the Treasury rate for each expiry, then prices a cash digital. A separate per-expiry decreasing, convex call-price curve is also shown when its quote and one-tick stability checks pass. Wide bounds remain visible; quote tightness and held-out quote fit describe market-input robustness, not realized forecast accuracy. Sparse or contradictory strips cannot provide a precise market probability. American exercise, dividends, and missing individual quote timestamps still limit interpretation. The physical and risk-neutral probabilities are never combined into a recommendation score. +**Market-implied odds** are displayed separately as risk-neutral option-price context, not as a substitute for the physical forecast. The `regimelib` estimator fits one two-state distribution to validated call quotes using the Treasury rate for each expiry, then prices a cash digital. A separate per-expiry decreasing, convex call-price curve is also shown when its quote and one-tick stability checks pass. SSVI is listed separately but remains unavailable until coherent, rights-cleared multi-expiry quotes and defensible American-exercise and dividend treatment exist. Wide bounds remain visible; quote tightness and held-out quote fit describe market-input robustness, not realized forecast accuracy. Sparse or contradictory strips cannot provide a precise market probability. American exercise, dividends, and missing individual quote timestamps still limit interpretation. The physical and risk-neutral probabilities are never combined into a recommendation score. Visible chain pages refresh odds about every five minutes during the regular trading session; a visible watchlist polls for updated cached odds. If a later watchlist poll fails, the last loaded list stays visible with a dated warning and Retry action. After hours, quote-implied valuation uses the latest completed session's official daily close and its dated rate. A watched contract's last valid market value from the latest completed session remains visible as dated context across refreshes and restarts, then disappears when a newer session completes. The compact stock summary shows its quote time, and market-session labels describe the source state at fetch. The fetch time is a snapshot time, not a claim about each option's quote timestamp; fetched and retrieved times include the browser's local timezone, while market quote time is labeled ET. Legacy strategy forecast snapshots remain in the local database for historical continuity but are never served as current odds. **Check expiry results** requests a separate background close-based outcome update. @@ -129,7 +129,9 @@ Watch jobs, watched contracts, close-based observations, append-only forecast is From `backend/`, `uv run --group research python scripts/evaluate_predictive.py IREN --refresh` refreshes public Yahoo closes and prints a retrospective rolling-origin report for the older per-origin EWMA/empirical selection policy, alongside its EWMA baseline comparator. That policy is research-only and does not choose the app's default forecast. Without `--refresh`, the command reads only the verified cache; `--as-of YYYY-MM-DD` evaluates a past completed session. To freeze the deterministic 50-stock convenience sample from eligible, verified cached Nasdaq stocks, run `uv run --group research python scripts/freeze_audit_cohort.py`. Then screen a challenger with `uv run --group research python scripts/evaluate_predictive.py --replay-cohort --candidate student_t_ewma --period screen`. These immutable snapshots contain current-vintage history, so replay is retrospective screening, never as-issued evidence. -Live forecast attempts for each method are recorded in an append-only DuckDB ledger, including unavailable attempts. The app's per-model evidence uses only issuances carrying the currently displayed model version. Its candidate coverage is conditional on recorded current-version candidate cells; cells without EWMA count as failures. Older generic-version attempts and missed origins remain outside this rate and are not silently counted as current-version failures. Once exact expiry-session closes have matured and passed the Nasdaq/Yahoo cross-check, `uv run --group research python scripts/evaluate_predictive.py --ledger-contest --candidate student_t_ewma --period holdout --holdout-start YYYY-MM-DD` reports paired as-issued Brier score, log loss, full-distribution score, calibration, availability, rejection reasons, and latency by horizon, moneyness, and volatility regime. Contract sides and strikes sharing one stock close are averaged into one ticker-origin-horizon unit; overlapping intervals are purged and calendar-date blocks are bootstrapped with all tickers together. These reports measure accuracy on matched observations and never change the selected model. An older `forecast_champions.json` file, if present, is left untouched and ignored. +Live forecast attempts for each method are recorded in an append-only DuckDB ledger, including unavailable attempts. Expected capture windows are recorded separately so missed windows are visible. The app's per-model evidence uses only issuances carrying the currently displayed model version. Its candidate coverage is conditional on recorded current-version candidate cells; cells without EWMA count as failures. Older generic-version attempts and missed origins remain outside this rate and are not silently counted as current-version failures. Once exact expiry-session closes have matured and passed the Nasdaq/Yahoo cross-check, `uv run --group research python scripts/evaluate_predictive.py --ledger-contest --candidate student_t_ewma --period holdout --holdout-start YYYY-MM-DD` reports paired as-issued Brier score, log loss, full-distribution score, calibration, availability, rejection reasons, and latency by horizon, moneyness, and volatility regime. Contract sides and strikes sharing one stock close are averaged into one ticker-origin-horizon unit; overlapping intervals are purged and calendar-date blocks are bootstrapped with all tickers together. The comparison dialog shows a calendar-block interval adjusted for the ten predeclared physical-model comparisons **within each horizon band** when independent support exists; comparisons across bands remain exploratory. These reports measure accuracy on matched observations and never change the selected model. An older `forecast_champions.json` file, if present, is left untouched and ignored. + +The [source-rights audit](docs/source-rights.md) explains why this release does not expand Yahoo downloads or store Nasdaq option quotes for training. Pooled NGBoost also needs its exact trained-artifact hash recorded with each issued forecast before live numbers can be shown. Earnings-jump, IV-informed physical, and SSVI market estimates require qualified inputs; their unavailable reasons remain visible in the comparison dialog. `uv run --group research python scripts/evaluate_intraday_open.py IREN` screens the intraday method on historical Opens; prospectively timestamped 10:00, 13:00, and 15:30 ET snapshots are recorded for separate accuracy measurement. `uv run --group research python scripts/evaluate_intraday_prospective.py` scores matched, matured as-issued snapshots against both the dated-close forecast and simple quote reanchor, with window coverage and latency. `uv run --group research python scripts/evaluate_market_curve.py` reports market-curve quote fit, coverage, rejections, and latency. Market-implied odds are judged on quote consistency rather than realized-outcome Brier score. These reports describe stock-close forecasts and option-quote fits; they do not claim retrospective option-trading performance. diff --git a/backend/pyproject.toml b/backend/pyproject.toml index 881436a..1ec5c5a 100644 --- a/backend/pyproject.toml +++ b/backend/pyproject.toml @@ -16,6 +16,7 @@ dependencies = [ "pydantic>=2.13", "regimelib==0.1.0", "scipy>=1.18,<2", + "statsmodels==0.15.0", "uvicorn[standard]>=0.32", "yfinance>=1.7", ] @@ -28,7 +29,10 @@ dev = [ "httpx2>=2.9", "ruff>=0.15", ] -research = [] +research = [ + "ngboost==0.5.11", + "scikit-learn==1.6.1", +] [tool.ruff] target-version = "py312" diff --git a/backend/scripts/capture_sec_events.py b/backend/scripts/capture_sec_events.py new file mode 100644 index 0000000..49036c7 --- /dev/null +++ b/backend/scripts/capture_sec_events.py @@ -0,0 +1,33 @@ +"""Capture explicit earnings dates from a bounded set of one issuer's recent 8-Ks.""" + +from __future__ import annotations + +import argparse +import os +from pathlib import Path + +from stocksweeper.config import load_settings +from stocksweeper.forecast.sec_events import capture_recent_sec_schedules + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("ticker", help="Exact SEC-listed ticker, e.g. AAPL") + parser.add_argument("cik", type=int, help="SEC CIK for that ticker") + parser.add_argument("--user-agent", help="Declared SEC contact (or use environment variable)") + parser.add_argument("--data-dir", type=Path, help="override STOCKSWEEPER_DATA_DIR") + args = parser.parse_args() + user_agent = args.user_agent or os.environ.get("HYPEROPTIONS_SEC_USER_AGENT") + if not user_agent: + parser.error("set HYPEROPTIONS_SEC_USER_AGENT to a declared name and contact email") + added = capture_recent_sec_schedules( + args.data_dir or load_settings().resolved_data_dir(), + args.ticker.upper(), + args.cik, + user_agent, + ) + print(f"forward schedules added: {added}; actual releases are recorded separately") + + +if __name__ == "__main__": + main() diff --git a/backend/scripts/evaluate_predictive.py b/backend/scripts/evaluate_predictive.py index d47ad9c..0b6d3ab 100644 --- a/backend/scripts/evaluate_predictive.py +++ b/backend/scripts/evaluate_predictive.py @@ -226,7 +226,10 @@ def main() -> None: help="atomically save a replay report for the local app comparison view", ) parser.add_argument( - "--candidate", choices=("empirical_scaled", "student_t_ewma", "gjr_garch_t") + "--candidate", choices=( + "empirical_scaled", "student_t_ewma", "gjr_garch_t", "ohlc_har", + "skew_t_ewma", "egarch_skew_t", "markov_switching", "ngboost_pooled", + ) ) parser.add_argument( "--provenance", choices=("as_issued", "immutable_replay"), default="as_issued" diff --git a/backend/src/options_api/live_quant.py b/backend/src/options_api/live_quant.py index 26e9e4c..2f73c73 100644 --- a/backend/src/options_api/live_quant.py +++ b/backend/src/options_api/live_quant.py @@ -33,6 +33,8 @@ from stocksweeper.forecast.ledger import ForecastIssuance from stocksweeper.forecast.predictive import PredictiveDistribution +SSVI_VERSION = "ssvi-market-v1" + @dataclass(frozen=True) class LiveQuant: @@ -160,10 +162,7 @@ def quant_for_contract( if physical_shadow is not None: model_views = [] quote_for_model = market_odds.underlying_quote(ticker) - for method in ( - "lognormal_ewma", "empirical_scaled", "student_t_ewma", "gjr_garch_t", - "intraday_shadow", - ): + for method in MODEL_VERSIONS: if predictive.status == "unavailable": view = PredictiveOddsView( method=method, status="unavailable", reason=predictive.reason, @@ -248,8 +247,15 @@ def quant_for_contract( "bid_ask_fit": None, }}) market_models = ( - (market, market_odds.lookup_curve(ticker, side, expiry_text, strike, root)) - if physical_shadow is not None else (market,) + ( + market, + market_odds.lookup_curve(ticker, side, expiry_text, strike, root), + MarketOddsView( + method="ssvi", status="unavailable", + reason="rights_cleared_option_history_unavailable", + model_version=SSVI_VERSION, + ), + ) if physical_shadow is not None else (market,) ) last_good = ( market_odds.lookup_last_good(ticker, side, expiry_text, strike, root) diff --git a/backend/src/options_api/main.py b/backend/src/options_api/main.py index f3ce054..22f6a62 100644 --- a/backend/src/options_api/main.py +++ b/backend/src/options_api/main.py @@ -22,7 +22,7 @@ from options_api.chain import load_cash_secured_puts, load_covered_calls from options_api.greeks import empty_greeks from options_api.intraday_capture import capture_intraday_window -from options_api.live_quant import quant_for_contract +from options_api.live_quant import SSVI_VERSION, quant_for_contract from options_api.market_calendar import first_session_after_completed from options_api.market_watch import MarketWatchOdds from options_api.models import ( @@ -39,7 +39,7 @@ ) from options_api.nasdaq import NasdaqError, create_http_client from options_api.outcomes import CloseProvider -from options_api.physical_shadow_capture import PhysicalShadowCapture +from options_api.physical_shadow_capture import MODEL_VERSIONS, PhysicalShadowCapture from options_api.predictive_watch import PredictiveWatchOdds from options_api.prospective_panel import ProspectivePanel from options_api.service import OptionChainService @@ -48,6 +48,7 @@ from options_api.watchlist import WatchlistService, router as watchlist_router from stocksweeper.config import Settings, load_settings from stocksweeper.forecast.labels import collect_matured_labels +from stocksweeper.forecast.capture_windows import reconcile_capture_windows from stocksweeper.forecast.predictive import PredictiveForecaster from stocksweeper.pipeline.jobs import JobManager from stocksweeper.storage.db import single_instance @@ -270,17 +271,14 @@ async def _load_page( PredictiveOddsView( method=method, status="unavailable", reason="contract_terms_ambiguous" ) - for method in ( - "lognormal_ewma", "empirical_scaled", "student_t_ewma", - "gjr_garch_t", "intraday_shadow", - ) + for method in MODEL_VERSIONS ] contract.market_models = [ MarketOddsView( method=method, status="unavailable", reason="Contract terms cannot be verified", ) - for method in ("regimelib", "constrained_call_curve") + for method in ("regimelib", "constrained_call_curve", "ssvi") ] contract.hypothetical_risk = HypotheticalRiskView( reason="Contract terms cannot be verified" @@ -322,6 +320,12 @@ async def _load_page( reason="Refreshing odds for displayed quotes", ) for method in ("regimelib", "constrained_call_curve") + ] + [ + MarketOddsView( + method="ssvi", status="unavailable", + reason="rights_cleared_option_history_unavailable", + model_version=SSVI_VERSION, + ) ] ) contract.hypothetical_risk = result.risk @@ -484,11 +488,28 @@ async def refresh_outcomes() -> None: async def capture_intraday() -> None: prepared: set[tuple[date, str]] = set() captured: set[tuple[date, str]] = set() + reconciled: set[tuple[date, str]] = set() + try: + now = datetime.now(UTC) + watched = await asyncio.to_thread(app.state.watchlist.store.list) + await asyncio.to_thread( + reconcile_capture_windows, settings.resolved_data_dir(), watched, now + ) + day = now.astimezone(_NY).date() + reconciled = { + (day, window) + for window, local_time in _INTRADAY_WINDOWS.items() + if datetime.combine(day, local_time, _NY).astimezone(UTC) + + timedelta(minutes=5) <= now + } + except Exception: + LOG.exception("intraday capture-window reconciliation failed") while True: now = datetime.now(UTC) day = now.astimezone(_NY).date() prepared = {key for key in prepared if key[0] == day} captured = {key for key in captured if key[0] == day} + reconciled = {key for key in reconciled if key[0] == day} for window, local_time in _INTRADAY_WINDOWS.items(): target = datetime.combine(day, local_time, _NY).astimezone(UTC) key = (day, window) @@ -545,6 +566,22 @@ async def capture_intraday() -> None: captured.add(key) except Exception: LOG.exception("intraday snapshot capture failed for %s", window) + ended = { + (day, window) + for window, local_time in _INTRADAY_WINDOWS.items() + if datetime.combine(day, local_time, _NY).astimezone(UTC) + + timedelta(minutes=5) <= now + } + if ended - reconciled: + try: + watched = await asyncio.to_thread(app.state.watchlist.store.list) + await asyncio.to_thread( + reconcile_capture_windows, + settings.resolved_data_dir(), watched, now, + ) + reconciled.update(ended) + except Exception: + LOG.exception("intraday capture-window reconciliation failed") await asyncio.sleep(30) async def capture_panel() -> None: diff --git a/backend/src/options_api/models.py b/backend/src/options_api/models.py index 6ff522d..94bc6f7 100644 --- a/backend/src/options_api/models.py +++ b/backend/src/options_api/models.py @@ -15,7 +15,9 @@ GreeksSource = Literal["bid", "mid"] MarketSource = Literal["nasdaq", "yahoo"] PhysicalModel = Literal[ - "lognormal_ewma", "empirical_scaled", "student_t_ewma", "gjr_garch_t", "intraday_shadow" + "lognormal_ewma", "empirical_scaled", "student_t_ewma", "gjr_garch_t", + "ohlc_har", "skew_t_ewma", "egarch_skew_t", "markov_switching", + "ngboost_pooled", "earnings_jump", "iv_physical", "intraday_shadow", ] @@ -111,7 +113,7 @@ class TickerSearchResponse(BaseModel): class MarketOddsView(BaseModel): - method: Literal["regimelib", "constrained_call_curve"] | None = None + method: Literal["regimelib", "constrained_call_curve", "ssvi"] | None = None status: Literal["pending", "available", "unavailable"] = "pending" itm_pct_tenths: int | None = None otm_pct_tenths: int | None = None diff --git a/backend/src/options_api/physical_shadow_capture.py b/backend/src/options_api/physical_shadow_capture.py index 37e402e..c77dc75 100644 --- a/backend/src/options_api/physical_shadow_capture.py +++ b/backend/src/options_api/physical_shadow_capture.py @@ -18,6 +18,7 @@ from stocksweeper.forecast.ledger import ForecastIssuance from stocksweeper.forecast.physical_contest import ( EMPIRICAL_SHADOW_VERSION, GJR_VERSION, STUDENT_VERSION, + HAR_VERSION, SKEW_T_VERSION, EGARCH_VERSION, MARKOV_VERSION, NGBOOST_VERSION, PhysicalShadowForecaster, ShadowForecast, ) from stocksweeper.forecast.predictive import BASELINE_VERSION, PredictiveDistribution @@ -25,15 +26,23 @@ LOG = logging.getLogger(__name__) _MAX_BATCH_CONTRACTS = 512 _MAX_PENDING_BATCHES = 4 -_MAX_CACHED_GROUPS = 512 +_MAX_CACHED_GROUPS = 96 _RETRY_SECONDS = 30 MODEL_VERSIONS = { "lognormal_ewma": BASELINE_VERSION, "empirical_scaled": EMPIRICAL_SHADOW_VERSION, "student_t_ewma": STUDENT_VERSION, "gjr_garch_t": GJR_VERSION, + "ohlc_har": HAR_VERSION, + "skew_t_ewma": SKEW_T_VERSION, + "egarch_skew_t": EGARCH_VERSION, + "markov_switching": MARKOV_VERSION, + "ngboost_pooled": NGBOOST_VERSION, + "earnings_jump": "earnings-jump-v1", + "iv_physical": "iv-physical-v1", "intraday_shadow": INTRADAY_VERSION, } +CLOSE_METHODS = tuple(method for method in MODEL_VERSIONS if method != "intraday_shadow") Entry = tuple[ForecastIssuance, PredictiveDistribution | None] @@ -69,6 +78,10 @@ def candidate( return ShadowForecast(None, base.reason or "completed_close_forecast_unavailable", 0, 0) if base.horizon_sessions > 25 and method != "lognormal_ewma": return ShadowForecast(None, "candidate_horizon_unsupported", 0, 0) + if method == "iv_physical": + return ShadowForecast(None, "rights_cleared_option_history_unavailable", 0, 0) + if method == "earnings_jump": + return ShadowForecast(None, "verified_release_time_history_unavailable", 0, 0) if method == base.method and method != "intraday_shadow": return ShadowForecast(base, None, 0, 0) key = base.ticker, base.as_of, expiry, contract_since, base.data_hash @@ -157,7 +170,7 @@ def _capacity_entries(entries: list[Entry]) -> list[Entry]: None, ) for issue, _ in entries - for method in ("lognormal_ewma", "empirical_scaled", "student_t_ewma", "gjr_garch_t") + for method in CLOSE_METHODS ] def _record_capacity(self, entries: list[Entry]) -> None: @@ -181,10 +194,7 @@ def _cache_capacity( continue rejected = { method: ShadowForecast(None, "shadow_capacity_exceeded", 0, 0) - for method in ( - "lognormal_ewma", "empirical_scaled", "student_t_ewma", - "gjr_garch_t", "intraday_shadow", - ) + for method in MODEL_VERSIONS } if distribution is not None and distribution.method == "lognormal_ewma": rejected["lognormal_ewma"] = ShadowForecast(distribution, None, 0, 0) @@ -286,12 +296,7 @@ def _capture_sync( regime = ( "low" if volatility < 0.02 else "medium" if volatility < 0.05 else "high" ) if volatility is not None else "unknown" - for method in ( - "lognormal_ewma", - "empirical_scaled", - "student_t_ewma", - "gjr_garch_t", - ): + for method in CLOSE_METHODS: candidate = candidates.get(method) if candidates is not None else None distribution = candidate.distribution if candidate is not None else None reason = ( @@ -303,6 +308,10 @@ def _capture_sync( ) if candidate is not None: reason = candidate.reason or "candidate_unavailable" + elif method == "iv_physical": + reason = "rights_cleared_option_history_unavailable" + elif method == "earnings_jump": + reason = "verified_release_time_history_unavailable" if distribution is not None and ( issue.status != "available" or distribution.data_hash != issue.data_hash diff --git a/backend/src/options_api/predictive_watch.py b/backend/src/options_api/predictive_watch.py index e9d1625..98e928e 100644 --- a/backend/src/options_api/predictive_watch.py +++ b/backend/src/options_api/predictive_watch.py @@ -337,7 +337,9 @@ def view_for_distribution( atm_pct_tenths=1000 - itm - otm, simulation_error_95_pct_tenths=( ceil(980 / sqrt(distribution.support)) - if distribution.method in {"student_t_ewma", "gjr_garch_t"} + if distribution.method not in { + "lognormal_ewma", "empirical_scaled", "intraday_shadow" + } and distribution.support > 0 else None ), diff --git a/backend/src/options_api/watchlist.py b/backend/src/options_api/watchlist.py index 5d99ef7..75b292e 100644 --- a/backend/src/options_api/watchlist.py +++ b/backend/src/options_api/watchlist.py @@ -16,7 +16,7 @@ from pydantic import BaseModel, ConfigDict, Field from options_api.contract_identity import make_watch_key, parse_watch_key, strike_exact -from options_api.live_quant import MODEL_VERSIONS, quant_for_contract +from options_api.live_quant import MODEL_VERSIONS, SSVI_VERSION, quant_for_contract from options_api.market_calendar import ( expiry_session_completed, first_session_after_completed, @@ -40,6 +40,7 @@ YahooCloseProvider, resolve_outcome, ) +from stocksweeper.forecast.capture_windows import reconcile_capture_windows from stocksweeper.pipeline.jobs import Job, JobBusy, JobManager from stocksweeper.storage.db import connect, rows @@ -204,7 +205,13 @@ def add( assert item is not None return item, created - def delete(self, watch_id: str) -> bool: + def delete(self, watch_id: str, *, now: datetime | None = None) -> bool: + watched = next((item for item in self.list() if item.id == watch_id), None) + if watched is None: + return False + reconcile_capture_windows( + self.path.parent, [watched], now or datetime.now(UTC), contract_key=watched.watch_key + ) with connect(self.path) as connection: found = rows(connection, "SELECT id FROM watches WHERE id = ?", [watch_id]) if not found: @@ -509,6 +516,7 @@ async def get_watchlist( for method, version in ( ("regimelib", _MODEL_VERSION), ("constrained_call_curve", _CURVE_VERSION), + ("ssvi", SSVI_VERSION), ) ] item.hypothetical_risk = HypotheticalRiskView( @@ -675,7 +683,7 @@ async def add_watch( async def delete_watch(request: Request, watch_id: str) -> None: if len(watch_id) != 32 or any(char not in "0123456789abcdef" for char in watch_id): raise HTTPException(status_code=404, detail="Watch not found") - if not request.app.state.watchlist.store.delete(watch_id): + if not await asyncio.to_thread(request.app.state.watchlist.store.delete, watch_id): raise HTTPException(status_code=404, detail="Watch not found") diff --git a/backend/src/stocksweeper/forecast/capture_windows.py b/backend/src/stocksweeper/forecast/capture_windows.py new file mode 100644 index 0000000..bfe022d --- /dev/null +++ b/backend/src/stocksweeper/forecast/capture_windows.py @@ -0,0 +1,195 @@ +"""Append-only denominators for expected and missed intraday watch captures.""" + +from __future__ import annotations + +import hashlib +from collections.abc import Iterable +from datetime import UTC, date, datetime, time, timedelta +from pathlib import Path +from typing import Protocol +from zoneinfo import ZoneInfo + +import duckdb + +from options_api.contract_identity import parse_watch_key +from options_api.market_calendar import _calendar +from stocksweeper.storage.db import connect, rows + +_NY = ZoneInfo("America/New_York") +_WINDOWS = {"10:00": time(10), "13:00": time(13), "15:30": time(15, 30)} +_GRACE = timedelta(minutes=5) +_LOOKBACK_DAYS = 35 + + +def capture_window_counts(data_dir: Path, since: datetime, as_of: datetime) -> dict[str, int]: + """Read distinct expected windows, with later captures superseding misses. + + The 35-day limit matches restart reconciliation. This reads existing events + without creating a database or changing the append-only event log. + """ + if ( + since.tzinfo is None + or as_of.tzinfo is None + or not timedelta(0) <= as_of - since <= timedelta(days=_LOOKBACK_DAYS) + ): + raise ValueError("capture count interval must be aware and at most 35 days") + counts = {"expected": 0, "captured": 0, "missed": 0, "pending": 0} + path = data_dir / "results.duckdb" + if not path.is_file(): + return counts + # DuckDB cannot mix read-only and read-write connections for one file in + # the running app. Use its normal connection mode but issue SELECTs only. + with duckdb.connect(str(path)) as connection: + tables = {row[0] for row in connection.execute("SHOW TABLES").fetchall()} + if "forecast_capture_window_events" not in tables: + return counts + events = rows( + connection, + """SELECT contract_key, target_at, event, recorded_at + FROM forecast_capture_window_events + WHERE target_at >= ? AND target_at <= ? AND recorded_at <= ?""", + [since, as_of, as_of], + ) + observed: dict[tuple[str, datetime], set[str]] = {} + for row in events: + key = (row["contract_key"], row["target_at"].astimezone(UTC)) + observed.setdefault(key, set()).add(row["event"]) + for (_, target), states in observed.items(): + if "expected" not in states: + continue + counts["expected"] += 1 + state = ( + "captured" + if "captured" in states + else ("missed" if "missed" in states or target + _GRACE <= as_of else "pending") + ) + counts[state] += 1 + return counts + + +class WatchedContract(Protocol): + watch_key: str + created_at: datetime + expiration: date + + +def reconcile_capture_windows( + data_dir: Path, + watches: Iterable[WatchedContract], + now: datetime, + *, + contract_key: str | None = None, +) -> dict[str, int]: + """Record expected windows and outcomes for the current watchlist. + + Call at startup and after every window. Restarts recover up to 35 calendar + days of missed windows for watches still present. Deleted watches cannot be + reconstructed from the current table and are deliberately excluded. + """ + if now.tzinfo is None: + raise ValueError("capture reconciliation time must be timezone-aware") + if contract_key is not None and parse_watch_key(contract_key) is None: + raise ValueError("invalid contract key for capture accounting") + now = now.astimezone(UTC) + current_day = now.astimezone(_NY).date() + calendar = _calendar() + start = current_day - timedelta(days=_LOOKBACK_DAYS) + sessions = calendar.sessions_in_range(start.isoformat(), current_day.isoformat()) + targets: dict[tuple[str, datetime], str] = {} + for watch in watches: + if contract_key is not None and watch.watch_key != contract_key: + continue + if parse_watch_key(watch.watch_key) is None or watch.created_at.tzinfo is None: + raise ValueError("invalid watched contract for capture accounting") + created_at = watch.created_at.astimezone(UTC) + for session in sessions: + day = session.date() + if day > watch.expiration: + break + opened = calendar.session_open(session).to_pydatetime() + closed = calendar.session_close(session).to_pydatetime() + for window, wall_time in _WINDOWS.items(): + target = datetime.combine(day, wall_time, _NY).astimezone(UTC) + if opened <= target < closed and created_at <= target <= now: + targets[(watch.watch_key, target)] = window + with connect(data_dir / "results.duckdb") as connection: + connection.execute( + """CREATE TABLE IF NOT EXISTS forecast_capture_window_events ( + event_key VARCHAR PRIMARY KEY, + contract_key VARCHAR NOT NULL, + target_at TIMESTAMPTZ NOT NULL, + snapshot_window VARCHAR NOT NULL, + event VARCHAR NOT NULL, + reason VARCHAR, + recorded_at TIMESTAMPTZ NOT NULL + )""" + ) + # A watch may be removed after its target but before the grace period + # ends. Keep reconciling already-recorded expectations after removal. + persisted_filter = " AND contract_key = ?" if contract_key is not None else "" + persisted_params = [datetime.combine(start, time.min, _NY).astimezone(UTC), now] + if contract_key is not None: + persisted_params.append(contract_key) + persisted = rows( + connection, + """SELECT contract_key, target_at, snapshot_window + FROM forecast_capture_window_events + WHERE event = 'expected' AND target_at >= ? AND target_at <= ?""" + + persisted_filter, + persisted_params, + ) + for row in persisted: + window = row["snapshot_window"] + if window in _WINDOWS: + targets[(row["contract_key"], row["target_at"].astimezone(UTC))] = window + if not targets: + return {"expected": 0, "captured": 0, "missed": 0, "pending": 0} + earliest = min(target for _, target in targets) + issued_filter = " AND contract_key = ?" if contract_key is not None else "" + issued_params = [earliest, now] + if contract_key is not None: + issued_params.append(contract_key) + issued = rows( + connection, + """SELECT DISTINCT contract_key, issued_at, snapshot_window + FROM forecast_issuances + WHERE snapshot_window IS NOT NULL AND issued_at >= ? AND issued_at <= ?""" + + issued_filter, + issued_params, + ) + captured = set() + for row in issued: + window = row["snapshot_window"] + if window not in _WINDOWS: + continue + issued_at = row["issued_at"].astimezone(UTC) + day = issued_at.astimezone(_NY).date() + target = datetime.combine(day, _WINDOWS[window], _NY).astimezone(UTC) + key = (row["contract_key"], target) + if key in targets and target <= issued_at < target + _GRACE: + captured.add(key) + counts = {"expected": len(targets), "captured": 0, "missed": 0, "pending": 0} + new_events = [] + for (key, target), window in targets.items(): + state = ( + "captured" + if (key, target) in captured + else ("missed" if target + _GRACE <= now else "pending") + ) + counts[state] += 1 + for event, reason in ( + ("expected", None), + (state, "no_issuance_in_window" if state == "missed" else None), + ): + if event == "pending": + continue + event_key = hashlib.sha256( + f"{key}|{target.isoformat()}|{event}".encode() + ).hexdigest() + new_events.append((event_key, key, target, window, event, reason, now)) + connection.executemany( + """INSERT OR IGNORE INTO forecast_capture_window_events + VALUES (?, ?, ?, ?, ?, ?, ?)""", + new_events, + ) + return counts diff --git a/backend/src/stocksweeper/forecast/evidence_reports.py b/backend/src/stocksweeper/forecast/evidence_reports.py index ec79281..43088e8 100644 --- a/backend/src/stocksweeper/forecast/evidence_reports.py +++ b/backend/src/stocksweeper/forecast/evidence_reports.py @@ -16,6 +16,7 @@ from typing import Any from options_api.market_calendar import session_close +from stocksweeper.forecast.capture_windows import capture_window_counts from options_api.intraday_capture import _COMPARATOR_VERSION from options_api.intraday_shadow import _VERSION as INTRADAY_VERSION from stocksweeper.forecast.calendar import SessionCalendar @@ -25,6 +26,11 @@ EMPIRICAL_SHADOW_VERSION, GJR_VERSION, STUDENT_VERSION, + HAR_VERSION, + SKEW_T_VERSION, + EGARCH_VERSION, + MARKOV_VERSION, + NGBOOST_VERSION, ) from stocksweeper.forecast.physical_evaluation import ContestRow, evaluate_band from stocksweeper.forecast.predictive import BASELINE_VERSION @@ -33,6 +39,13 @@ "empirical_scaled": EMPIRICAL_SHADOW_VERSION, "student_t_ewma": STUDENT_VERSION, "gjr_garch_t": GJR_VERSION, + "ohlc_har": HAR_VERSION, + "skew_t_ewma": SKEW_T_VERSION, + "egarch_skew_t": EGARCH_VERSION, + "markov_switching": MARKOV_VERSION, + "ngboost_pooled": NGBOOST_VERSION, + "earnings_jump": "earnings-jump-v1", + "iv_physical": "iv-physical-v1", } BANDS = {"1": range(1, 2), "2-5": range(2, 6), "6-25": range(6, 26)} @@ -232,10 +245,12 @@ def _summary( log_loss = dict(band_report["log_loss"]) if dates < 20: brier["bootstrap_95"] = None + brier["bootstrap_familywise_95"] = None log_loss["bootstrap_95"] = None if reference_only: brier["paired_delta"] = None brier["bootstrap_95"] = None + brier["bootstrap_familywise_95"] = None log_loss["paired_delta"] = None log_loss["bootstrap_95"] = None return { @@ -291,6 +306,7 @@ def build_model_evidence( calendar = SessionCalendar() # ponytail: cap the daily ledger scan at three years; use offline grouped reports if it grows. since = as_of.date() - timedelta(days=1096) + capture_counts = capture_window_counts(data_dir, as_of - timedelta(days=35), as_of) result: dict[tuple[str, str], dict[str, Any]] = {} for band in BANDS: baseline_rows = ledger_band_rows( @@ -379,6 +395,7 @@ def build_model_evidence( ), "coverage_basis": "recorded_intraday_contract_windows", "coverage_limit": report["denominator_limit"], + "capture_windows_all_horizons": capture_counts, "rejection_reasons": summary["rejection_reasons"], "brier": { "baseline": summary["mean"]["brier"]["dated_close"], diff --git a/backend/src/stocksweeper/forecast/physical_contest.py b/backend/src/stocksweeper/forecast/physical_contest.py index 32fa268..c3794b4 100644 --- a/backend/src/stocksweeper/forecast/physical_contest.py +++ b/backend/src/stocksweeper/forecast/physical_contest.py @@ -9,14 +9,20 @@ from collections import OrderedDict from dataclasses import dataclass, replace from datetime import date, datetime, timedelta -from math import exp, isfinite, sqrt +from math import exp, isfinite, log, pi, sqrt from time import perf_counter import numpy as np import polars as pl +from scipy.optimize import nnls from scipy.stats import t as student_t from stocksweeper.forecast.market import clean_completed, price_hash +from stocksweeper.forecast.pooled_ngboost import ( + VERSION as NGBOOST_VERSION, + load_pooled_model, + predict_pooled_params, +) from stocksweeper.forecast.predictive import ( BASELINE_VERSION, _BASELINE_QUANTILES, @@ -28,15 +34,22 @@ PredictiveDistribution, PredictiveForecaster, ) +from stocksweeper.forecast.provenance import SourceRightsUnverified SCENARIOS = 4096 MAX_HORIZON = 25 STUDENT_VERSION = "student-t-ewma60-shadow-v1" GJR_VERSION = "gjr-garch11-t-shadow-v1" EMPIRICAL_SHADOW_VERSION = "empirical-ewma60-shadow-v1" +HAR_VERSION = "ohlc-har-proxy-v1" +SKEW_T_VERSION = "skew-t-ewma60-v1" +EGARCH_VERSION = "egarch11-skewt-v1" +MARKOV_VERSION = "markov-switching-variance-v1" +NEW_METHODS = ("ohlc_har", "skew_t_ewma", "egarch_skew_t", "markov_switching", "ngboost_pooled") _MAX_FIT_CACHE = 512 _T_FIT_LIMIT_MS = 2000.0 _GJR_FIT_LIMIT_MS = 5000.0 +_NEW_FIT_LIMIT_MS = 5000.0 @dataclass(frozen=True) @@ -57,6 +70,18 @@ class _Fits: gjr_reason: str | None t_fit_ms: float gjr_fit_ms: float + har_parameters: tuple[float, ...] | None = None + har_reason: str | None = None + har_fit_ms: float = 0.0 + skew_parameters: tuple[float, ...] | None = None + skew_reason: str | None = None + skew_fit_ms: float = 0.0 + egarch_parameters: tuple[float, ...] | None = None + egarch_reason: str | None = None + egarch_fit_ms: float = 0.0 + markov_parameters: tuple[float, ...] | None = None + markov_reason: str | None = None + markov_fit_ms: float = 0.0 def _seed(ticker: str, as_of: date, method: str, horizon: int) -> int: @@ -83,7 +108,7 @@ def _fit_student( return None, "student_history_short" try: degrees, _, scale = student_t.fit(innovations, floc=0) - except (FloatingPointError, ValueError, RuntimeError): + except (FloatingPointError, ValueError, RuntimeError, np.linalg.LinAlgError): return None, "student_fit_failed" if not all(map(isfinite, (degrees, scale))) or not 2.05 < degrees <= 200 or scale <= 0: return None, "student_parameters_invalid" @@ -121,7 +146,10 @@ def _fit_gjr( beta = float(parameters["beta[1]"]) nu = float(parameters["nu"]) last_variance = float(result.conditional_volatility[-1]) ** 2 - except (FloatingPointError, KeyError, ValueError, RuntimeError, IndexError): + except ( + FloatingPointError, KeyError, ValueError, RuntimeError, IndexError, + np.linalg.LinAlgError, + ): return None, "gjr_fit_failed" values = (omega, alpha, gamma, beta, nu, last_variance) if ( @@ -129,13 +157,17 @@ def _fit_gjr( or omega <= 0 or alpha < 0 or beta < 0 - or alpha + gamma < 0 + # arch's optimizer can return a boundary value a few ulps below zero. + # Reject a material violation, but normalize harmless numerical noise. + or alpha + gamma < -1e-10 or alpha + gamma / 2 + beta >= 1 or nu <= 2.05 or last_variance <= 0 ): return None, "gjr_parameters_invalid" - return values, None + if alpha + gamma < 0: + gamma = -alpha + return (omega, alpha, gamma, beta, nu, last_variance), None def _student_terminal( @@ -184,12 +216,257 @@ def _gjr_terminal( return tuple(sorted(terminal.tolist())) +def _fit_har(frame: pl.DataFrame) -> tuple[tuple[float, ...] | None, str | None]: + """Fit next-day variance to lagged 1/5/22 daily OHLC variance proxies.""" + splits = np.asarray(frame["stock_splits"].to_list(), dtype=float) + split_index = np.flatnonzero(splits > 0) + if split_index.size: + frame = frame.slice(int(split_index[-1])) + if frame.height < 224: + return None, "har_history_short" + opens = np.asarray(frame["open"].to_list(), dtype=float) + highs = np.asarray(frame["high"].to_list(), dtype=float) + lows = np.asarray(frame["low"].to_list(), dtype=float) + closes = np.asarray(frame["close"].to_list(), dtype=float) + overnight = np.log(opens[1:] / closes[:-1]) + daily = overnight**2 + np.log(highs[1:] / lows[1:]) ** 2 / (4 * log(2)) + if not np.isfinite(daily).all() or np.any(daily < 0) or np.mean(daily) <= 1e-12: + return None, "har_inputs_invalid" + predictors = np.asarray( + [ + [1.0, daily[t - 1], daily[t - 5 : t].mean(), daily[t - 22 : t].mean()] + for t in range(22, len(daily)) + ] + ) + targets = daily[22:] + if len(targets) < 180: + return None, "har_history_short" + try: + coefficients, _ = nnls(predictors, targets) + except (FloatingPointError, ValueError, RuntimeError, np.linalg.LinAlgError): + return None, "har_fit_failed" + if not np.isfinite(coefficients).all() or coefficients.sum() <= 0: + return None, "har_parameters_invalid" + return tuple(map(float, (*coefficients, *daily[-22:]))), None + + +def _har_terminal( + spot: float, horizon: int, parameters: tuple[float, ...], seed: int +) -> tuple[float, ...]: + coefficients = np.asarray(parameters[:4]) + lagged = list(parameters[4:]) + if len(lagged) != 22: + raise ValueError("invalid HAR lags") + rng = np.random.default_rng(seed) + total = np.zeros(SCENARIOS) + for _ in range(horizon): + variance = float( + coefficients @ np.array([1.0, lagged[-1], np.mean(lagged[-5:]), np.mean(lagged)]) + ) + if not isfinite(variance) or variance <= 0: + raise ValueError("invalid HAR variance") + total += rng.standard_normal(SCENARIOS) * sqrt(variance) + lagged.pop(0) + lagged.append(variance) + return _terminal_prices(spot, total) + + +def _fit_arch_skew( + returns: np.ndarray, split_days: np.ndarray, *, egarch: bool +) -> tuple[tuple[float, ...] | None, str | None]: + label = "egarch" if egarch else "skew_ewma" + minimum = 500 if egarch else 180 + if split_days.any(): + returns = returns[int(np.flatnonzero(split_days)[-1]) + 1 :] + if len(returns) < minimum: + return ( + None, + f"{label}_split_safe_history_short" if split_days.any() else f"{label}_history_short", + ) + try: + from arch.univariate import EGARCH, EWMAVariance, SkewStudent, ZeroMean + except ImportError: + return None, "research_dependency_unavailable" + try: + model = ZeroMean(returns * 100) + model.volatility = EGARCH(p=1, o=1, q=1) if egarch else EWMAVariance(_EWMA_DECAY) + model.distribution = SkewStudent() + result = model.fit(disp="off", show_warning=False, options={"maxiter": 250}) + if result.convergence_flag != 0: + return None, f"{label}_nonconverged" + params = result.params + eta, skew = float(params["eta"]), float(params["lambda"]) + last_variance = float(result.conditional_volatility[-1]) ** 2 + if egarch: + omega = float(params["omega"]) + alpha = float(params["alpha[1]"]) + gamma = float(params["gamma[1]"]) + beta = float(params["beta[1]"]) + last_z = float(returns[-1] * 100 / sqrt(last_variance)) + values = (omega, alpha, gamma, beta, eta, skew, last_variance, last_z) + else: + values = (eta, skew, last_variance) + except ( + FloatingPointError, KeyError, ValueError, RuntimeError, IndexError, + np.linalg.LinAlgError, + ): + return None, f"{label}_fit_failed" + if ( + not all(map(isfinite, values)) + or eta <= 2.05 + or eta > 300 + or abs(skew) >= 0.99 + or last_variance <= 0 + or (egarch and (beta < 0 or beta >= 1)) + ): + return None, f"{label}_parameters_invalid" + return values, None + + +def _skew_terminal( + spot: float, horizon: int, parameters: tuple[float, ...], last_return: float, seed: int +) -> tuple[float, ...]: + from arch.univariate import SkewStudent + + eta, skew, last_variance = parameters + rng = np.random.default_rng(seed) + draw = SkewStudent(seed=rng).simulate([eta, skew]) + variance = np.full( + SCENARIOS, _EWMA_DECAY * last_variance + (1 - _EWMA_DECAY) * (last_return * 100) ** 2 + ) + total = np.zeros(SCENARIOS) + for _ in range(horizon): + realized = np.sqrt(variance) * draw(SCENARIOS) + total += realized / 100 + variance = _EWMA_DECAY * variance + (1 - _EWMA_DECAY) * realized**2 + return _terminal_prices(spot, total) + + +def _egarch_terminal( + spot: float, horizon: int, parameters: tuple[float, ...], seed: int +) -> tuple[float, ...]: + from arch.univariate import SkewStudent + + omega, alpha, gamma, beta, eta, skew, last_variance, last_z = parameters + rng = np.random.default_rng(seed) + draw = SkewStudent(seed=rng).simulate([eta, skew]) + log_variance = np.full(SCENARIOS, log(last_variance)) + previous_z = np.full(SCENARIOS, last_z) + total = np.zeros(SCENARIOS) + for _ in range(horizon): + log_variance = ( + omega + + alpha * (np.abs(previous_z) - sqrt(2 / pi)) + + gamma * previous_z + + beta * log_variance + ) + if not np.isfinite(log_variance).all() or np.any(np.abs(log_variance) > 50): + raise ValueError("invalid EGARCH variance") + previous_z = draw(SCENARIOS) + total += np.exp(log_variance / 2) * previous_z / 100 + return _terminal_prices(spot, total) + + +def _fit_markov( + returns: np.ndarray, split_days: np.ndarray +) -> tuple[tuple[float, ...] | None, str | None]: + if split_days.any(): + returns = returns[int(np.flatnonzero(split_days)[-1]) + 1 :] + if len(returns) < 500: + return ( + None, + "markov_split_safe_history_short" if split_days.any() else "markov_history_short", + ) + try: + from statsmodels.tsa.regime_switching.markov_regression import MarkovRegression + except ImportError: + return None, "research_dependency_unavailable" + try: + result = MarkovRegression( + returns * 100, k_regimes=2, trend="n", switching_variance=True + ).fit(disp=False, em_iter=5, maxiter=100, search_reps=0) + if not result.mle_retvals.get("converged", False): + return None, "markov_nonconverged" + named = dict(zip(result.model.param_names, result.params, strict=True)) + probabilities = np.asarray(result.filtered_marginal_probabilities[-1], dtype=float) + values = tuple( + map( + float, + ( + named["p[0->0]"], + named["p[1->0]"], + named["sigma2[0]"], + named["sigma2[1]"], + probabilities[0], + probabilities[1], + ), + ) + ) + except ( + FloatingPointError, KeyError, ValueError, RuntimeError, IndexError, + np.linalg.LinAlgError, + ): + return None, "markov_fit_failed" + p00, p10, var0, var1, p0, p1 = values + if ( + not all(map(isfinite, values)) + or not 0 <= p00 <= 1 + or not 0 <= p10 <= 1 + or var0 <= 0 + or var1 <= 0 + or not 0 <= p0 <= 1 + or not 0 <= p1 <= 1 + or abs(p0 + p1 - 1) > 1e-6 + ): + return None, "markov_parameters_invalid" + return values, None + + +def _markov_terminal( + spot: float, horizon: int, parameters: tuple[float, ...], seed: int +) -> tuple[float, ...]: + p00, p10, var0, var1, p0, _ = parameters + rng = np.random.default_rng(seed) + regime0 = rng.random(SCENARIOS) < p0 + total = np.zeros(SCENARIOS) + for _ in range(horizon): + regime0 = rng.random(SCENARIOS) < np.where(regime0, p00, p10) + total += rng.standard_normal(SCENARIOS) * np.sqrt(np.where(regime0, var0, var1)) / 100 + return _terminal_prices(spot, total) + + +def _terminal_prices(spot: float, log_returns: np.ndarray) -> tuple[float, ...]: + with np.errstate(over="ignore", invalid="ignore", under="ignore"): + prices = spot * np.exp(log_returns) + if not np.isfinite(prices).all() or np.any(prices <= 0): + raise ValueError("candidate scenarios invalid") + return tuple(sorted(prices.tolist())) + + class PhysicalShadowForecaster: """Fit once per verified ticker/session without fetching on lookup.""" def __init__(self, forecaster: PredictiveForecaster) -> None: self.forecaster = forecaster self._fits: OrderedDict[tuple[str, date, str], _Fits] = OrderedDict() + self._pooled_artifact: dict[str, object] | None = None + self._pooled_reason: str | None = None + self._pooled_checked = False + + def _pooled_model(self) -> tuple[dict[str, object] | None, str | None]: + if not self._pooled_checked: + self._pooled_checked = True + try: + load_pooled_model(self.forecaster.data_dir) + # The issuance ledger cannot yet identify which trained artifact produced a result. + self._pooled_reason = "training_artifact_traceability_unavailable" + except FileNotFoundError: + self._pooled_reason = "training_source_rights_unverified" + except SourceRightsUnverified: + self._pooled_reason = "training_source_rights_unverified" + except (OSError, ValueError, TypeError, KeyError): + self._pooled_reason = "pooled_artifact_invalid" + return self._pooled_artifact, self._pooled_reason def _fit( self, ticker: str, as_of: date, frame: pl.DataFrame, digest: str @@ -216,12 +493,54 @@ def _fit( t_ms = (perf_counter() - t_started) * 1000 if t_ms > _T_FIT_LIMIT_MS: t_parameters, t_reason = None, "student_fit_latency_exceeded" + try: + import arch # noqa: F401 - keep the cold import outside the fit latency budget + except ImportError: + pass # _fit_gjr reports the unavailable dependency gjr_started = perf_counter() gjr_parameters, gjr_reason = _fit_gjr(returns, split_days) gjr_ms = (perf_counter() - gjr_started) * 1000 if gjr_ms > _GJR_FIT_LIMIT_MS: gjr_parameters, gjr_reason = None, "gjr_fit_latency_exceeded" fits = _Fits(returns, t_parameters, t_reason, gjr_parameters, gjr_reason, t_ms, gjr_ms) + started_method = perf_counter() + params, reason = _fit_har(bounded) + elapsed = (perf_counter() - started_method) * 1000 + fits = replace( + fits, + har_parameters=params if elapsed <= _NEW_FIT_LIMIT_MS else None, + har_reason=reason if elapsed <= _NEW_FIT_LIMIT_MS else "har_fit_latency_exceeded", + har_fit_ms=elapsed, + ) + started_method = perf_counter() + params, reason = _fit_arch_skew(returns, split_days, egarch=False) + elapsed = (perf_counter() - started_method) * 1000 + fits = replace( + fits, + skew_parameters=params if elapsed <= _NEW_FIT_LIMIT_MS else None, + skew_reason=reason + if elapsed <= _NEW_FIT_LIMIT_MS + else "skew_ewma_fit_latency_exceeded", + skew_fit_ms=elapsed, + ) + started_method = perf_counter() + params, reason = _fit_arch_skew(returns, split_days, egarch=True) + elapsed = (perf_counter() - started_method) * 1000 + fits = replace( + fits, + egarch_parameters=params if elapsed <= _NEW_FIT_LIMIT_MS else None, + egarch_reason=reason if elapsed <= _NEW_FIT_LIMIT_MS else "egarch_fit_latency_exceeded", + egarch_fit_ms=elapsed, + ) + started_method = perf_counter() + params, reason = _fit_markov(returns, split_days) + elapsed = (perf_counter() - started_method) * 1000 + fits = replace( + fits, + markov_parameters=params if elapsed <= _NEW_FIT_LIMIT_MS else None, + markov_reason=reason if elapsed <= _NEW_FIT_LIMIT_MS else "markov_fit_latency_exceeded", + markov_fit_ms=elapsed, + ) self._fits[key] = fits while len(self._fits) > _MAX_FIT_CACHE: self._fits.popitem(last=False) @@ -236,7 +555,7 @@ def cached_candidate( ) -> ShadowForecast: """Build only one candidate from an existing fit; never optimize here.""" started = perf_counter() - if method not in ("empirical_scaled", "student_t_ewma", "gjr_garch_t"): + if method not in ("empirical_scaled", "student_t_ewma", "gjr_garch_t", *NEW_METHODS): raise ValueError("unknown physical challenger") key = (current.ticker, current.as_of, current.data_hash) fits = self._fits.get(key) @@ -280,7 +599,7 @@ def cached_candidate( except ValueError: return ShadowForecast(None, "candidate_scenarios_invalid", fits.t_fit_ms, 0) version, fit_ms = STUDENT_VERSION, fits.t_fit_ms - else: + elif method == "gjr_garch_t": if fits.gjr_parameters is None: return ShadowForecast(None, fits.gjr_reason, fits.gjr_fit_ms, 0) try: @@ -294,6 +613,72 @@ def cached_candidate( except ValueError: return ShadowForecast(None, "candidate_scenarios_invalid", fits.gjr_fit_ms, 0) version, fit_ms = GJR_VERSION, fits.gjr_fit_ms + elif method == "ohlc_har": + if fits.har_parameters is None: + return ShadowForecast(None, fits.har_reason, fits.har_fit_ms, 0) + try: + prices = _har_terminal( + current.spot, + horizon, + fits.har_parameters, + _seed(current.ticker, current.as_of, method, horizon), + ) + except ValueError: + return ShadowForecast(None, "candidate_scenarios_invalid", fits.har_fit_ms, 0) + version, fit_ms = HAR_VERSION, fits.har_fit_ms + elif method == "skew_t_ewma": + if fits.skew_parameters is None: + return ShadowForecast(None, fits.skew_reason, fits.skew_fit_ms, 0) + try: + prices = _skew_terminal( + current.spot, + horizon, + fits.skew_parameters, + float(fits.returns[-1]), + _seed(current.ticker, current.as_of, method, horizon), + ) + except ValueError: + return ShadowForecast(None, "candidate_scenarios_invalid", fits.skew_fit_ms, 0) + version, fit_ms = SKEW_T_VERSION, fits.skew_fit_ms + elif method == "egarch_skew_t": + if fits.egarch_parameters is None: + return ShadowForecast(None, fits.egarch_reason, fits.egarch_fit_ms, 0) + try: + prices = _egarch_terminal( + current.spot, + horizon, + fits.egarch_parameters, + _seed(current.ticker, current.as_of, method, horizon), + ) + except ValueError: + return ShadowForecast(None, "candidate_scenarios_invalid", fits.egarch_fit_ms, 0) + version, fit_ms = EGARCH_VERSION, fits.egarch_fit_ms + elif method == "markov_switching": + if fits.markov_parameters is None: + return ShadowForecast(None, fits.markov_reason, fits.markov_fit_ms, 0) + try: + prices = _markov_terminal( + current.spot, + horizon, + fits.markov_parameters, + _seed(current.ticker, current.as_of, method, horizon), + ) + except ValueError: + return ShadowForecast(None, "candidate_scenarios_invalid", fits.markov_fit_ms, 0) + version, fit_ms = MARKOV_VERSION, fits.markov_fit_ms + else: + if clean is None or price_hash(clean) != current.data_hash: + return ShadowForecast(None, "candidate_input_unverified", 0, 0) + artifact, reason = self._pooled_model() + if artifact is None: + return ShadowForecast(None, reason, 0, 0) + try: + mean, scale = predict_pooled_params(artifact, clean, horizon, current.as_of) + rng = np.random.default_rng(_seed(current.ticker, current.as_of, method, horizon)) + prices = _terminal_prices(current.spot, rng.normal(mean, scale, SCENARIOS)) + except ValueError as exc: + return ShadowForecast(None, str(exc), 0, 0) + version, fit_ms = NGBOOST_VERSION, 0.0 if not prices or not all(isfinite(price) and price > 0 for price in prices): return ShadowForecast(None, "candidate_scenarios_invalid", fit_ms, 0) distribution = replace( @@ -305,7 +690,10 @@ def cached_candidate( weights=(1 / len(prices),) * len(prices), ) return ShadowForecast( - distribution, None, fit_ms, (perf_counter() - started) * 1000, + distribution, + None, + fit_ms, + (perf_counter() - started) * 1000, independent_blocks if method == "empirical_scaled" else None, ) @@ -318,7 +706,13 @@ def forecast_candidates( contract_since: date | None = None, standard_terms: bool = True, ) -> dict[str, ShadowForecast]: - names = ("lognormal_ewma", "empirical_scaled", "student_t_ewma", "gjr_garch_t") + names = ( + "lognormal_ewma", + "empirical_scaled", + "student_t_ewma", + "gjr_garch_t", + *NEW_METHODS, + ) current = self.forecaster.forecast( ticker, as_of, expiry, contract_since=contract_since, standard_terms=standard_terms ) @@ -345,10 +739,12 @@ def forecast_candidates( "lognormal_ewma": ShadowForecast(baseline, None, 0, (perf_counter() - started) * 1000) } if horizon > MAX_HORIZON: - results.update({ - name: ShadowForecast(None, "shadow_horizon_unsupported", 0, 0) - for name in names[1:] - }) + results.update( + { + name: ShadowForecast(None, "shadow_horizon_unsupported", 0, 0) + for name in names[1:] + } + ) return results try: frame = self.forecaster.prices.read(ticker) diff --git a/backend/src/stocksweeper/forecast/physical_evaluation.py b/backend/src/stocksweeper/forecast/physical_evaluation.py index 26c0df8..51c28a5 100644 --- a/backend/src/stocksweeper/forecast/physical_evaluation.py +++ b/backend/src/stocksweeper/forecast/physical_evaluation.py @@ -16,6 +16,7 @@ Band = Literal["1", "2-5", "6-25"] _BANDS: dict[Band, tuple[int, int]] = {"1": (1, 1), "2-5": (2, 5), "6-25": (6, 25)} _BASELINE = "lognormal_ewma" +_PREDECLARED_PHYSICAL_COMPARISONS = 10 @dataclass(frozen=True) @@ -284,6 +285,7 @@ def evaluate_band( date_blocks = sorted(by_date) brier_ci: tuple[float, float] | None = None log_ci: tuple[float, float] | None = None + brier_familywise_ci: tuple[float, float] | None = None if date_blocks: rng = np.random.default_rng(20260927) sampled = rng.integers(0, len(date_blocks), size=(bootstrap_samples, len(date_blocks))) @@ -299,6 +301,11 @@ def evaluate_band( log_boot = log_sums[sampled].sum(axis=1) / sampled_counts brier_ci = (_quantile(brier_boot.tolist(), 0.025), _quantile(brier_boot.tolist(), 0.975)) log_ci = (_quantile(log_boot.tolist(), 0.025), _quantile(log_boot.tolist(), 0.975)) + tail = 0.05 / (2 * _PREDECLARED_PHYSICAL_COMPARISONS) + brier_familywise_ci = ( + _quantile(brier_boot.tolist(), tail), + _quantile(brier_boot.tolist(), 1 - tail), + ) subgroup_units: dict[tuple[str, str], list[tuple[date, float]]] = defaultdict(list) for (category, value, _, origin, _), deltas in subgroup_values.items(): subgroup_units[(category, value)].append((origin, mean(deltas))) @@ -351,6 +358,8 @@ def calibration(bins: list[list[tuple[float, float]]]) -> list[dict[str, float | "candidate": mean(item["brier_candidate"] for item in units) if units else None, "paired_delta": brier_delta, "bootstrap_95": brier_ci, + "bootstrap_familywise_95": brier_familywise_ci, + "comparison_count": _PREDECLARED_PHYSICAL_COMPARISONS, }, "log_loss": { "baseline": mean(item["log_baseline"] for item in units) if units else None, diff --git a/backend/src/stocksweeper/forecast/pooled_ngboost.py b/backend/src/stocksweeper/forecast/pooled_ngboost.py new file mode 100644 index 0000000..932ac57 --- /dev/null +++ b/backend/src/stocksweeper/forecast/pooled_ngboost.py @@ -0,0 +1,362 @@ +"""Offline NGBoost training and safe, bounded live inference. + +The current public Yahoo cohort is deliberately not licensed for this training +path. A qualified frozen cohort is required before an artifact can be built. +""" + +from __future__ import annotations + +import hashlib +import json +from datetime import date +from math import exp, isfinite, log +from pathlib import Path +from uuid import uuid4 + +import numpy as np +import polars as pl + +from stocksweeper.forecast.provenance import load_training_cohort + +VERSION = "ngboost-pooled-v1" +FEATURES = ( + "horizon", + "last_return", + "mean_return_5", + "mean_return_22", + "volatility_5", + "volatility_22", + "range_variance_5", + "range_variance_22", + "volume_ratio_log", +) +MAX_ARTIFACT_BYTES = 2_000_000 +MAX_STAGES = 100 +MIN_TICKERS = 20 + + +def _features(frame: pl.DataFrame, index: int, horizon: int) -> tuple[float, ...] | None: + if index < 22 or not 1 <= horizon <= 25: + return None + tail = frame.slice(index - 22, 23) + if tail.height != 23 or any(float(value) > 0 for value in tail["stock_splits"]): + return None + close = np.asarray(tail["close"].to_list(), dtype=float) + high = np.asarray(tail["high"].to_list(), dtype=float) + low = np.asarray(tail["low"].to_list(), dtype=float) + volume = np.asarray(tail["volume"].to_list(), dtype=float) + if ( + not all(np.isfinite(values).all() for values in (close, high, low, volume)) + or np.any(close <= 0) + or np.any(low <= 0) + or np.any(volume <= 0) + ): + return None + returns = np.diff(np.log(close)) + range_variance = np.log(high[1:] / low[1:]) ** 2 / (4 * log(2)) + features = ( + float(horizon), + float(returns[-1]), + float(returns[-5:].mean()), + float(returns.mean()), + float(returns[-5:].std()), + float(returns.std()), + float(range_variance[-5:].mean()), + float(range_variance.mean()), + float(log(volume[-1] / volume[1:].mean())), + ) + return features if all(map(isfinite, features)) else None + + +def _cohort_digest(cohort: tuple[tuple[str, pl.DataFrame, str], ...]) -> str: + return hashlib.sha256( + "|".join(f"{ticker}:{digest}" for ticker, _, digest in sorted(cohort)).encode() + ).hexdigest() + + +def _training_rows( + cohort: tuple[tuple[str, pl.DataFrame, str], ...], +) -> tuple[np.ndarray, np.ndarray, date, str]: + if len(cohort) < MIN_TICKERS or len(cohort) > 100: + raise ValueError("pooled_training_cohort_size_invalid") + if len({ticker for ticker, _, _ in cohort}) != len(cohort): + raise ValueError("pooled_training_cohort_duplicate") + through = {frame["ts"][-1] for _, frame, _ in cohort} + if len(through) != 1: + raise ValueError("pooled_training_vintage_mismatch") + inputs: list[tuple[float, ...]] = [] + targets: list[float] = [] + for _, frame, _ in cohort: + if frame.height < 500 or frame.height > 756: + raise ValueError("pooled_training_history_invalid") + close = np.asarray(frame["close"].to_list(), dtype=float) + splits = np.asarray(frame["stock_splits"].to_list(), dtype=float) + # Fixed, evenly spaced origins keep training size and ticker weight bounded. + for origin in sorted(set(np.linspace(60, frame.height - 26, 12, dtype=int))): + features = _features(frame, int(origin), 1) + if features is None: + continue + for horizon in range(1, 26): + if np.any(splits[origin + 1 : origin + horizon + 1] > 0): + continue + outcome = float(log(close[origin + horizon] / close[origin])) + if isfinite(outcome): + inputs.append((float(horizon), *features[1:])) + targets.append(outcome) + if len(inputs) < 5_000: + raise ValueError("pooled_training_support_insufficient") + return np.asarray(inputs), np.asarray(targets), through.pop(), _cohort_digest(cohort) + + +def _tree_payload(tree: object) -> dict[str, list[float] | list[int]]: + raw = tree.tree_ + return { + "left": raw.children_left.tolist(), + "right": raw.children_right.tolist(), + "feature": raw.feature.tolist(), + "threshold": raw.threshold.tolist(), + "value": raw.value[:, 0, 0].tolist(), + } + + +def _export_model( + fitted: object, + through: date, + cohort_hash: str, + bounds: tuple[np.ndarray, np.ndarray], + rights: dict[str, str], +) -> dict[str, object]: + return { + "version": VERSION, + "features": FEATURES, + "training_through": through.isoformat(), + "cohort_hash": cohort_hash, + "cohort_manifest_hash": rights["manifest_hash"], + "rights_status": "approved_for_training", + "source_name": rights["source_name"], + "license_reference": rights["license_reference"], + "initial": np.asarray(fitted.init_params, dtype=float).tolist(), + "learning_rate": float(fitted.learning_rate), + "lower": bounds[0].tolist(), + "upper": bounds[1].tolist(), + "stages": [ + { + "scaling": float(scale), + "columns": np.asarray(columns, dtype=int).tolist(), + "trees": [_tree_payload(tree) for tree in models], + } + for models, scale, columns in zip( + fitted.base_models, fitted.scalings, fitted.col_idxs, strict=True + ) + ], + } + + +def _validate_model(payload: dict[str, object]) -> None: + if ( + payload.get("version") != VERSION + or tuple(payload.get("features", ())) != FEATURES + or payload.get("rights_status") != "approved_for_training" + or not isinstance(payload.get("source_name"), str) + or not payload["source_name"] + or not isinstance(payload.get("license_reference"), str) + or not payload["license_reference"] + or not isinstance(payload.get("cohort_hash"), str) + or len(payload["cohort_hash"]) != 64 + or not isinstance(payload.get("cohort_manifest_hash"), str) + or len(payload["cohort_manifest_hash"]) != 64 + or any(character not in "0123456789abcdef" for character in payload["cohort_hash"]) + or any(character not in "0123456789abcdef" for character in payload["cohort_manifest_hash"]) + ): + raise ValueError("pooled_artifact_invalid") + date.fromisoformat(str(payload["training_through"])) + initial = payload["initial"] + lower, upper = payload["lower"], payload["upper"] + rate = float(payload["learning_rate"]) + stages = payload["stages"] + if ( + len(initial) != 2 + or len(lower) != len(FEATURES) + or len(upper) != len(FEATURES) + or not 0 < rate <= 1 + or not 1 <= len(stages) <= MAX_STAGES + or not all(isfinite(float(value)) for value in (*initial, *lower, *upper)) + or any(float(low) > float(high) for low, high in zip(lower, upper, strict=True)) + ): + raise ValueError("pooled_artifact_invalid") + for stage in stages: + columns = stage["columns"] + if ( + not isfinite(float(stage["scaling"])) + or not columns + or any( + not isinstance(column, int) or not 0 <= column < len(FEATURES) for column in columns + ) + or len(stage["trees"]) != 2 + ): + raise ValueError("pooled_artifact_invalid") + for tree in stage["trees"]: + left, right, feature, threshold, value = ( + tree[name] for name in ("left", "right", "feature", "threshold", "value") + ) + size = len(left) + if ( + not 1 <= size <= 63 + or any(len(items) != size for items in (right, feature, threshold, value)) + or not all(isfinite(float(item)) for item in (*threshold, *value)) + ): + raise ValueError("pooled_artifact_invalid") + for node in range(size): + if left[node] == -1 and right[node] == -1: + continue + if ( + not isinstance(left[node], int) + or not isinstance(right[node], int) + or not node < left[node] < size + or not node < right[node] < size + or not isinstance(feature[node], int) + or not 0 <= feature[node] < len(columns) + ): + raise ValueError("pooled_artifact_invalid") + + +def _artifact_dir(data_dir: Path) -> Path: + return data_dir / "forecast" / "training" / "ngboost" + + +def train_pooled_model(data_dir: Path) -> Path: + """Train only from a rights-qualified immutable cohort; never from live requests.""" + cohort = load_training_cohort(data_dir, require_training_rights=True) + metadata = json.loads((data_dir / "forecast" / "training" / "cohort.json").read_text()) + if ( + metadata.get("rights_status") != "approved_for_training" + or not metadata.get("source_name") + or not metadata.get("license_reference") + or not metadata.get("manifest_hash") + ): + raise ValueError("training_source_rights_unverified") + from ngboost import NGBRegressor + from ngboost.distns import Normal + from sklearn.tree import DecisionTreeRegressor + + inputs, targets, through, cohort_hash = _training_rows(cohort) + fitted = NGBRegressor( + Dist=Normal, + Base=DecisionTreeRegressor(max_depth=2, min_samples_leaf=40, random_state=0), + n_estimators=60, + learning_rate=0.05, + random_state=0, + verbose=False, + ).fit(inputs, targets) + margins = np.maximum(np.ptp(inputs, axis=0) * 0.25, 1e-12) + payload = _export_model( + fitted, + through, + cohort_hash, + (inputs.min(axis=0) - margins, inputs.max(axis=0) + margins), + metadata, + ) + _validate_model(payload) + encoded = json.dumps(payload, sort_keys=True, separators=(",", ":"), allow_nan=False).encode() + if len(encoded) > MAX_ARTIFACT_BYTES: + raise ValueError("pooled_artifact_too_large") + digest = hashlib.sha256(encoded).hexdigest() + root = _artifact_dir(data_dir) + root.mkdir(parents=True, exist_ok=True) + output = root / f"model-{digest}.json" + if output.exists(): + if output.read_bytes() != encoded: + raise ValueError("pooled_artifact_hash_conflict") + else: + temporary = root / f".{uuid4().hex}.tmp" + temporary.write_bytes(encoded) + temporary.replace(output) + pointer = root / "latest.json" + temporary = root / f".{uuid4().hex}.tmp" + temporary.write_text(json.dumps({"sha256": digest}, sort_keys=True)) + temporary.replace(pointer) + return output + + +def load_pooled_model(data_dir: Path) -> dict[str, object]: + root = _artifact_dir(data_dir) + pointer = root / "latest.json" + if not pointer.exists(): + raise FileNotFoundError("pooled_training_artifact_unavailable") + if pointer.stat().st_size > 1024: + raise ValueError("pooled_artifact_invalid") + digest = json.loads(pointer.read_text())["sha256"] + if ( + not isinstance(digest, str) + or len(digest) != 64 + or any(character not in "0123456789abcdef" for character in digest) + ): + raise ValueError("pooled_artifact_invalid") + path = root / f"model-{digest}.json" + if not path.is_file(): + raise ValueError("pooled_artifact_invalid") + if path.stat().st_size > MAX_ARTIFACT_BYTES: + raise ValueError("pooled_artifact_invalid") + encoded = path.read_bytes() + if hashlib.sha256(encoded).hexdigest() != digest: + raise ValueError("pooled_artifact_invalid") + payload = json.loads(encoded) + if not isinstance(payload, dict): + raise ValueError("pooled_artifact_invalid") + _validate_model(payload) + # An artifact's self-described license is insufficient: verify the frozen + # cohort with the qualified loader before making its predictions live. + cohort = load_training_cohort(data_dir, require_training_rights=True) + manifest = json.loads((data_dir / "forecast" / "training" / "cohort.json").read_text()) + if ( + _cohort_digest(cohort) != payload["cohort_hash"] + or manifest.get("manifest_hash") != payload["cohort_manifest_hash"] + or manifest.get("source_name") != payload["source_name"] + or manifest.get("license_reference") != payload["license_reference"] + ): + raise ValueError("pooled_artifact_provenance_mismatch") + return payload + + +def _tree_predict( + tree: dict[str, object], columns: list[int], features: tuple[float, ...] +) -> float: + node = 0 + for _ in range(len(tree["left"])): + if tree["left"][node] == -1: + return float(tree["value"][node]) + feature = features[columns[tree["feature"][node]]] + node = tree["left"][node] if feature <= tree["threshold"][node] else tree["right"][node] + raise ValueError("pooled_artifact_invalid") + + +def predict_pooled_params( + payload: dict[str, object], frame: pl.DataFrame, horizon: int, as_of: date +) -> tuple[float, float]: + if date.fromisoformat(payload["training_through"]) > as_of: + raise ValueError("pooled_artifact_future_leak") + features = _features(frame, frame.height - 1, horizon) + if features is None: + raise ValueError("pooled_inputs_invalid") + if any( + value < low or value > high + for value, low, high in zip(features, payload["lower"], payload["upper"], strict=True) + ): + raise ValueError("pooled_inputs_out_of_domain") + parameters = [float(value) for value in payload["initial"]] + for stage in payload["stages"]: + factor = payload["learning_rate"] * stage["scaling"] + for index, tree in enumerate(stage["trees"]): + parameters[index] -= factor * _tree_predict(tree, stage["columns"], features) + mean, log_scale = parameters + if not all(map(isfinite, parameters)) or not -20 < log_scale < 5: + raise ValueError("pooled_parameters_invalid") + return mean, exp(log_scale) + + +if __name__ == "__main__": + import sys + + if len(sys.argv) != 2: + raise SystemExit("usage: python -m stocksweeper.forecast.pooled_ngboost DATA_DIR") + print(train_pooled_model(Path(sys.argv[1]))) diff --git a/backend/src/stocksweeper/forecast/predictive.py b/backend/src/stocksweeper/forecast/predictive.py index 0ec9a6f..b6346d2 100644 --- a/backend/src/stocksweeper/forecast/predictive.py +++ b/backend/src/stocksweeper/forecast/predictive.py @@ -86,7 +86,18 @@ class PredictiveDistribution: reason: str | None method: ( Literal[ - "lognormal_ewma", "empirical_scaled", "student_t_ewma", "gjr_garch_t", "intraday_shadow" + "lognormal_ewma", + "empirical_scaled", + "student_t_ewma", + "gjr_garch_t", + "ohlc_har", + "skew_t_ewma", + "egarch_skew_t", + "markov_switching", + "ngboost_pooled", + "earnings_jump", + "iv_physical", + "intraday_shadow", ] | None ) @@ -700,6 +711,9 @@ def forecast_baseline( standard_terms: bool = True, ) -> PredictiveDistribution: return self.forecast( - ticker, as_of, expiry, contract_since=contract_since, + ticker, + as_of, + expiry, + contract_since=contract_since, standard_terms=standard_terms, ) diff --git a/backend/src/stocksweeper/forecast/provenance.py b/backend/src/stocksweeper/forecast/provenance.py new file mode 100644 index 0000000..8a2ace9 --- /dev/null +++ b/backend/src/stocksweeper/forecast/provenance.py @@ -0,0 +1,272 @@ +"""Immutable, hash-verified local inputs for retrospective model training. + +These are frozen from an existing verified cache, never fetched here. They are +current-vintage replay inputs, not evidence of what was available at an old origin. +""" + +from __future__ import annotations + +import hashlib +import json +from collections.abc import Iterable +from datetime import UTC, date, datetime +from pathlib import Path + +import polars as pl + +from options_api.models import normalize_ticker +from stocksweeper.forecast.audit import AuditCohort, _write_once, read_audit_cohort +from stocksweeper.forecast.calendar import SessionCalendar +from stocksweeper.forecast.market import ( + CacheIntegrityError, + ForecastPriceStore, + PRICE_COLUMNS, + YahooForecastProvider, + clean_completed, + price_hash, +) + +TRAINING_SIZE = 100 +TRAINING_BARS = 756 +MIN_TRAINING_BARS = 500 +_PROVENANCE = "immutable_current_vintage_training" +_UNQUALIFIED_RIGHTS = "training_rights_unverified" +_APPROVED_RIGHTS = "approved_for_training" + + +class SourceRightsUnverified(ValueError): + """A valid local snapshot does not grant permission to train on its source.""" + + +def _has_approval(metadata: dict[str, object]) -> bool: + source = metadata.get("source_name") + reference = metadata.get("license_reference") + return ( + isinstance(source, str) + and bool(source.strip()) + and isinstance(reference, str) + and bool(reference.strip()) + and metadata.get("source") == source + ) + + +def _vintage_paths(data_dir: Path, ticker: str, session: date, digest: str) -> tuple[Path, Path]: + if ( + normalize_ticker(ticker) != ticker + or len(digest) != 64 + or any(character not in "0123456789abcdef" for character in digest) + ): + raise ValueError("invalid price vintage identity") + root = data_dir / "forecast" / "vintages" / ticker / session.isoformat() + return root / f"{digest}.parquet", root / f"{digest}.json" + + +def read_price_vintage( + data_dir: Path, ticker: str, session: date, digest: str +) -> tuple[pl.DataFrame, dict[str, str]]: + path, manifest = _vintage_paths(data_dir, ticker, session, digest) + try: + metadata = json.loads(manifest.read_text()) + frame = pl.read_parquet(path) + expected_schema = {name: pl.Date if name == "ts" else pl.Float64 for name in PRICE_COLUMNS} + rights = metadata.get("rights_status") if isinstance(metadata, dict) else None + manifest_digest = metadata.get("manifest_hash") if isinstance(metadata, dict) else None + signed_metadata = ( + {key: value for key, value in metadata.items() if key != "manifest_hash"} + if isinstance(metadata, dict) + else {} + ) + if ( + not isinstance(metadata, dict) + or metadata.get("ticker") != ticker + or metadata.get("through_session") != session.isoformat() + or metadata.get("data_hash") != digest + or metadata.get("provenance") != _PROVENANCE + or rights not in (_UNQUALIFIED_RIGHTS, _APPROVED_RIGHTS) + or (rights == _APPROVED_RIGHTS and not _has_approval(metadata)) + or manifest_digest != _manifest_hash(signed_metadata) + or metadata.get("price_basis") != "split-normalized, dividend-unadjusted" + or frame.schema != expected_schema + or frame.is_empty() + or frame["ts"][-1] != session + or tuple(frame["ts"]) != SessionCalendar().sessions(frame["ts"][0], session) + or clean_completed(frame, session, SessionCalendar()).height != frame.height + or price_hash(frame) != digest + or datetime.fromisoformat(metadata["source_retrieved_at"]).tzinfo is None + or datetime.fromisoformat(metadata["frozen_at"]).tzinfo is None + ): + raise ValueError("vintage manifest does not match prices") + except (OSError, KeyError, TypeError, ValueError, pl.exceptions.PolarsError) as exc: + raise CacheIntegrityError(f"invalid immutable price vintage for {ticker}") from exc + return frame, metadata + + +def freeze_price_vintage( + data_dir: Path, ticker: str, completed_session: date +) -> tuple[pl.DataFrame, str]: + """Copy up to three years from an already-local, verified completed cache.""" + store = ForecastPriceStore(data_dir, YahooForecastProvider()) + source = store.read(ticker) + if source is None or source["ts"][-1] != completed_session: + raise ValueError("verified latest completed price bar is unavailable") + frame = source.tail(TRAINING_BARS) + calendar = SessionCalendar() + if ( + frame.height < MIN_TRAINING_BARS + or tuple(frame["ts"]) != calendar.sessions(frame["ts"][0], completed_session) + or clean_completed(frame, completed_session, calendar).height != frame.height + or any(value > 0 for value in frame["stock_splits"]) + ): + raise ValueError("contiguous split-safe training history is unavailable") + try: + source_metadata = json.loads(store.path(ticker).with_suffix(".json").read_text()) + if ( + not isinstance(source_metadata, dict) + or source_metadata.get("hash") != price_hash(source) + or source_metadata.get("through_session") != completed_session.isoformat() + ): + raise ValueError("price cache changed while freezing vintage") + retrieved_at = datetime.fromisoformat(source_metadata["retrieved_at"]) + if retrieved_at.tzinfo is None: + raise ValueError("source retrieval time is not timezone-aware") + except (OSError, KeyError, TypeError, ValueError) as exc: + raise CacheIntegrityError("price cache changed while freezing vintage") from exc + digest = price_hash(frame) + path, manifest = _vintage_paths(data_dir, ticker, completed_session, digest) + _write_once(path, frame.write_parquet) + metadata = { + "ticker": ticker, + "through_session": completed_session.isoformat(), + "data_hash": digest, + "source_hash": source_metadata["hash"], + "source": source_metadata["source"], + "source_retrieved_at": retrieved_at.astimezone(UTC).isoformat(), + "frozen_at": datetime.now(UTC).isoformat(), + "price_basis": source_metadata["price_basis"], + "provenance": _PROVENANCE, + "rights_status": _UNQUALIFIED_RIGHTS, + } + metadata["manifest_hash"] = _manifest_hash(metadata) + _write_once( + manifest, lambda temporary: temporary.write_text(json.dumps(metadata, sort_keys=True)) + ) + verified, _ = read_price_vintage(data_dir, ticker, completed_session, digest) + return verified, digest + + +def _cohort_manifest(data_dir: Path) -> Path: + return data_dir / "forecast" / "training" / "cohort.json" + + +def _manifest_hash(metadata: dict[str, object]) -> str: + encoded = json.dumps(metadata, sort_keys=True, separators=(",", ":")).encode() + return hashlib.sha256(encoded).hexdigest() + + +def _audit_identity(audit: AuditCohort) -> str: + return _manifest_hash( + {"members": sorted((member.ticker, member.data_hash) for member in audit.members)} + ) + + +def load_training_cohort( + data_dir: Path, *, require_training_rights: bool = True +) -> tuple[tuple[str, pl.DataFrame, str], ...]: + """Load a frozen cohort, rejecting changed members and changed manifest.""" + try: + metadata = json.loads(_cohort_manifest(data_dir).read_text()) + digest = metadata.pop("manifest_hash") + if ( + metadata.get("provenance") != _PROVENANCE + or metadata.get("rights_status") not in (_UNQUALIFIED_RIGHTS, _APPROVED_RIGHTS) + or _manifest_hash(metadata) != digest + or len(metadata["members"]) != metadata["size"] + ): + raise ValueError("invalid training cohort manifest") + rights = metadata["rights_status"] + if rights == _APPROVED_RIGHTS and not _has_approval(metadata): + raise ValueError("approved source and license reference are required") + session = date.fromisoformat(metadata["completed_session"]) + audit = read_audit_cohort(data_dir) + if audit is None or metadata.get("audit_cohort_hash") != _audit_identity(audit): + raise ValueError("separate audit cohort changed or disappeared") + excluded = {member.ticker for member in audit.members} + result = [] + for member in metadata["members"]: + ticker, member_hash = member["ticker"], member["data_hash"] + if ticker in excluded: + raise ValueError("training cohort overlaps audit cohort") + frame, vintage = read_price_vintage(data_dir, ticker, session, member_hash) + if vintage["rights_status"] != rights or ( + rights == _APPROVED_RIGHTS + and ( + vintage["source_name"] != metadata["source_name"] + or vintage["license_reference"] != metadata["license_reference"] + ) + ): + raise ValueError("training member source rights differ from cohort") + result.append((ticker, frame, member_hash)) + if len({ticker for ticker, _, _ in result}) != len(result): + raise ValueError("duplicate training cohort ticker") + if require_training_rights and rights != _APPROVED_RIGHTS: + raise SourceRightsUnverified(_UNQUALIFIED_RIGHTS) + return tuple(result) + except SourceRightsUnverified: + raise + except (OSError, KeyError, TypeError, ValueError, CacheIntegrityError) as exc: + raise CacheIntegrityError("invalid immutable training cohort") from exc + + +def freeze_training_cohort( + data_dir: Path, + eligible_tickers: Iterable[str], + completed_session: date, + *, + size: int = TRAINING_SIZE, + exclude_tickers: Iterable[str] = (), +) -> tuple[tuple[str, pl.DataFrame, str], ...]: + """Deterministically freeze only verified local Nasdaq-member caches.""" + manifest = _cohort_manifest(data_dir) + if manifest.exists(): + existing = load_training_cohort(data_dir, require_training_rights=False) + if len(existing) != size or any( + frame["ts"][-1] != completed_session for _, frame, _ in existing + ): + raise ValueError("existing frozen cohort has another size or session") + return existing + if size <= 0: + raise ValueError("training cohort size must be positive") + audit = read_audit_cohort(data_dir) + if audit is None: + raise ValueError("freeze the separate audit cohort first") + excluded = set(exclude_tickers) | {member.ticker for member in audit.members} + eligible = sorted(set(eligible_tickers) - excluded) + ranked = sorted( + (ticker for ticker in eligible if normalize_ticker(ticker) == ticker), + key=lambda ticker: (hashlib.sha256(ticker.encode()).hexdigest(), ticker), + ) + members = [] + for ticker in ranked: + try: + _, digest = freeze_price_vintage(data_dir, ticker, completed_session) + except (CacheIntegrityError, OSError, ValueError): + continue + members.append({"ticker": ticker, "data_hash": digest}) + if len(members) == size: + break + if len(members) != size: + raise ValueError(f"only {len(members)} verified disjoint training caches; need {size}") + metadata: dict[str, object] = { + "provenance": _PROVENANCE, + "rights_status": _UNQUALIFIED_RIGHTS, + "audit_cohort_hash": _audit_identity(audit), + "completed_session": completed_session.isoformat(), + "size": size, + "members": members, + "frozen_at": datetime.now(UTC).isoformat(), + } + metadata["manifest_hash"] = _manifest_hash(metadata) + _write_once( + manifest, lambda temporary: temporary.write_text(json.dumps(metadata, sort_keys=True)) + ) + return load_training_cohort(data_dir, require_training_rights=False) diff --git a/backend/src/stocksweeper/forecast/sec_events.py b/backend/src/stocksweeper/forecast/sec_events.py new file mode 100644 index 0000000..38d5449 --- /dev/null +++ b/backend/src/stocksweeper/forecast/sec_events.py @@ -0,0 +1,630 @@ +"""Conservative point-in-time SEC earnings schedule evidence. + +An 8-K announcing a *future* date is useful. An earnings release or a filing +accepted on the event date is not proof that its date was known beforehand. +""" + +from __future__ import annotations + +import hashlib +import json +import re +import threading +import time as clock +from collections.abc import Callable +from dataclasses import dataclass +from datetime import UTC, date, datetime +from html.parser import HTMLParser +from pathlib import Path +from zoneinfo import ZoneInfo + +import httpx + +from options_api.models import normalize_ticker +from stocksweeper.storage.db import connect, rows + +_NY = ZoneInfo("America/New_York") +_ACCESSION = re.compile(r"\d{10}-\d{2}-\d{6}\Z") +_FILENAME = re.compile(r"[A-Za-z0-9][A-Za-z0-9._-]{0,127}\Z") +_DATE = ( + r"(?P(?:January|February|March|April|May|June|July|August|September|" + r"October|November|December|Jan\.?|Feb\.?|Mar\.?|Apr\.?|Jun\.?|Jul\.?|" + r"Aug\.?|Sep\.?|Sept\.?|Oct\.?|Nov\.?|Dec\.?)\s+\d{1,2},?\s+20\d{2})" +) +_SCHEDULE = re.compile( + r"(?:will|plans to|expects to|to)\s+(?:report|release|announce)\b.{0,150}?" + r"(?:earnings|financial results|quarterly results)\b.{0,180}?\bon\s+" + _DATE, + re.IGNORECASE, +) +_ACTUAL_PREFIX = re.compile( + r"\bOn\s+" + _DATE + r"\b.{0,120}?\b(?:reported|announced|released)\b" + r".{0,150}?\b(?:earnings|financial results|quarterly results)\b", + re.IGNORECASE, +) +_ACTUAL_SUFFIX = re.compile( + r"\b(?:reported|announced|released)\b.{0,150}?" + r"\b(?:earnings|financial results|quarterly results)\b.{0,100}?\bon\s+" + _DATE, + re.IGNORECASE, +) +_TIME = re.compile( + r"\bat\s+(\d{1,2})(?::(\d{2}))?\s*(a\.?m\.?|p\.?m\.?)\s*(?:ET|EST|EDT|Eastern Time)\b", + re.IGNORECASE, +) +_PERIOD = re.compile( + r"\b(first|second|third|fourth|1st|2nd|3rd|4th)\s+quarter\s+(20\d{2})\b", re.IGNORECASE +) +_MAX_BYTES = 1_000_000 +_MAX_FILINGS = 5 +_rate_lock = threading.Lock() +_last_request = 0.0 + + +class _Text(HTMLParser): + def __init__(self) -> None: + super().__init__(convert_charrefs=True) + self.parts: list[str] = [] + self.skip = 0 + + def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: + if tag in ("script", "style"): + self.skip += 1 + + def handle_endtag(self, tag: str) -> None: + if tag in ("script", "style") and self.skip: + self.skip -= 1 + + def handle_data(self, data: str) -> None: + if not self.skip: + self.parts.append(data) + + +@dataclass(frozen=True) +class SecSchedule: + ticker: str + cik: int + accession: str + accepted_at: datetime + retrieved_at: datetime + event_date: date + event_at: datetime | None + series_key: str | None + document_url: str + document_hash: str + evidence: str + + @property + def knowledge_at(self) -> datetime: + return max(self.accepted_at, self.retrieved_at) + + +@dataclass(frozen=True) +class SecActual: + ticker: str + cik: int + accession: str + accepted_at: datetime + retrieved_at: datetime + event_date: date + document_url: str + document_hash: str + evidence: str + + +def _filing_url(cik: int, accession: str, document: str) -> str: + if ( + not 0 < cik <= 9_999_999_999 + or not _ACCESSION.fullmatch(accession) + or not _FILENAME.fullmatch(document) + ): + raise ValueError("invalid SEC filing identity") + return f"https://www.sec.gov/Archives/edgar/data/{cik}/{accession.replace('-', '')}/{document}" + + +def _fetch(client: httpx.Client, url: str, user_agent: str) -> bytes: + global _last_request + if not ( + url.startswith("https://www.sec.gov/Archives/edgar/data/") + or url.startswith("https://data.sec.gov/submissions/CIK") + ): + raise ValueError("SEC fetch host or path is not allowed") + if "@" not in user_agent or len(user_agent) > 200: + raise ValueError("SEC requests need a declared contact User-Agent") + with _rate_lock: + delay = 0.25 - (clock.monotonic() - _last_request) + if delay > 0: + clock.sleep(delay) + _last_request = clock.monotonic() + with client.stream( + "GET", + url, + headers={"User-Agent": user_agent, "Accept-Encoding": "identity"}, + follow_redirects=False, + ) as response: + if response.status_code != 200 or response.is_redirect: + raise ValueError(f"SEC document unavailable (HTTP {response.status_code})") + if int(response.headers.get("Content-Length", "0")) > _MAX_BYTES: + raise ValueError("SEC response exceeds byte limit") + content = bytearray() + for chunk in response.iter_bytes(): + content.extend(chunk) + if len(content) > _MAX_BYTES: + raise ValueError("SEC response exceeds byte limit") + return bytes(content) + + +def _document_text(content: bytes) -> str: + parser = _Text() + parser.feed(content.decode("utf-8", errors="replace")) + return " ".join(" ".join(parser.parts).split()) + + +def _event_date(text: str) -> date | None: + normalized = text.replace(".", "").replace(",", "") + for layout in ("%B %d %Y", "%b %d %Y"): + try: + return datetime.strptime(normalized, layout).date() + except ValueError: + pass + return None + + +def parse_forward_schedule( + ticker: str, + cik: int, + accession: str, + accepted_at: datetime, + retrieved_at: datetime, + document: str, + content: bytes, +) -> SecSchedule | None: + """Accept an explicit future date in a dated 8-K document, never infer one.""" + url = _filing_url(cik, accession, document) + if ( + normalize_ticker(ticker) != ticker + or accepted_at.tzinfo is None + or retrieved_at.tzinfo is None + ): + raise ValueError("SEC schedule identity or time is invalid") + if len(content) > _MAX_BYTES: + raise ValueError("SEC document exceeds byte limit") + text = _document_text(content) + matches = list(_SCHEDULE.finditer(text)) + if len(matches) != 1: + return None # Ambiguous or absent announcement. + match = matches[0] + event_date = _event_date(match.group("date")) + if event_date is None: + return None + if event_date <= accepted_at.astimezone(_NY).date() or retrieved_at < accepted_at: + return None + context = text[max(0, match.start() - 100) : min(len(text), match.end() + 100)] + # A nearby filing timestamp must not become the announced event time. + after_date = text[match.end() : min(len(text), match.end() + 100)].split(".", 1)[0] + local_time = _TIME.search(after_date) + event_at = None + if local_time: + hour = int(local_time.group(1)) % 12 + ( + 12 if local_time.group(3).lower().startswith("p") else 0 + ) + minute = int(local_time.group(2) or 0) + if hour < 24 and minute < 60: + event_at = datetime( + event_date.year, event_date.month, event_date.day, hour, minute, tzinfo=_NY + ).astimezone(UTC) + period = _PERIOD.search(context) + quarter = { + "first": 1, + "1st": 1, + "second": 2, + "2nd": 2, + "third": 3, + "3rd": 3, + "fourth": 4, + "4th": 4, + } + series_key = f"{period.group(2)}Q{quarter[period.group(1).lower()]}" if period else None + return SecSchedule( + ticker, + cik, + accession, + accepted_at.astimezone(UTC), + retrieved_at.astimezone(UTC), + event_date, + event_at, + series_key, + url, + hashlib.sha256(content).hexdigest(), + text[match.start() : match.end()][:300], + ) + + +def parse_actual_results( + ticker: str, + cik: int, + accession: str, + accepted_at: datetime, + retrieved_at: datetime, + document: str, + content: bytes, +) -> SecActual | None: + """Require an explicitly dated results announcement near filing acceptance.""" + url = _filing_url(cik, accession, document) + if ( + normalize_ticker(ticker) != ticker + or accepted_at.tzinfo is None + or retrieved_at.tzinfo is None + ): + raise ValueError("SEC actual-results identity or time is invalid") + if len(content) > _MAX_BYTES or retrieved_at < accepted_at: + raise ValueError("SEC actual-results content or retrieval time is invalid") + text = _document_text(content) + matches = list(_ACTUAL_PREFIX.finditer(text)) + list(_ACTUAL_SUFFIX.finditer(text)) + if len(matches) != 1: + return None + match = matches[0] + event_date = _event_date(match.group("date")) + accepted_day = accepted_at.astimezone(_NY).date() + if event_date is None or not 0 <= (accepted_day - event_date).days <= 3: + return None + return SecActual( + ticker, + cik, + accession, + accepted_at.astimezone(UTC), + retrieved_at.astimezone(UTC), + event_date, + url, + hashlib.sha256(content).hexdigest(), + text[match.start() : match.end()][:300], + ) + + +def record_actual(data_dir: Path, actual: SecActual) -> bool: + identity = ( + actual.ticker, + actual.cik, + actual.accession, + actual.document_hash, + actual.event_date.isoformat(), + ) + key = hashlib.sha256(json.dumps(identity).encode()).hexdigest() + with connect(data_dir / "results.duckdb") as connection: + connection.execute( + """CREATE TABLE IF NOT EXISTS sec_earnings_actuals ( + event_key VARCHAR PRIMARY KEY, ticker VARCHAR NOT NULL, cik BIGINT NOT NULL, + accession VARCHAR NOT NULL, accepted_at TIMESTAMPTZ NOT NULL, + retrieved_at TIMESTAMPTZ NOT NULL, event_date DATE NOT NULL, + document_url VARCHAR NOT NULL, document_hash VARCHAR NOT NULL, + evidence VARCHAR NOT NULL + )""" + ) + before = connection.execute( + "SELECT count(*) FROM sec_earnings_actuals WHERE event_key = ?", [key] + ).fetchone()[0] + connection.execute( + """INSERT OR IGNORE INTO sec_earnings_actuals + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", + [ + key, + actual.ticker, + actual.cik, + actual.accession, + actual.accepted_at, + actual.retrieved_at, + actual.event_date, + actual.document_url, + actual.document_hash, + actual.evidence, + ], + ) + return before == 0 + + +def record_schedule(data_dir: Path, schedule: SecSchedule) -> bool: + """Append one source revision; return whether it was newly recorded.""" + if schedule.knowledge_at.date() > schedule.event_date: + raise ValueError("earnings schedule was not known before event") + identity = ( + schedule.ticker, + schedule.cik, + schedule.accession, + schedule.document_hash, + schedule.event_date.isoformat(), + schedule.event_at.isoformat() if schedule.event_at else None, + ) + key = hashlib.sha256(json.dumps(identity).encode()).hexdigest() + with connect(data_dir / "results.duckdb") as connection: + connection.execute( + """CREATE TABLE IF NOT EXISTS sec_earnings_schedules ( + event_key VARCHAR PRIMARY KEY, ticker VARCHAR NOT NULL, cik BIGINT NOT NULL, + accession VARCHAR NOT NULL, accepted_at TIMESTAMPTZ NOT NULL, + retrieved_at TIMESTAMPTZ NOT NULL, event_date DATE NOT NULL, + event_at TIMESTAMPTZ, series_key VARCHAR, document_url VARCHAR NOT NULL, + document_hash VARCHAR NOT NULL, evidence VARCHAR NOT NULL + )""" + ) + before = connection.execute( + "SELECT count(*) FROM sec_earnings_schedules WHERE event_key = ?", [key] + ).fetchone()[0] + connection.execute( + """INSERT OR IGNORE INTO sec_earnings_schedules + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", + [ + key, + schedule.ticker, + schedule.cik, + schedule.accession, + schedule.accepted_at, + schedule.retrieved_at, + schedule.event_date, + schedule.event_at, + schedule.series_key, + schedule.document_url, + schedule.document_hash, + schedule.evidence, + ], + ) + return before == 0 + + +def known_forward_schedules( + data_dir: Path, ticker: str, as_of: datetime, through: date +) -> tuple[dict[str, object], ...]: + """Return only schedules retrieved by the origin, preserving revision rows.""" + known = _schedules_as_of(data_dir, ticker, as_of) + current_day = as_of.astimezone(_NY).date() + return tuple( + item + for item in known + if current_day < item["event_date"] <= through + ) + + +def _schedules_as_of( + data_dir: Path, ticker: str, as_of: datetime +) -> tuple[dict[str, object], ...]: + if normalize_ticker(ticker) != ticker or as_of.tzinfo is None: + raise ValueError("invalid SEC event lookup") + with connect(data_dir / "results.duckdb") as connection: + connection.execute( + """CREATE TABLE IF NOT EXISTS sec_earnings_schedules ( + event_key VARCHAR PRIMARY KEY, ticker VARCHAR NOT NULL, cik BIGINT NOT NULL, + accession VARCHAR NOT NULL, accepted_at TIMESTAMPTZ NOT NULL, + retrieved_at TIMESTAMPTZ NOT NULL, event_date DATE NOT NULL, + event_at TIMESTAMPTZ, series_key VARCHAR, document_url VARCHAR NOT NULL, + document_hash VARCHAR NOT NULL, evidence VARCHAR NOT NULL + )""" + ) + found = rows( + connection, + """SELECT * FROM sec_earnings_schedules WHERE ticker = ? + AND accepted_at <= ? AND retrieved_at <= ? + ORDER BY accepted_at, retrieved_at, event_key""", + [ticker, as_of, as_of], + ) + return tuple(found) + + +def effective_forward_schedules( + data_dir: Path, ticker: str, as_of: datetime, through: date +) -> tuple[dict[str, object], ...]: + """Latest observed revision per explicitly identified fiscal period.""" + # Choose the latest known revision before applying the date window. An old + # future date must not survive a later revision moving it into the past. + known = _schedules_as_of(data_dir, ticker, as_of) + latest: dict[str, dict[str, object]] = {} + for item in known: + series = item["series_key"] + if not isinstance(series, str): + continue + previous = latest.get(series) + order = (item["accepted_at"], item["retrieved_at"], item["event_key"]) + if previous is None or order > ( + previous["accepted_at"], + previous["retrieved_at"], + previous["event_key"], + ): + latest[series] = item + current_day = as_of.astimezone(_NY).date() + return tuple( + latest[key] + for key in sorted(latest) + if current_day < latest[key]["event_date"] <= through + ) + + +def verified_past_earnings_events( + data_dir: Path, ticker: str, as_of: datetime +) -> tuple[dict[str, object], ...]: + """Match an actual results release to a schedule known before that event. + + Only explicit dates in SEC documents qualify; no event is inferred from an + 8-K filing date. Schedule revisions are kept, then the last prior revision + for the exact realized date is selected without rewriting source records. + """ + if normalize_ticker(ticker) != ticker or as_of.tzinfo is None: + raise ValueError("invalid SEC event lookup") + with connect(data_dir / "results.duckdb") as connection: + actual_tables = {row[0] for row in connection.execute("SHOW TABLES").fetchall()} + if not {"sec_earnings_actuals", "sec_earnings_schedules"} <= actual_tables: + return () + actuals = rows( + connection, + """SELECT * FROM sec_earnings_actuals + WHERE ticker = ? AND accepted_at <= ? AND retrieved_at <= ? + ORDER BY event_date, accepted_at, retrieved_at, event_key""", + [ticker, as_of, as_of], + ) + schedules = rows( + connection, + """SELECT * FROM sec_earnings_schedules WHERE ticker = ? + ORDER BY event_date, accepted_at, retrieved_at, event_key""", + [ticker], + ) + result = [] + seen_dates: set[date] = set() + for actual in actuals: + day = actual["event_date"] + if day in seen_dates or day >= as_of.astimezone(_NY).date(): + continue + prior_by_series: dict[str, dict[str, object]] = {} + for schedule in schedules: + series = schedule["series_key"] + if not isinstance(series, str) or schedule["cik"] != actual["cik"]: + continue + if ( + schedule["accepted_at"].astimezone(_NY).date() >= day + or schedule["retrieved_at"].astimezone(_NY).date() >= day + ): + continue + previous = prior_by_series.get(series) + order = (schedule["accepted_at"], schedule["retrieved_at"], schedule["event_key"]) + if previous is None or order > ( + previous["accepted_at"], + previous["retrieved_at"], + previous["event_key"], + ): + prior_by_series[series] = schedule + matching = [item for item in prior_by_series.values() if item["event_date"] == day] + if len(matching) != 1: + continue + latest = matching[0] + result.append( + { + "ticker": ticker, + "cik": actual["cik"], + "event_date": day, + "actual_accession": actual["accession"], + "actual_accepted_at": actual["accepted_at"], + "actual_retrieved_at": actual["retrieved_at"], + "actual_document_hash": actual["document_hash"], + "schedule_accession": latest["accession"], + "schedule_accepted_at": latest["accepted_at"], + "schedule_retrieved_at": latest["retrieved_at"], + "schedule_document_hash": latest["document_hash"], + } + ) + seen_dates.add(day) + return tuple(result) + + +def _one_exhibit_991(index: bytes) -> str | None: + """SEC archive index has filenames, not reliable exhibit type metadata.""" + try: + listing = json.loads(index)["directory"]["item"] + matches = [] + for item in listing: + name = item["name"] + if not isinstance(name, str) or not _FILENAME.fullmatch(name): + continue + if not name.lower().endswith((".htm", ".html", ".txt")): + continue + compact = re.sub(r"[^a-z0-9]", "", name.lower()) + if re.search(r"(?:exhibit|ex)991(?!\d)", compact): + matches.append(name) + return matches[0] if len(matches) == 1 else None + except (KeyError, TypeError, ValueError, json.JSONDecodeError): + return None + + +def capture_recent_sec_schedules( + data_dir: Path, + ticker: str, + cik: int, + user_agent: str, + *, + now: datetime | None = None, + client: httpx.Client | None = None, + observed_time: Callable[[], datetime] | None = None, +) -> int: + """Inspect at most five recent 8-Ks and one exhibit each, offline-safe. + + This intentionally does not infer a calendar from SEC filing history. Only + explicit forward dates and actual result dates become separate records. + """ + if normalize_ticker(ticker) != ticker or not 0 < cik <= 9_999_999_999: + raise ValueError("invalid SEC issuer") + if now is not None and now.tzinfo is None: + raise ValueError("SEC capture time must be timezone-aware") + + def capture_time() -> datetime: + value = observed_time() if observed_time is not None else now or datetime.now(UTC) + if value.tzinfo is None: + raise ValueError("SEC capture time must be timezone-aware") + return value.astimezone(UTC) + + owned_client = client is None + client = client or httpx.Client( + timeout=httpx.Timeout(10, connect=5), follow_redirects=False, trust_env=False + ) + try: + url = f"https://data.sec.gov/submissions/CIK{cik:010d}.json" + response = json.loads(_fetch(client, url, user_agent)) + observed_at = capture_time() + if response.get("cik") != cik or ticker not in response.get("tickers", []): + raise ValueError("SEC CIK does not match ticker") + recent = response["filings"]["recent"] + selected = [ + (form, accession, document, accepted) + for form, accession, document, accepted in zip( + recent["form"], + recent["accessionNumber"], + recent["primaryDocument"], + recent["acceptanceDateTime"], + strict=True, + ) + if form in ("8-K", "8-K/A") + ][:_MAX_FILINGS] + inserted = 0 + for _, accession, document, accepted in selected: + accepted_at = datetime.fromisoformat(accepted) + if accepted_at.tzinfo is None: + accepted_at = accepted_at.replace(tzinfo=_NY) + if accepted_at > observed_at: + continue + content = _fetch(client, _filing_url(cik, accession, document), user_agent) + primary_retrieved_at = capture_time() + schedule = parse_forward_schedule( + ticker, cik, accession, accepted_at, primary_retrieved_at, document, content + ) + actual = parse_actual_results( + ticker, cik, accession, accepted_at, primary_retrieved_at, document, content + ) + if schedule is None or actual is None: + index = _fetch(client, _filing_url(cik, accession, "index.json"), user_agent) + exhibit = _one_exhibit_991(index) + if exhibit is not None: + exhibit_content = _fetch( + client, _filing_url(cik, accession, exhibit), user_agent + ) + exhibit_retrieved_at = capture_time() + if schedule is None: + schedule = parse_forward_schedule( + ticker, + cik, + accession, + accepted_at, + exhibit_retrieved_at, + exhibit, + exhibit_content, + ) + if actual is None: + actual = parse_actual_results( + ticker, + cik, + accession, + accepted_at, + exhibit_retrieved_at, + exhibit, + exhibit_content, + ) + # A first-time scan may discover an old announcement only after + # its event. Keep the record_schedule invariant and continue. + if schedule is not None and schedule.knowledge_at.date() <= schedule.event_date: + inserted += record_schedule(data_dir, schedule) + if actual is not None: + record_actual(data_dir, actual) + return inserted + finally: + if owned_client: + client.close() diff --git a/backend/tests/test_evidence_reports.py b/backend/tests/test_evidence_reports.py index 8db95c1..3f9cd44 100644 --- a/backend/tests/test_evidence_reports.py +++ b/backend/tests/test_evidence_reports.py @@ -7,6 +7,7 @@ from stocksweeper.forecast.evidence_reports import ( CANDIDATE_VERSIONS, + _summary, build_model_evidence, ledger_band_rows, load_replay_report, @@ -14,7 +15,7 @@ ) from stocksweeper.forecast.ledger import ForecastLedger from stocksweeper.forecast.calendar import SessionCalendar -from stocksweeper.forecast.physical_evaluation import evaluate_band +from stocksweeper.forecast.physical_evaluation import ContestRow, evaluate_band from stocksweeper.forecast.predictive import BASELINE_VERSION @@ -22,7 +23,10 @@ def test_unlabelled_ledger_reports_zero_and_no_significance(tmp_path): evidence = build_model_evidence( tmp_path, ForecastLedger(tmp_path), datetime(2026, 9, 27, tzinfo=UTC) ) - assert len(evidence) == 15 + assert len(evidence) == 3 * (len(CANDIDATE_VERSIONS) + 2) + assert evidence[("intraday_shadow", "1")]["prospective"]["capture_windows_all_horizons"] == { + "expected": 0, "captured": 0, "missed": 0, "pending": 0, + } for (method, band), result in evidence.items(): assert method and band prospective = result["prospective"] @@ -35,6 +39,65 @@ def test_unlabelled_ledger_reports_zero_and_no_significance(tmp_path): assert result["retrospective"] is None +def test_adjusted_interval_needs_twenty_independent_scored_date_blocks() -> None: + calendar = SessionCalendar() + origins = calendar.sessions(date(2026, 1, 5), date(2026, 7, 1))[::5][:20] + assert len(origins) == 20 + rows = [] + for origin in origins: + expiry = calendar.offset(origin, 1) + for side, observed, baseline, candidate in ( + ("call", True, 0.4, 0.6), + ("put", False, 0.6, 0.4), + ): + for method, probability, score in ( + ("lognormal_ewma", baseline, 3.4), + ("ohlc_har", candidate, 3.1), + ): + rows.append(ContestRow( + ticker="TEST", origin=origin, expiry_session=expiry, + horizon=1, strike="100", side=side, method=method, + probability=probability, observed_itm=observed, + provenance="as_issued", input_vintage="frozen-input", + contract_id=f"TEST:{origin}:{side}", crps=score, + )) + + def reported(input_rows): + band = evaluate_band(input_rows, "ohlc_har", "1", bootstrap_samples=500) + return _summary( + band, provenance="as_issued", generated_at="2026-09-28T00:00:00Z", + report_hash="test", model_version="ohlc-har-v1", + ) + + enough = reported(rows) + assert enough["ticker_origin_horizon_units"] == 20 # Call and put share one close. + assert enough["independent_date_blocks"] == 20 + assert enough["brier"]["comparison_count"] == 10 + ordinary = enough["brier"]["bootstrap_95"] + adjusted = enough["brier"]["bootstrap_familywise_95"] + assert ordinary is not None and adjusted is not None + assert adjusted[0] <= ordinary[0] <= ordinary[1] <= adjusted[1] + assert enough["significance"] == "exploratory_interval" + assert enough["crps"]["scored_units"] == 20 + + too_few = reported([ + row for row in rows if row.origin != origins[-1] + ]) + assert too_few["independent_date_blocks"] == 19 + assert too_few["brier"]["bootstrap_95"] is None + assert too_few["brier"]["bootstrap_familywise_95"] is None + assert too_few["significance"] == "not_estimable" + + no_labels = reported([]) + assert no_labels["ticker_origin_horizon_units"] == 0 + assert no_labels["brier"]["candidate"] is None + assert no_labels["brier"]["bootstrap_familywise_95"] is None + assert no_labels["crps"] == { + "scored_units": 0, "baseline": None, "candidate": None, + } + assert no_labels["significance"] == "not_estimable" + + def test_replay_report_is_explicitly_current_vintage_and_detects_tampering(tmp_path): band = evaluate_band([], "student_t_ewma", "1", bootstrap_samples=100) report = {"provenance": "immutable_replay", "bands": {"1": band}} diff --git a/backend/tests/test_evidence_selection_regressions.py b/backend/tests/test_evidence_selection_regressions.py index 26cd063..d20d801 100644 --- a/backend/tests/test_evidence_selection_regressions.py +++ b/backend/tests/test_evidence_selection_regressions.py @@ -14,6 +14,7 @@ from options_api.intraday_capture import _COMPARATOR_VERSION from options_api.intraday_shadow import _VERSION as INTRADAY_VERSION from options_api.predictive_watch import PredictiveWatchOdds +from options_api.physical_shadow_capture import MODEL_VERSIONS from options_api.watchlist import OutcomeView, WatchItem, get_watchlist from stocksweeper.forecast.calendar import SessionCalendar from stocksweeper.forecast.calibration import build_calibration @@ -166,7 +167,7 @@ def panel_coverage(self, **_kwargs): @pytest.mark.asyncio -async def test_expired_watch_keeps_five_physical_two_market_and_selected_risk() -> None: +async def test_expired_watch_keeps_all_models_and_selected_risk() -> None: expiry = date(2026, 9, 25) report = {"prospective": {"ticker_origin_horizon_units": 0}} item = WatchItem( @@ -191,12 +192,9 @@ async def test_expired_watch_keeps_five_physical_two_market_and_selected_risk() body = response.model_dump(mode="json") saved = body["items"][0] assert body["model_evidence"] == {"student_t_ewma:1": report} - assert [model["method"] for model in saved["physical_models"]] == [ - "lognormal_ewma", "empirical_scaled", "student_t_ewma", "gjr_garch_t", - "intraday_shadow", - ] + assert [model["method"] for model in saved["physical_models"]] == list(MODEL_VERSIONS) assert [model["method"] for model in saved["market_models"]] == [ - "regimelib", "constrained_call_curve", + "regimelib", "constrained_call_curve", "ssvi", ] assert all(model["status"] == "unavailable" for model in saved["physical_models"]) assert all(model["status"] == "unavailable" for model in saved["market_models"]) diff --git a/backend/tests/test_forecast_provenance.py b/backend/tests/test_forecast_provenance.py new file mode 100644 index 0000000..cce6937 --- /dev/null +++ b/backend/tests/test_forecast_provenance.py @@ -0,0 +1,767 @@ +"""Offline provenance checks: frozen prices, missed windows and SEC knowledge time.""" + +from __future__ import annotations + +import json +import hashlib +import importlib.util +import sys +from dataclasses import dataclass +from datetime import UTC, date, datetime, timedelta +from decimal import Decimal +from pathlib import Path + +import httpx +import polars as pl +import pytest + +from options_api.contract_identity import make_watch_key +from stocksweeper.forecast.audit import freeze_audit_cohort +from stocksweeper.forecast.calendar import SessionCalendar +from stocksweeper.forecast.capture_windows import capture_window_counts, reconcile_capture_windows +from stocksweeper.forecast.market import CacheIntegrityError, ForecastPriceStore, price_hash +from stocksweeper.forecast.provenance import ( + SourceRightsUnverified, + freeze_price_vintage, + freeze_training_cohort, + load_training_cohort, + read_price_vintage, +) +from stocksweeper.forecast.sec_events import ( + capture_recent_sec_schedules, + effective_forward_schedules, + known_forward_schedules, + parse_actual_results, + parse_forward_schedule, + record_actual, + record_schedule, + verified_past_earnings_events, +) +from stocksweeper.storage.db import connect, rows + +SESSION = date(2026, 9, 25) + + +def _prices(last: date = SESSION, count: int = 501) -> pl.DataFrame: + days = SessionCalendar().sessions(date(2023, 1, 1), last)[-count:] + close = [100 + i / 100 for i in range(count)] + return pl.DataFrame( + { + "ts": days, + "open": close, + "high": [x + 1 for x in close], + "low": [x - 1 for x in close], + "close": close, + "volume": [1000.0] * count, + "dividends": [0.0] * count, + "stock_splits": [0.0] * count, + } + ) + + +class Provider: + def __init__(self, frame: pl.DataFrame) -> None: + self.frame = frame + + def fetch(self, ticker, start, end): + frame = self.frame.filter(pl.col("ts") <= end) + return frame if start is None else frame.filter(pl.col("ts") >= start) + + +def test_local_vintages_preserve_revisions_and_reject_tampering(tmp_path): + provider = Provider(_prices()) + store = ForecastPriceStore(tmp_path, provider) + store.update("AAPL", SESSION) + first, first_hash = freeze_price_vintage(tmp_path, "AAPL", SESSION) + assert first.height == 501 + provider.frame = provider.frame.with_columns( + pl.when(pl.col("ts") == SESSION).then(106.0).otherwise(pl.col("close")).alias("close"), + pl.when(pl.col("ts") == SESSION).then(107.0).otherwise(pl.col("high")).alias("high"), + ) + store.update("AAPL", SESSION, full_refresh=True) + _, second_hash = freeze_price_vintage(tmp_path, "AAPL", SESSION) + assert first_hash != second_hash + assert read_price_vintage(tmp_path, "AAPL", SESSION, first_hash)[0].equals(first) + path = ( + tmp_path / "forecast" / "vintages" / "AAPL" / SESSION.isoformat() / f"{first_hash}.parquet" + ) + path.chmod(0o600) + path.write_bytes(b"changed") + with pytest.raises(CacheIntegrityError): + read_price_vintage(tmp_path, "AAPL", SESSION, first_hash) + + +def test_vintage_freeze_rejects_cache_manifest_race(tmp_path, monkeypatch): + ForecastPriceStore(tmp_path, Provider(_prices())).update("AAPL", SESSION) + original_read = ForecastPriceStore.read + + def raced_read(store, ticker): + frame = original_read(store, ticker) + manifest = store.path(ticker).with_suffix(".json") + metadata = json.loads(manifest.read_text()) + metadata["hash"] = "0" * 64 + manifest.write_text(json.dumps(metadata)) + return frame + + monkeypatch.setattr(ForecastPriceStore, "read", raced_read) + with pytest.raises(CacheIntegrityError, match="changed while freezing"): + freeze_price_vintage(tmp_path, "AAPL", SESSION) + + +def test_frozen_vintage_requires_its_manifest_hash(tmp_path): + ForecastPriceStore(tmp_path, Provider(_prices())).update("AAPL", SESSION) + _, digest = freeze_price_vintage(tmp_path, "AAPL", SESSION) + manifest = tmp_path / "forecast" / "vintages" / "AAPL" / SESSION.isoformat() / f"{digest}.json" + metadata = json.loads(manifest.read_text()) + metadata.pop("manifest_hash") + metadata["source_retrieved_at"] = datetime(2026, 9, 27, tzinfo=UTC).isoformat() + manifest.chmod(0o600) + manifest.write_text(json.dumps(metadata)) + with pytest.raises(CacheIntegrityError, match="invalid immutable price vintage"): + read_price_vintage(tmp_path, "AAPL", SESSION, digest) + + +def _signed(metadata: dict[str, object]) -> str: + return hashlib.sha256( + json.dumps(metadata, sort_keys=True, separators=(",", ":")).encode() + ).hexdigest() + + +def test_rights_approved_synthetic_cohort_loads_for_trainer_without_mocking(tmp_path): + from stocksweeper.forecast.pooled_ngboost import ( + _training_rows, load_pooled_model, train_pooled_model, + ) + + ForecastPriceStore(tmp_path, Provider(_prices())).update("AAPL", SESSION) + audit = freeze_audit_cohort(tmp_path, ["AAPL"], SESSION, size=1) + source = "Project synthetic fixture" + license_reference = "project-owned-synthetic-test-data-v1" + frame = _prices() + data_hash = price_hash(frame) + observed = datetime(2026, 9, 26, 12, tzinfo=UTC).isoformat() + members = [] + for index in range(20): + ticker = f"T{chr(65 + index)}A" + root = tmp_path / "forecast" / "vintages" / ticker / SESSION.isoformat() + root.mkdir(parents=True) + frame.write_parquet(root / f"{data_hash}.parquet") + vintage: dict[str, object] = { + "ticker": ticker, + "through_session": SESSION.isoformat(), + "data_hash": data_hash, + "source_hash": data_hash, + "source": source, + "source_name": source, + "license_reference": license_reference, + "source_retrieved_at": observed, + "frozen_at": observed, + "price_basis": "split-normalized, dividend-unadjusted", + "provenance": "immutable_current_vintage_training", + "rights_status": "approved_for_training", + } + vintage["manifest_hash"] = _signed(vintage) + (root / f"{data_hash}.json").write_text(json.dumps(vintage)) + members.append({"ticker": ticker, "data_hash": data_hash}) + cohort: dict[str, object] = { + "provenance": "immutable_current_vintage_training", + "rights_status": "approved_for_training", + "source": source, + "source_name": source, + "license_reference": license_reference, + "audit_cohort_hash": _signed( + {"members": sorted((member.ticker, member.data_hash) for member in audit.members)} + ), + "completed_session": SESSION.isoformat(), + "size": len(members), + "members": members, + "frozen_at": observed, + } + cohort["manifest_hash"] = _signed(cohort) + manifest = tmp_path / "forecast" / "training" / "cohort.json" + manifest.parent.mkdir(parents=True) + manifest.write_text(json.dumps(cohort)) + loaded = load_training_cohort(tmp_path) + assert len(loaded) == 20 + inputs, targets, through, _ = _training_rows(loaded) + assert len(inputs) >= 5_000 and len(targets) == len(inputs) and through == SESSION + artifact = train_pooled_model(tmp_path) + assert artifact.is_file() + assert load_pooled_model(tmp_path)["cohort_manifest_hash"] == cohort["manifest_hash"] + cohort["license_reference"] = "another-source-right" + cohort["manifest_hash"] = _signed( + {key: value for key, value in cohort.items() if key != "manifest_hash"} + ) + manifest.write_text(json.dumps(cohort)) + with pytest.raises(CacheIntegrityError, match="invalid immutable training cohort"): + load_training_cohort(tmp_path) + with pytest.raises(CacheIntegrityError, match="invalid immutable training cohort"): + load_pooled_model(tmp_path) + + +def test_training_cohort_is_disjoint_and_rights_unqualified(tmp_path): + for ticker in ("AAPL", "MSFT", "NVDA"): + ForecastPriceStore(tmp_path, Provider(_prices())).update(ticker, SESSION) + audit = freeze_audit_cohort(tmp_path, ["AAPL", "MSFT", "NVDA"], SESSION, size=1) + members = freeze_training_cohort(tmp_path, ["AAPL", "MSFT", "NVDA"], SESSION, size=2) + assert {ticker for ticker, _, _ in members}.isdisjoint({item.ticker for item in audit.members}) + assert len(load_training_cohort(tmp_path, require_training_rights=False)) == 2 + with pytest.raises(SourceRightsUnverified, match="training_rights_unverified"): + load_training_cohort(tmp_path) + with pytest.raises(ValueError, match="another size or session"): + freeze_training_cohort(tmp_path, ["AAPL", "MSFT", "NVDA"], SESSION, size=1) + audit_manifest = tmp_path / "forecast" / "audit" / "cohort.json" + audit_missing = audit_manifest.with_suffix(".missing") + audit_manifest.rename(audit_missing) + try: + with pytest.raises(CacheIntegrityError, match="invalid immutable training cohort"): + load_training_cohort(tmp_path, require_training_rights=False) + finally: + audit_missing.rename(audit_manifest) + manifest = tmp_path / "forecast" / "training" / "cohort.json" + contents = json.loads(manifest.read_text()) + assert contents["rights_status"] == "training_rights_unverified" + contents["size"] = 1 + manifest.chmod(0o600) + manifest.write_text(json.dumps(contents)) + with pytest.raises(CacheIntegrityError): + load_training_cohort(tmp_path, require_training_rights=False) + + +def test_missing_bar_and_recent_split_reject_frozen_training_history(tmp_path): + frame = _prices() + missing = frame.filter(pl.col("ts") != frame["ts"][200]) + ForecastPriceStore(tmp_path / "gap", Provider(missing)).update("AAPL", SESSION) + with pytest.raises(ValueError, match="contiguous split-safe"): + freeze_price_vintage(tmp_path / "gap", "AAPL", SESSION) + split = frame.with_columns( + pl.when(pl.col("ts") == frame["ts"][300]) + .then(2.0) + .otherwise(pl.col("stock_splits")) + .alias("stock_splits") + ) + ForecastPriceStore(tmp_path / "split", Provider(split)).update("AAPL", SESSION) + with pytest.raises(ValueError, match="contiguous split-safe"): + freeze_price_vintage(tmp_path / "split", "AAPL", SESSION) + + +@dataclass(frozen=True) +class Watch: + watch_key: str + created_at: datetime + expiration: date + + +def test_capture_denominator_recovers_missed_window_then_captured_revision(tmp_path): + key = make_watch_key("AAPL", "AAPL", "call", "2026-10-30", Decimal("100")) + watch = Watch(key, datetime(2026, 9, 24, 20, tzinfo=UTC), date(2026, 10, 30)) + now = datetime(2026, 9, 25, 15, 6, tzinfo=UTC) + first = reconcile_capture_windows(tmp_path, [watch], now) + assert first == {"expected": 1, "captured": 0, "missed": 1, "pending": 0} + assert reconcile_capture_windows(tmp_path, [watch], now) == first + with connect(tmp_path / "results.duckdb") as connection: + connection.execute( + """INSERT INTO forecast_issuances + (idempotency_key, contract_key, ticker, root, side, expiration, + expiry_session, strike_exact, terms_note, issued_at, status, + provenance, snapshot_window) + VALUES (?, ?, 'AAPL', 'AAPL', 'call', '2026-10-30', '2026-10-30', + '100.000', 'standard', ?, 'unavailable', 'as_issued', '10:00')""", + ["a" * 64, key, datetime(2026, 9, 25, 14, 1, tzinfo=UTC)], + ) + second = reconcile_capture_windows(tmp_path, [watch], now) + assert second["captured"] == 1 and second["missed"] == 0 + assert capture_window_counts( + tmp_path, datetime(2026, 9, 25, 14, tzinfo=UTC), now + ) == {"expected": 1, "captured": 1, "missed": 0, "pending": 0} + with connect(tmp_path / "results.duckdb") as connection: + events = rows(connection, "SELECT event FROM forecast_capture_window_events") + assert sorted(row["event"] for row in events) == ["captured", "expected", "missed"] + + +def test_removed_watch_still_resolves_persisted_expected_window(tmp_path): + key = make_watch_key("AAPL", "AAPL", "call", "2026-10-30", Decimal("100")) + watch = Watch(key, datetime(2026, 9, 24, 20, tzinfo=UTC), date(2026, 10, 30)) + pending = reconcile_capture_windows( + tmp_path, [watch], datetime(2026, 9, 25, 14, 2, tzinfo=UTC) + ) + assert pending == {"expected": 1, "captured": 0, "missed": 0, "pending": 1} + assert capture_window_counts( + tmp_path, + datetime(2026, 9, 25, 14, tzinfo=UTC), + datetime(2026, 9, 25, 14, 2, tzinfo=UTC), + ) == pending + resolved = reconcile_capture_windows( + tmp_path, [], datetime(2026, 9, 25, 14, 6, tzinfo=UTC) + ) + assert resolved == {"expected": 1, "captured": 0, "missed": 1, "pending": 0} + with connect(tmp_path / "results.duckdb") as connection: + events = rows(connection, "SELECT event FROM forecast_capture_window_events") + assert sorted(row["event"] for row in events) == ["expected", "missed"] + assert capture_window_counts( + tmp_path, + datetime(2026, 9, 25, 14, tzinfo=UTC), + datetime(2026, 9, 25, 14, 6, tzinfo=UTC), + ) == resolved + + +def test_capture_counts_are_read_only_and_reject_unbounded_interval(tmp_path): + start = datetime(2026, 9, 25, 14, tzinfo=UTC) + assert capture_window_counts(tmp_path, start, start) == { + "expected": 0, "captured": 0, "missed": 0, "pending": 0, + } + assert not (tmp_path / "results.duckdb").exists() + with connect(tmp_path / "results.duckdb"): + # The live app can hold a write-capable connection during evidence reads. + assert capture_window_counts(tmp_path, start, start)["expected"] == 0 + with pytest.raises(ValueError, match="at most 35 days"): + capture_window_counts(tmp_path, start - timedelta(days=36), start) + + +def test_early_close_does_not_expect_closed_window(tmp_path): + watch = Watch( + make_watch_key("AAPL", "AAPL", "put", "2026-12-04", Decimal("100")), + datetime(2026, 11, 26, 20, tzinfo=UTC), + date(2026, 12, 4), + ) + # 2026-11-27 closes at 13:00 ET: 13:00 and 15:30 are not capture windows. + report = reconcile_capture_windows(tmp_path, [watch], datetime(2026, 11, 27, 21, tzinfo=UTC)) + assert report["expected"] == 1 and report["missed"] == 1 + + +def test_capture_accounting_indexes_issuances_by_window(tmp_path, monkeypatch): + from stocksweeper.forecast import capture_windows + + watches = [ + Watch( + make_watch_key("AAPL", "AAPL", "call", "2026-10-30", Decimal(index + 1)), + datetime(2026, 9, 24, 20, tzinfo=UTC), + date(2026, 10, 30), + ) + for index in range(128) + ] + reads = [0] + + class Tracked(dict): + def __getitem__(self, field): + reads[0] += 1 + return super().__getitem__(field) + + issued = [ + Tracked( + contract_key=watch.watch_key, + snapshot_window="10:00", + issued_at=datetime(2026, 9, 25, 14, 1, tzinfo=UTC), + ) + for watch in watches + ] + monkeypatch.setattr( + capture_windows, + "rows", + lambda _connection, sql, _params: [] if "forecast_capture_window_events" in sql else issued, + ) + counts = reconcile_capture_windows(tmp_path, watches, datetime(2026, 9, 25, 15, 6, tzinfo=UTC)) + assert counts["captured"] == 128 + assert reads[0] <= 128 * 4 # One pass through issuances, independent of watch count. + + +def _announcement(day: str) -> bytes: + return ( + f"

Acme will report third quarter 2026 financial results on {day} " + "at 4:30 PM Eastern Time.

" + ).encode() + + +def _results(day: str) -> bytes: + return ( + f"

On {day}, Acme announced its third quarter 2026 financial results.

" + ).encode() + + +def test_sec_schedule_revisions_obey_retrieval_knowledge_time(tmp_path): + accepted = datetime(2026, 9, 28, 16, tzinfo=UTC) + first = parse_forward_schedule( + "ACME", + 12345, + "0000012345-26-000001", + accepted, + accepted + timedelta(minutes=5), + "ex99-1.htm", + _announcement("November 5, 2026"), + ) + assert first is not None and first.series_key == "2026Q3" + assert first.event_at == datetime(2026, 11, 5, 21, 30, tzinfo=UTC) + assert record_schedule(tmp_path, first) + assert not record_schedule(tmp_path, first) + second = parse_forward_schedule( + "ACME", + 12345, + "0000012345-26-000002", + accepted + timedelta(days=1), + accepted + timedelta(days=1, minutes=5), + "ex99-1.htm", + _announcement("November 6, 2026"), + ) + assert second is not None and record_schedule(tmp_path, second) + before = known_forward_schedules( + tmp_path, "ACME", accepted + timedelta(minutes=4), date(2026, 11, 30) + ) + middle = known_forward_schedules( + tmp_path, "ACME", accepted + timedelta(hours=1), date(2026, 11, 30) + ) + after = known_forward_schedules( + tmp_path, "ACME", accepted + timedelta(days=2), date(2026, 11, 30) + ) + assert before == () + assert [row["event_date"] for row in middle] == [date(2026, 11, 5)] + assert [row["event_date"] for row in after] == [date(2026, 11, 5), date(2026, 11, 6)] + assert ( + parse_forward_schedule( + "ACME", + 12345, + "0000012345-26-000003", + accepted, + accepted + timedelta(minutes=5), + "ex99-1.htm", + b"

Acme reported earnings today.

", + ) + is None + ) + + +def test_realized_results_require_explicit_release_and_prior_schedule(tmp_path): + accepted = datetime(2026, 9, 28, 16, tzinfo=UTC) + schedule = parse_forward_schedule( + "ACME", + 12345, + "0000012345-26-000001", + accepted, + accepted + timedelta(minutes=5), + "ex99-1.htm", + _announcement("November 5, 2026"), + ) + assert schedule is not None and record_schedule(tmp_path, schedule) + actual_accepted = datetime(2026, 11, 6, 12, tzinfo=UTC) + actual = parse_actual_results( + "ACME", + 12345, + "0000012345-26-000002", + actual_accepted, + actual_accepted + timedelta(minutes=5), + "ex99-1.htm", + _results("November 5, 2026"), + ) + assert actual is not None and record_actual(tmp_path, actual) + assert not record_actual(tmp_path, actual) + before = verified_past_earnings_events(tmp_path, "ACME", actual_accepted + timedelta(minutes=4)) + after = verified_past_earnings_events(tmp_path, "ACME", actual_accepted + timedelta(hours=1)) + assert before == () + assert len(after) == 1 + assert after[0]["event_date"] == date(2026, 11, 5) + assert after[0]["schedule_accession"] == schedule.accession + assert ( + parse_actual_results( + "ACME", + 12345, + "0000012345-26-000003", + actual_accepted, + actual_accepted + timedelta(minutes=5), + "ex99-1.htm", + b"

Acme reported results.

", + ) + is None + ) + + +def test_same_ticker_different_cik_cannot_validate_actual_release(tmp_path): + accepted = datetime(2026, 9, 28, 16, tzinfo=UTC) + schedule = parse_forward_schedule( + "ACME", 12345, "0000012345-26-000001", accepted, + accepted + timedelta(minutes=5), "ex99-1.htm", _announcement("November 5, 2026"), + ) + assert schedule is not None and record_schedule(tmp_path, schedule) + actual_accepted = datetime(2026, 11, 6, 12, tzinfo=UTC) + actual = parse_actual_results( + "ACME", 54321, "0000054321-26-000001", actual_accepted, + actual_accepted + timedelta(minutes=5), "ex99-1.htm", _results("November 5, 2026"), + ) + assert actual is not None and record_actual(tmp_path, actual) + assert ( + verified_past_earnings_events(tmp_path, "ACME", actual_accepted + timedelta(hours=1)) + == () + ) + + +def test_later_schedule_revision_does_not_validate_old_event_date(tmp_path): + accepted = datetime(2026, 9, 28, 16, tzinfo=UTC) + for index, day in enumerate(("November 5, 2026", "November 6, 2026"), start=1): + timestamp = accepted + timedelta(days=index - 1) + schedule = parse_forward_schedule( + "ACME", + 12345, + f"0000012345-26-{index:06d}", + timestamp, + timestamp + timedelta(minutes=5), + "ex99-1.htm", + _announcement(day), + ) + assert schedule is not None and record_schedule(tmp_path, schedule) + current = effective_forward_schedules( + tmp_path, "ACME", accepted + timedelta(days=2), date(2026, 11, 30) + ) + assert [row["event_date"] for row in current] == [date(2026, 11, 6)] + actual_time = datetime(2026, 11, 6, 12, tzinfo=UTC) + old_date_actual = parse_actual_results( + "ACME", + 12345, + "0000012345-26-000003", + actual_time, + actual_time + timedelta(minutes=5), + "ex99-1.htm", + _results("November 5, 2026"), + ) + assert old_date_actual is not None and record_actual(tmp_path, old_date_actual) + assert verified_past_earnings_events(tmp_path, "ACME", actual_time + timedelta(hours=1)) == () + + +def test_expired_revision_does_not_leave_an_older_future_schedule_active(tmp_path): + accepted = datetime(2026, 10, 1, 16, tzinfo=UTC) + for index, day in enumerate(("November 10, 2026", "November 3, 2026"), start=1): + observed = accepted + timedelta(days=index - 1) + schedule = parse_forward_schedule( + "ACME", + 12345, + f"0000012345-26-{index:06d}", + observed, + observed + timedelta(minutes=5), + "ex99-1.htm", + _announcement(day), + ) + assert schedule is not None and record_schedule(tmp_path, schedule) + as_of = datetime(2026, 11, 5, 16, tzinfo=UTC) + assert [row["event_date"] for row in known_forward_schedules( + tmp_path, "ACME", as_of, date(2026, 11, 30) + )] == [date(2026, 11, 10)] + assert effective_forward_schedules(tmp_path, "ACME", as_of, date(2026, 11, 30)) == () + + +def test_schedule_does_not_take_an_unrelated_filing_time_as_event_time(): + accepted = datetime(2026, 9, 28, 16, tzinfo=UTC) + content = ( + b"

Filed at 9:00 AM ET. Acme will report third quarter 2026 " + b"financial results on November 5, 2026.

" + ) + schedule = parse_forward_schedule( + "ACME", 12345, "0000012345-26-000001", accepted, + accepted + timedelta(minutes=5), "ex99-1.htm", content, + ) + assert schedule is not None and schedule.event_at is None + + +def test_sec_fetch_is_bounded_and_checks_cik_and_filing_path(tmp_path): + accepted = "2026-09-28T12:00:00" + accession = "0000012345-26-000001" + submissions = { + "cik": 12345, + "tickers": ["ACME"], + "filings": { + "recent": { + "form": ["8-K"], + "accessionNumber": [accession], + "primaryDocument": ["announcement.htm"], + "acceptanceDateTime": [accepted], + } + }, + } + urls = [] + + def handler(request): + urls.append(str(request.url)) + if request.url.host == "data.sec.gov": + return httpx.Response(200, json=submissions) + if request.url.path.endswith("index.json"): + return httpx.Response(200, json={"directory": {"item": []}}) + return httpx.Response(200, content=_announcement("November 5, 2026")) + + client = httpx.Client(transport=httpx.MockTransport(handler), follow_redirects=False) + assert ( + capture_recent_sec_schedules( + tmp_path, + "ACME", + 12345, + "HyperOptions test@example.com", + now=datetime(2026, 9, 28, 17, tzinfo=UTC), + client=client, + ) + == 1 + ) + assert urls == [ + "https://data.sec.gov/submissions/CIK0000012345.json", + "https://www.sec.gov/Archives/edgar/data/12345/000001234526000001/announcement.htm", + "https://www.sec.gov/Archives/edgar/data/12345/000001234526000001/index.json", + ] + assert ( + capture_recent_sec_schedules( + tmp_path, + "ACME", + 12345, + "HyperOptions test@example.com", + now=datetime(2026, 9, 28, 17, tzinfo=UTC), + client=client, + ) + == 0 + ) + submissions["cik"] = 54321 + with pytest.raises(ValueError, match="CIK does not match"): + capture_recent_sec_schedules( + tmp_path, + "ACME", + 12345, + "HyperOptions test@example.com", + client=client, + ) + + +def test_late_first_sec_scan_skips_expired_schedule_and_continues(tmp_path): + old_accession = "0000012345-26-000001" + new_accession = "0000012345-26-000002" + submissions = { + "cik": 12345, + "tickers": ["ACME"], + "filings": {"recent": { + "form": ["8-K", "8-K"], + "accessionNumber": [old_accession, new_accession], + "primaryDocument": ["old.htm", "new.htm"], + "acceptanceDateTime": ["2026-09-28T12:00:00", "2026-10-31T12:00:00"], + }}, + } + + def handler(request): + if request.url.host == "data.sec.gov": + return httpx.Response(200, json=submissions) + if request.url.path.endswith("index.json"): + return httpx.Response(200, json={"directory": {"item": []}}) + if request.url.path.endswith("old.htm"): + return httpx.Response(200, content=_announcement("October 1, 2026")) + return httpx.Response(200, content=_announcement("November 5, 2026")) + + client = httpx.Client(transport=httpx.MockTransport(handler), follow_redirects=False) + assert capture_recent_sec_schedules( + tmp_path, "ACME", 12345, "HyperOptions test@example.com", + now=datetime(2026, 11, 1, 17, tzinfo=UTC), client=client, + ) == 1 + current = effective_forward_schedules( + tmp_path, "ACME", datetime(2026, 11, 1, 17, tzinfo=UTC), date(2026, 11, 30) + ) + assert [row["accession"] for row in current] == [new_accession] + + +def test_sec_exhibit_991_is_bounded_and_ambiguous_index_is_ignored(tmp_path): + submissions = { + "cik": 12345, + "tickers": ["ACME"], + "filings": { + "recent": { + "form": ["8-K"], + "accessionNumber": ["0000012345-26-000001"], + "primaryDocument": ["primary.htm"], + "acceptanceDateTime": ["2026-09-28T12:00:00"], + } + }, + } + names = ["exhibit99-1.htm"] + urls = [] + + def handler(request): + urls.append(str(request.url)) + if request.url.host == "data.sec.gov": + return httpx.Response(200, json=submissions) + if request.url.path.endswith("index.json"): + return httpx.Response( + 200, json={"directory": {"item": [{"name": name} for name in names]}} + ) + if request.url.path.endswith("exhibit99-1.htm"): + return httpx.Response(200, content=_announcement("November 5, 2026")) + return httpx.Response(200, content=b"

Item 9.01: exhibit attached.

") + + client = httpx.Client(transport=httpx.MockTransport(handler), follow_redirects=False) + now = datetime(2026, 9, 28, 17, tzinfo=UTC) + assert ( + capture_recent_sec_schedules( + tmp_path, "ACME", 12345, "HyperOptions test@example.com", now=now, client=client + ) + == 1 + ) + assert len(urls) == 4 and urls[-1].endswith("exhibit99-1.htm") + names.append("ex99_1.htm") + urls.clear() + assert ( + capture_recent_sec_schedules( + tmp_path / "ambiguous", + "ACME", + 12345, + "HyperOptions test@example.com", + now=now, + client=client, + ) + == 0 + ) + assert len(urls) == 3 + + +def test_sec_schedule_knowledge_time_is_document_fetch_completion(tmp_path): + accession = "0000012345-26-000001" + submissions = { + "cik": 12345, + "tickers": ["ACME"], + "filings": {"recent": { + "form": ["8-K"], + "accessionNumber": [accession], + "primaryDocument": ["primary.htm"], + "acceptanceDateTime": ["2026-09-28T12:00:00"], + }}, + } + + def handler(request): + if request.url.host == "data.sec.gov": + return httpx.Response(200, json=submissions) + if request.url.path.endswith("index.json"): + return httpx.Response( + 200, json={"directory": {"item": [{"name": "exhibit99-1.htm"}]}} + ) + if request.url.path.endswith("exhibit99-1.htm"): + return httpx.Response(200, content=_announcement("November 5, 2026")) + return httpx.Response(200, content=b"

Item 9.01: exhibit attached.

") + + started = datetime(2026, 9, 28, 17, tzinfo=UTC) + times = iter((started, started + timedelta(minutes=1), started + timedelta(minutes=2))) + client = httpx.Client(transport=httpx.MockTransport(handler), follow_redirects=False) + assert capture_recent_sec_schedules( + tmp_path, "ACME", 12345, "HyperOptions test@example.com", + client=client, observed_time=lambda: next(times), + ) == 1 + before = started + timedelta(minutes=1, seconds=30) + after = started + timedelta(minutes=2) + assert known_forward_schedules(tmp_path, "ACME", before, date(2026, 11, 30)) == () + assert known_forward_schedules(tmp_path, "ACME", after, date(2026, 11, 30))[0][ + "retrieved_at" + ] == after + + +def test_sec_capture_cli_uses_existing_data_dir_override(tmp_path, monkeypatch, capsys): + path = Path(__file__).parents[1] / "scripts" / "capture_sec_events.py" + spec = importlib.util.spec_from_file_location("capture_sec_events_test", path) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + observed = [] + monkeypatch.setenv("STOCKSWEEPER_DATA_DIR", str(tmp_path)) + monkeypatch.setenv("HYPEROPTIONS_SEC_USER_AGENT", "HyperOptions test@example.com") + monkeypatch.setattr( + module, + "capture_recent_sec_schedules", + lambda *arguments: observed.append(arguments) or 2, + ) + monkeypatch.setattr(sys, "argv", [str(path), "acme", "12345"]) + module.main() + assert observed == [(tmp_path, "ACME", 12345, "HyperOptions test@example.com")] + assert "forward schedules added: 2" in capsys.readouterr().out diff --git a/backend/tests/test_physical_capture.py b/backend/tests/test_physical_capture.py index 259613f..1966829 100644 --- a/backend/tests/test_physical_capture.py +++ b/backend/tests/test_physical_capture.py @@ -27,7 +27,12 @@ EXPIRY = date(2026, 10, 2) RETRIEVED = datetime(2026, 9, 25, 21, tzinfo=UTC) ISSUED = datetime(2026, 9, 26, 12, tzinfo=UTC) -METHODS = {"lognormal_ewma", "empirical_scaled", "student_t_ewma", "gjr_garch_t"} +METHODS = set(MODEL_VERSIONS) - {"intraday_shadow"} +FITTED_METHODS = METHODS - {"earnings_jump", "iv_physical"} +QUALIFIED_INPUT_REASONS = { + "earnings_jump": "verified_release_time_history_unavailable", + "iv_physical": "rights_cleared_option_history_unavailable", +} def _distribution( @@ -91,7 +96,7 @@ def _capture(tmp_path, *, verified: bool = True) -> PhysicalShadowCapture: def _candidates(*, digest: str = "a" * 64) -> dict[str, ShadowForecast]: return { method: ShadowForecast(_distribution(method, digest=digest), None, 1.0, 1.0) - for method in METHODS + for method in FITTED_METHODS } @@ -109,11 +114,19 @@ def forecast(*args, **kwargs): rows = ForecastLedger(tmp_path).evaluation_rows() assert len(calls) == 1 # Fit once per ticker, input session, and expiry. assert {row["method"] for row in rows} == METHODS - assert all(row["status"] == "available" for row in rows) + assert all(row["status"] == "available" for row in rows if row["method"] in FITTED_METHODS) + assert { + row["method"]: row["unavailable_reason"] for row in rows + if row["status"] == "unavailable" + } == QUALIFIED_INPUT_REASONS assert all(row["input_session"] == INPUT and row["data_hash"] == "a" * 64 for row in rows) - assert all(row["distribution_hash"] for row in rows) + assert all(row["distribution_hash"] for row in rows if row["status"] == "available") + assert all(row["distribution_hash"] is None for row in rows if row["status"] == "unavailable") assert all(row["provenance"] == "as_issued" for row in rows) - assert all(row["prepare_ms"] == 1.0 and row["lookup_ms"] == 1.0 for row in rows) + assert all( + row["prepare_ms"] == 1.0 and row["lookup_ms"] == 1.0 + for row in rows if row["status"] == "available" + ) baseline = next(row for row in rows if row["method"] == "lognormal_ewma") assert baseline["itm_probability"] == pytest.approx(0.5) assert baseline["otm_probability"] == pytest.approx(0.5) @@ -216,14 +229,20 @@ def test_long_horizon_capture_does_not_record_false_baseline_outage(tmp_path) -> ) capture._capture_sync([(issue, distribution)]) - coverage = ForecastLedger(tmp_path).coverage() + coverage = ForecastLedger(tmp_path).evaluation_rows() baseline = [row for row in coverage if row["model_version"] == distribution.model_version] assert baseline and all(row["status"] == "available" for row in baseline) assert { - row["unavailable_reason"] - for row in coverage - if row["model_version"] != distribution.model_version - } == {"shadow_horizon_unsupported"} + row["method"]: row["unavailable_reason"] + for row in coverage if row["method"] in FITTED_METHODS - {"lognormal_ewma"} + } == { + method: "shadow_horizon_unsupported" + for method in FITTED_METHODS - {"lognormal_ewma"} + } + assert { + row["method"]: row["unavailable_reason"] + for row in coverage if row["method"] in QUALIFIED_INPUT_REASONS + } == QUALIFIED_INPUT_REASONS def test_shadow_failure_cannot_mask_valid_live_method(tmp_path) -> None: @@ -310,13 +329,24 @@ def test_unavailable_first_contract_does_not_suppress_valid_peer(tmp_path) -> No ) capture._capture_sync([(unavailable, None), (contracts[1], _distribution())]) rows = ForecastLedger(tmp_path).evaluation_rows() - assert sum(row["status"] == "available" for row in rows) == 4 - assert {row["unavailable_reason"] for row in rows if row["status"] == "unavailable"} == { - "input_vintage_changed" - } + by_contract = {} + for row in rows: + by_contract.setdefault(row["contract_key"], {})[row["method"]] = row + valid = by_contract[contracts[1].contract_key] + invalid = by_contract[unavailable.contract_key] + assert set(valid) == set(invalid) == METHODS + assert all(valid[method]["status"] == "available" for method in FITTED_METHODS) + assert all( + invalid[method]["unavailable_reason"] == "input_vintage_changed" + for method in FITTED_METHODS + ) + assert { + method: valid[method]["unavailable_reason"] + for method in QUALIFIED_INPUT_REASONS + } == QUALIFIED_INPUT_REASONS -def test_manifest_mismatch_records_four_unavailable_attempts_without_fitting(tmp_path) -> None: +def test_manifest_mismatch_records_every_unavailable_attempt_without_fitting(tmp_path) -> None: capture = _capture(tmp_path, verified=False) capture.forecaster = SimpleNamespace( forecast_candidates=lambda *_args, **_kwargs: pytest.fail("unverified cache was fitted") @@ -326,7 +356,14 @@ def test_manifest_mismatch_records_four_unavailable_attempts_without_fitting(tmp rows = ForecastLedger(tmp_path).evaluation_rows() assert {row["method"] for row in rows} == METHODS assert all(row["status"] == "unavailable" for row in rows) - assert all(row["unavailable_reason"] == "input_provenance_unverified" for row in rows) + assert all( + row["unavailable_reason"] == "input_provenance_unverified" + for row in rows if row["method"] in FITTED_METHODS + ) + assert { + row["method"]: row["unavailable_reason"] + for row in rows if row["method"] in QUALIFIED_INPUT_REASONS + } == QUALIFIED_INPUT_REASONS assert all(row["itm_probability"] is None and row["distribution_hash"] is None for row in rows) assert capture.candidate( _distribution("lognormal_ewma"), "student_t_ewma", EXPIRY, INPUT @@ -341,9 +378,16 @@ def test_changed_input_vintage_is_rejected_even_if_fit_succeeds(tmp_path) -> Non capture._capture_sync([(_issue(), _distribution())]) rows = ForecastLedger(tmp_path).evaluation_rows() - assert len(rows) == 4 + assert len(rows) == len(METHODS) assert all(row["status"] == "unavailable" for row in rows) - assert all(row["unavailable_reason"] == "input_vintage_changed" for row in rows) + assert all( + row["unavailable_reason"] == "input_vintage_changed" + for row in rows if row["method"] in FITTED_METHODS + ) + assert { + row["method"]: row["unavailable_reason"] + for row in rows if row["method"] in QUALIFIED_INPUT_REASONS + } == QUALIFIED_INPUT_REASONS def test_candidate_fit_error_is_recorded_as_fit_failure_not_bad_provenance(tmp_path) -> None: @@ -358,7 +402,14 @@ def broken_fit(*_args, **_kwargs): rows = ForecastLedger(tmp_path).evaluation_rows() assert {row["method"] for row in rows} == METHODS assert all(row["status"] == "unavailable" for row in rows) - assert all(row["unavailable_reason"] == "shadow_fit_failed" for row in rows) + assert all( + row["unavailable_reason"] == "shadow_fit_failed" + for row in rows if row["method"] in FITTED_METHODS + ) + assert { + row["method"]: row["unavailable_reason"] + for row in rows if row["method"] in QUALIFIED_INPUT_REASONS + } == QUALIFIED_INPUT_REASONS assert capture.candidate( _distribution("lognormal_ewma"), "student_t_ewma", EXPIRY, INPUT ).reason == "shadow_fit_failed" @@ -381,8 +432,8 @@ def test_restart_does_not_duplicate_the_same_shadow_issuances(tmp_path) -> None: capture._capture_sync(entries) rows = ForecastLedger(tmp_path).evaluation_rows() - assert len(rows) == 4 - assert len({row["idempotency_key"] for row in rows}) == 4 + assert len(rows) == len(METHODS) + assert len({row["idempotency_key"] for row in rows}) == len(METHODS) def test_contracts_over_fit_capacity_leave_recorded_unavailable_attempts( @@ -400,15 +451,14 @@ def test_contracts_over_fit_capacity_leave_recorded_unavailable_attempts( capture._capture_sync(contracts) rows = ForecastLedger(tmp_path).evaluation_rows() - assert len(rows) == 8 + assert len(rows) == 2 * len(METHODS) assert {row["contract_key"] for row in rows} == { _issue("100.000").contract_key, _issue("110.000").contract_key, } - assert len([row for row in rows if row["status"] == "available"]) == 4 - assert ( - len([row for row in rows if row["unavailable_reason"] == "shadow_capacity_exceeded"]) == 4 - ) + assert len([row for row in rows if row["status"] == "available"]) == len(FITTED_METHODS) + capped = [row for row in rows if row["unavailable_reason"] == "shadow_capacity_exceeded"] + assert len(capped) == len(METHODS) skipped = max( (issue for issue, _ in contracts), key=lambda issue: hashlib.sha256(issue.contract_key.encode()).digest(), diff --git a/backend/tests/test_physical_contest.py b/backend/tests/test_physical_contest.py index d440cb3..f5bc3fe 100644 --- a/backend/tests/test_physical_contest.py +++ b/backend/tests/test_physical_contest.py @@ -2,11 +2,13 @@ from __future__ import annotations +import builtins from dataclasses import replace from datetime import UTC, date, datetime from importlib.util import module_from_spec, spec_from_file_location from math import isclose from pathlib import Path +from types import SimpleNamespace import numpy as np import polars as pl @@ -19,6 +21,9 @@ PhysicalShadowForecaster, ShadowForecast, _fit_gjr, + _fit_har, + _fit_arch_skew, + _fit_markov, _gjr_terminal, _student_terminal, ) @@ -92,6 +97,56 @@ def gjr(returns, splits): assert first["lognormal_ewma"].distribution.method == "lognormal_ewma" +def test_gjr_cold_import_is_outside_fit_limit_but_slow_fit_is_rejected(tmp_path, monkeypatch): + from stocksweeper.forecast import physical_contest + + session = date(2026, 9, 25) + frame = _bars(700, session) + shadow = PhysicalShadowForecaster(PredictiveForecaster(tmp_path, _Prices(frame))) + clock = [0.0] + fit_seconds = [0.015] + imported = [False] + result = SimpleNamespace( + convergence_flag=0, + params={"omega": 0.01, "alpha[1]": 0.05, "gamma[1]": 0.04, "beta[1]": 0.9, "nu": 6.0}, + conditional_volatility=np.ones(700), + ) + + def fit(**_kwargs): + clock[0] += fit_seconds[0] + return result + + fake_arch = SimpleNamespace(arch_model=lambda *_args, **_kwargs: SimpleNamespace(fit=fit)) + original_import = builtins.__import__ + + def timed_import(name, globals=None, locals=None, fromlist=(), level=0): + if name == "arch": + if not imported[0]: + clock[0] += 5.1 + imported[0] = True + return fake_arch + return original_import(name, globals, locals, fromlist, level) + + monkeypatch.setattr(builtins, "__import__", timed_import) + monkeypatch.setattr(physical_contest, "perf_counter", lambda: clock[0]) + + def skip(*_args, **_kwargs): + return None, None + + for name in ("_fit_student", "_fit_har", "_fit_arch_skew", "_fit_markov"): + monkeypatch.setattr(physical_contest, name, skip) + + fast, _ = shadow._fit("TEST", session, frame, "cold") + assert fast.gjr_reason is None + assert fast.gjr_fit_ms == pytest.approx(15.0) + assert shadow._fit("TEST", session, frame, "cold")[0] is fast + + fit_seconds[0] = 5.001 + slow, _ = shadow._fit("TEST", session, frame, "slow") + assert slow.gjr_reason == "gjr_fit_latency_exceeded" + assert slow.gjr_parameters is None + + def test_gjr_requires_long_split_safe_history_and_scenarios_are_seeded(): short = np.random.default_rng(0).normal(0, 0.01, 499) assert _fit_gjr(short, np.zeros(499, dtype=bool))[1] == "gjr_history_short" @@ -106,6 +161,126 @@ def test_gjr_requires_long_split_safe_history_and_scenarios_are_seeded(): ) +def test_singular_candidate_fit_reports_only_that_model_unavailable(monkeypatch): + import arch.univariate + import statsmodels.tsa.regime_switching.markov_regression as markov_regression + + def singular(*_args, **_kwargs): + raise np.linalg.LinAlgError("singular history") + + monkeypatch.setattr(arch.univariate, "ZeroMean", singular) + monkeypatch.setattr(markov_regression, "MarkovRegression", singular) + returns = np.random.default_rng(1).normal(0, 0.01, 500) + splits = np.zeros(500, dtype=bool) + assert _fit_arch_skew(returns, splits, egarch=False)[1] == "skew_ewma_fit_failed" + assert _fit_markov(returns, splits)[1] == "markov_fit_failed" + + +def test_gjr_boundary_roundoff_is_accepted_but_material_violation_is_not(monkeypatch): + import arch + + parameters = { + "omega": 0.01, + "alpha[1]": 0.1, + "gamma[1]": -0.1 - 8.58e-14, + "beta[1]": 0.8, + "nu": 6.0, + } + result = SimpleNamespace( + convergence_flag=0, params=parameters, conditional_volatility=np.ones(500) + ) + monkeypatch.setattr( + arch, "arch_model", lambda *_args, **_kwargs: SimpleNamespace(fit=lambda **_kwargs: result) + ) + returns = np.zeros(500) + splits = np.zeros(500, dtype=bool) + accepted, reason = _fit_gjr(returns, splits) + assert reason is None + assert accepted is not None and accepted[1] + accepted[2] == 0 + parameters["gamma[1]"] = -0.100001 + assert _fit_gjr(returns, splits)[1] == "gjr_parameters_invalid" + + +def test_added_completed_close_models_share_seeded_terminal_distribution(tmp_path): + session = date(2026, 9, 25) + frame = _bars(700, session) + forecaster = PredictiveForecaster(tmp_path, _Prices(frame)) + forecaster.prepare("TEST", session) + shadow = PhysicalShadowForecaster(forecaster) + now = datetime(2026, 9, 26, 12, tzinfo=UTC) + expiry = date(2026, 10, 9) + first = shadow.forecast_candidates("TEST", now, expiry) + repeat = shadow.forecast_candidates("TEST", now, expiry) + for name in ("ohlc_har", "skew_t_ewma", "egarch_skew_t", "markov_switching"): + distribution = first[name].distribution + assert distribution is not None, (name, first[name].reason) + assert distribution.prices == repeat[name].distribution.prices + assert len(distribution.prices) == 4096 + assert sum(distribution.weights) == pytest.approx(1) + assert distribution.probability("call", distribution.spot) + distribution.probability( + "put", distribution.spot + ) == pytest.approx(1) + assert distribution.probability("call", distribution.spot * 1.1) < distribution.probability( + "call", distribution.spot + ) + assert first["ngboost_pooled"].distribution is None + assert first["ngboost_pooled"].reason == "training_source_rights_unverified" + + +def test_added_models_reject_short_or_split_affected_history(): + frame = _bars(220, date(2026, 9, 25)) + returns = np.diff(np.log(np.asarray(frame["close"].to_list()))) + splits = np.zeros(len(returns), dtype=bool) + assert _fit_har(frame)[1] == "har_history_short" + assert _fit_arch_skew(returns, splits, egarch=True)[1] == "egarch_history_short" + assert _fit_markov(returns, splits)[1] == "markov_history_short" + long = _bars(700, date(2026, 9, 25)) + split = np.zeros(699, dtype=bool) + split[650] = True + long_returns = np.diff(np.log(np.asarray(long["close"].to_list()))) + assert _fit_arch_skew(long_returns, split, egarch=False)[1] == ( + "skew_ewma_split_safe_history_short" + ) + assert _fit_markov(long_returns, split)[1] == "markov_split_safe_history_short" + split[:] = False + split[100] = True + assert _fit_arch_skew(long_returns, split, egarch=False)[1] is None + assert _fit_markov(long_returns, split)[1] is None + long = long.with_columns( + pl.when(pl.int_range(pl.len()) == 650).then(2.0).otherwise(0.0).alias("stock_splits") + ) + assert _fit_har(long)[1] == "har_history_short" + + +def test_qualified_pooled_artifact_stays_unavailable_without_issuance_identity( + tmp_path, monkeypatch +): + session = date(2026, 9, 25) + forecaster = PredictiveForecaster(tmp_path, _Prices(_bars(220, session))) + forecaster.prepare("TEST", session) + shadow = PhysicalShadowForecaster(forecaster) + calls = {"load": 0, "predict": 0} + + def load(_data_dir): + calls["load"] += 1 + return {"qualified": True} + + def predict(_artifact, _frame, horizon, as_of): + calls["predict"] += 1 + raise AssertionError("untraceable pooled prediction must not be issued") + + monkeypatch.setattr("stocksweeper.forecast.physical_contest.load_pooled_model", load) + monkeypatch.setattr("stocksweeper.forecast.physical_contest.predict_pooled_params", predict) + now = datetime(2026, 9, 26, 12, tzinfo=UTC) + first = shadow.forecast_candidates("TEST", now, date(2026, 10, 2))["ngboost_pooled"] + repeat = shadow.forecast_candidates("TEST", now, date(2026, 10, 2))["ngboost_pooled"] + other = shadow.forecast_candidates("TEST", now, date(2026, 10, 9))["ngboost_pooled"] + for result in (first, repeat, other): + assert result.distribution is None + assert result.reason == "training_artifact_traceability_unavailable" + assert calls == {"load": 1, "predict": 0} + + @pytest.mark.parametrize("horizon", (25, 26, 40)) def test_baseline_remains_available_beyond_challenger_horizon(tmp_path, horizon): session = date(2026, 9, 25) @@ -122,7 +297,16 @@ def test_baseline_remains_available_beyond_challenger_horizon(tmp_path, horizon) assert all( forecasts[name].distribution is None and forecasts[name].reason == "shadow_horizon_unsupported" - for name in ("empirical_scaled", "student_t_ewma", "gjr_garch_t") + for name in ( + "empirical_scaled", + "student_t_ewma", + "gjr_garch_t", + "ohlc_har", + "skew_t_ewma", + "egarch_skew_t", + "markov_switching", + "ngboost_pooled", + ) ) @@ -386,7 +570,7 @@ def evaluation_rows(self, *, provenance): "unavailable_reason": "gjr_nonconverged", "distribution_hash": None, }, - ] + ] def iter_evaluation_rows(self, **kwargs): assert kwargs["methods"] == ("lognormal_ewma", "student_t_ewma") @@ -399,9 +583,7 @@ def panel_coverage(self, **_kwargs): def evaluation_skipped_attempts(self, provenance): return {} - report = ledger_contest( - Ledger(), SessionCalendar(), "as_issued", "student_t_ewma", None, "all" - ) + report = ledger_contest(Ledger(), SessionCalendar(), "as_issued", "student_t_ewma", None, "all") first = report["bands"]["1"] assert first["baseline_contract_forecasts_available"] == 1 assert first["contract_forecasts_available"] == 0 diff --git a/backend/tests/test_pooled_ngboost.py b/backend/tests/test_pooled_ngboost.py new file mode 100644 index 0000000..2322501 --- /dev/null +++ b/backend/tests/test_pooled_ngboost.py @@ -0,0 +1,96 @@ +"""Qualified pooled training is offline, causal, and pickle-free.""" + +from __future__ import annotations + +import json +from datetime import date +from math import log + +import numpy as np +import polars as pl +import pytest + +from stocksweeper.forecast import pooled_ngboost +from stocksweeper.forecast.calendar import SessionCalendar +from stocksweeper.forecast.provenance import SourceRightsUnverified + + +def _bars(count: int, last: date) -> pl.DataFrame: + sessions = SessionCalendar().sessions(date(2020, 1, 1), last)[-count:] + returns = np.random.default_rng(17).standard_t(5, count) * 0.012 + closes = (100 * np.exp(np.cumsum(returns))).tolist() + return pl.DataFrame( + { + "ts": sessions, + "open": closes, + "high": [price * 1.01 for price in closes], + "low": [price * 0.99 for price in closes], + "close": closes, + "volume": [1000.0] * count, + "dividends": [0.0] * count, + "stock_splits": [0.0] * count, + } + ) + + +def test_current_unqualified_cohort_cannot_train(tmp_path, monkeypatch): + def denied(*_args, **_kwargs): + raise SourceRightsUnverified("training_rights_unverified") + + monkeypatch.setattr(pooled_ngboost, "load_training_cohort", denied) + with pytest.raises(SourceRightsUnverified, match="training_rights_unverified"): + pooled_ngboost.train_pooled_model(tmp_path) + assert not (tmp_path / "forecast" / "training" / "ngboost").exists() + + +def test_qualified_synthetic_cohort_trains_and_roundtrips_without_pickle(tmp_path, monkeypatch): + session = date(2026, 9, 25) + frame = _bars(520, session) + cohort = tuple((f"FIXTURE{i}", frame, f"{i:064x}") for i in range(20)) + monkeypatch.setattr(pooled_ngboost, "load_training_cohort", lambda *_args, **_kwargs: cohort) + manifest = tmp_path / "forecast" / "training" / "cohort.json" + manifest.parent.mkdir(parents=True) + manifest.write_text( + json.dumps( + { + "rights_status": "approved_for_training", + "source_name": "Licensed synthetic fixture", + "license_reference": "test-license", + "manifest_hash": "f" * 64, + } + ) + ) + fitted = {} + original_export = pooled_ngboost._export_model + + def capture(model, *args): + fitted["model"] = model + return original_export(model, *args) + + monkeypatch.setattr(pooled_ngboost, "_export_model", capture) + path = pooled_ngboost.train_pooled_model(tmp_path) + assert path.suffix == ".json" + assert path.stat().st_size < pooled_ngboost.MAX_ARTIFACT_BYTES + artifact = pooled_ngboost.load_pooled_model(tmp_path) + mean, scale = pooled_ngboost.predict_pooled_params(artifact, frame, 5, session) + assert np.isfinite(mean) and np.isfinite(scale) and scale > 0 + features = pooled_ngboost._features(frame, frame.height - 1, 5) + direct = fitted["model"].pred_param(np.asarray([features]))[0] + np.testing.assert_allclose((mean, log(scale)), direct, rtol=1e-12, atol=1e-12) + assert (mean, scale) == pooled_ngboost.predict_pooled_params(artifact, frame, 5, session) + with pytest.raises(ValueError, match="pooled_artifact_future_leak"): + pooled_ngboost.predict_pooled_params(artifact, frame, 5, date(2026, 9, 24)) + with pytest.raises(ValueError, match="pooled_inputs_invalid"): + pooled_ngboost.predict_pooled_params(artifact, frame, 26, session) + + def denied(*_args, **_kwargs): + raise SourceRightsUnverified("training_rights_unverified") + + monkeypatch.setattr(pooled_ngboost, "load_training_cohort", denied) + with pytest.raises(SourceRightsUnverified, match="training_rights_unverified"): + pooled_ngboost.load_pooled_model(tmp_path) + monkeypatch.setattr(pooled_ngboost, "load_training_cohort", lambda *_args, **_kwargs: cohort) + + path.write_bytes(path.read_bytes() + b" ") + with pytest.raises(ValueError, match="pooled_artifact_invalid"): + pooled_ngboost.load_pooled_model(tmp_path) diff --git a/backend/tests/test_quant_api.py b/backend/tests/test_quant_api.py index f90bf9b..4051ff8 100644 --- a/backend/tests/test_quant_api.py +++ b/backend/tests/test_quant_api.py @@ -94,6 +94,9 @@ async def load(*_args, **_kwargs): ) default_response = client.get("/api/covered-calls/IREN?moneyness=all") invalid_response = client.get("/api/covered-calls/IREN?forecast_model=unknown") + rights_unavailable = client.get( + "/api/covered-calls/IREN?moneyness=all&forecast_model=iv_physical" + ) page = assemble_covered_calls(chain, info, history, today, now, "all") def failed_evidence(_entries): @@ -120,11 +123,26 @@ def failed_evidence(_entries): assert row["predictive_odds"]["method"] == "empirical_scaled" assert [view["method"] for view in row["physical_models"]] == [ "lognormal_ewma", "empirical_scaled", "student_t_ewma", - "gjr_garch_t", "intraday_shadow", + "gjr_garch_t", "ohlc_har", "skew_t_ewma", "egarch_skew_t", + "markov_switching", "ngboost_pooled", "earnings_jump", + "iv_physical", "intraday_shadow", ] assert [view["method"] for view in row["market_models"]] == [ - "regimelib", "constrained_call_curve", + "regimelib", "constrained_call_curve", "ssvi", ] + by_method = {view["method"]: view for view in row["physical_models"]} + assert by_method["iv_physical"]["status"] == "unavailable" + assert by_method["iv_physical"]["reason"] == ( + "rights_cleared_option_history_unavailable" + ) + assert by_method["earnings_jump"]["status"] == "unavailable" + assert by_method["earnings_jump"]["reason"] == ( + "verified_release_time_history_unavailable" + ) + assert row["market_models"][-1]["status"] == "unavailable" + assert row["market_models"][-1]["reason"] == ( + "rights_cleared_option_history_unavailable" + ) assert row["hypothetical_risk"]["forecast_method"] == "empirical_scaled" assert row["predictive_odds"]["model_version"] == "test-v1" assert row["hypothetical_risk"]["status"] == "available" @@ -159,3 +177,19 @@ def failed_evidence(_entries): assert all( view["status"] == "unavailable" for view in unavailable_row["physical_models"] ) + rights_row = next( + contract for group in rights_unavailable.json()["expirations"] + for contract in group["contracts"] if contract["watch_key"] is not None + ) + assert rights_row["predictive_odds"]["method"] == "iv_physical" + assert rights_row["predictive_odds"]["status"] == "unavailable" + assert rights_row["predictive_odds"]["reason"] == ( + "rights_cleared_option_history_unavailable" + ) + assert rights_row["predictive_odds"]["itm_pct_tenths"] is None + assert rights_row["hypothetical_risk"]["status"] == "unavailable" + assert rights_row["hypothetical_risk"]["forecast_method"] == "iv_physical" + assert next( + view for view in rights_row["physical_models"] + if view["method"] == "empirical_scaled" + )["status"] == "available" diff --git a/backend/tests/test_watchlist.py b/backend/tests/test_watchlist.py index a1e6755..b3c47e3 100644 --- a/backend/tests/test_watchlist.py +++ b/backend/tests/test_watchlist.py @@ -21,6 +21,7 @@ from options_api.models import MarketOddsView, OptionChainResponse, OptionQuote, StockInfoResponse from options_api.models import TickerListing from options_api.main import create_app +from options_api.physical_shadow_capture import MODEL_VERSIONS from options_api.outcomes import ( CloseHistory, OutcomeResult, @@ -31,6 +32,8 @@ from options_api.parser import parse_option_chain from options_api.watchlist import WatchStore, WatchlistService from stocksweeper.config import Settings +from stocksweeper.forecast.physical_contest import ShadowForecast +from stocksweeper.forecast.capture_windows import reconcile_capture_windows from stocksweeper.forecast.predictive import PredictiveDistribution from stocksweeper.pipeline.jobs import JobBusy, JobManager from stocksweeper.storage.db import connect @@ -41,6 +44,124 @@ D = Decimal +def test_removing_watch_preserves_due_capture_window(tmp_path: Path) -> None: + store = WatchStore(tmp_path) + expiry = date(2026, 10, 30) + key = make_watch_key("AAPL", "AAPL", "call", expiry.isoformat(), D("100")) + watch, _ = store.add( + key, "AAPL", "AAPL", "call", expiry, D("100"), + datetime(2026, 9, 28, 13, 59, tzinfo=UTC), + ) + assert store.delete(watch.id, now=datetime(2026, 9, 28, 14, 2, tzinfo=UTC)) + assert store.list() == [] + reconcile_capture_windows(tmp_path, [], datetime(2026, 9, 28, 14, 6, tzinfo=UTC)) + with connect(tmp_path / "results.duckdb") as connection: + events = connection.execute( + """SELECT event FROM forecast_capture_window_events + WHERE contract_key = ? AND snapshot_window = '10:00' ORDER BY event""", + [key], + ).fetchall() + assert [event for (event,) in events] == ["expected", "missed"] + + +def test_removing_watch_reconciles_only_that_contract(tmp_path: Path) -> None: + store = WatchStore(tmp_path) + created = datetime(2026, 9, 28, 13, 59, tzinfo=UTC) + keys = [ + make_watch_key("AAPL", "AAPL", "call", "2026-10-30", D(str(strike))) + for strike in (100, 105) + ] + watches = [ + store.add(key, "AAPL", "AAPL", "call", date(2026, 10, 30), D(str(strike)), created)[0] + for key, strike in zip(keys, (100, 105), strict=True) + ] + reconcile_capture_windows(tmp_path, watches, datetime(2026, 9, 28, 14, 2, tzinfo=UTC)) + assert store.delete(watches[0].id, now=datetime(2026, 9, 28, 14, 6, tzinfo=UTC)) + with connect(tmp_path / "results.duckdb") as connection: + events = connection.execute( + """SELECT contract_key, event FROM forecast_capture_window_events + WHERE contract_key IN (?, ?) ORDER BY contract_key, event""", + keys, + ).fetchall() + assert set(events) == { + (keys[0], "expected"), (keys[0], "missed"), (keys[1], "expected") + } + + +def test_watchlist_selects_each_new_method_without_substituting_ewma( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + now = datetime(2026, 9, 25, 21, tzinfo=UTC) + expiry = date(2026, 10, 2) + base = PredictiveDistribution( + ticker="IREN", status="available", reason=None, method="lognormal_ewma", + as_of=date(2026, 9, 25), expiry_session=expiry, horizon_sessions=5, + spot=100.0, daily_volatility=0.02, + model_version=MODEL_VERSIONS["lognormal_ewma"], support=60, + data_hash="verified-input", terminal_prices=(80.0, 120.0), weights=(0.5, 0.5), + ) + fitted = ( + "ohlc_har", "skew_t_ewma", "egarch_skew_t", "markov_switching", + "ngboost_pooled", + ) + qualified_reasons = { + "earnings_jump": "verified_release_time_history_unavailable", + "iv_physical": "rights_cleared_option_history_unavailable", + } + app = create_app( + clock=lambda: now, prefetch_universe=False, predictive_refresh=False, + research_settings=Settings(data_dir=tmp_path), + ) + with TestClient(app, base_url="http://127.0.0.1") as client: + app.state.watchlist.store.add( + make_watch_key("IREN", "IREN", "call", expiry.isoformat(), D("100")), + "IREN", "IREN", "call", expiry, D("100"), + datetime(2026, 9, 24, 18, tzinfo=UTC), + ) + market = app.state.market_odds + monkeypatch.setattr(market, "schedule", lambda _tickers: None) + monkeypatch.setattr(market, "lookup", lambda *_args: MarketOddsView(status="unavailable")) + monkeypatch.setattr(market, "lookup_last_good", lambda *_args: None) + monkeypatch.setattr(market, "entry_quote", lambda *_args: None) + predictive = app.state.predictive_odds + monkeypatch.setattr(predictive, "schedule", lambda _tickers: None) + monkeypatch.setattr(predictive, "distribution", lambda *_args, **_kwargs: base) + monkeypatch.setattr(predictive, "cache_retrieved_at", lambda *_args: now) + shadow = app.state.physical_shadow + monkeypatch.setattr(shadow, "submit", lambda _entries: None) + original_candidate = shadow.candidate + + def candidate(distribution, method, *args, **kwargs): + if method in fitted: + return ShadowForecast( + replace( + distribution, method=method, model_version=MODEL_VERSIONS[method], + terminal_prices=(90.0, 110.0), weights=(0.6, 0.4), + ), None, 1, 1, + ) + return original_candidate(distribution, method, *args, **kwargs) + + monkeypatch.setattr(shadow, "candidate", candidate) + for method in (*fitted, *qualified_reasons): + response = client.get(f"/api/watchlist?forecast_model={method}") + assert response.status_code == 200 + item = response.json()["items"][0] + selected = item["predictive_odds"] + by_method = {model["method"]: model for model in item["physical_models"]} + assert selected == by_method[method] + assert by_method["lognormal_ewma"]["status"] == "available" + assert item["hypothetical_risk"]["forecast_method"] == method + if method in fitted: + assert selected["status"] == "available" + assert (selected["itm_pct_tenths"], selected["otm_pct_tenths"]) == (400, 600) + else: + assert selected["status"] == "unavailable" + assert selected["reason"] == qualified_reasons[method] + assert selected["itm_pct_tenths"] is None + assert item["hypothetical_risk"]["status"] == "unavailable" + + + def test_saved_adjusted_root_cannot_suppress_standard_watch_forecast( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: diff --git a/backend/uv.lock b/backend/uv.lock index abc4b60..6b7470b 100644 --- a/backend/uv.lock +++ b/backend/uv.lock @@ -59,6 +59,28 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/ef/86/612d45473d0865d41934b0580fa05e6aa48167b502d0136e8bd9dd5aa581/arch-8.0.0-cp312-cp312-win_amd64.whl", hash = "sha256:8b13d261e0a681b3a8a2f9c588ab37a35500bca9f3bbcc6ca1ce2d999322651d", size = 930370, upload-time = "2025-10-21T08:42:14.667Z" }, ] +[[package]] +name = "autograd" +version = "1.9.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "numpy" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/9d/ac/4a8b58364ba865ab545251edbe475e5a35326814a5d274a2211d33ed8a5d/autograd-1.9.1.tar.gz", hash = "sha256:7818c5c69ddf9efb7da74097bd741a4c7920133f41966f68537223a15ef798bc", size = 2566975, upload-time = "2026-07-02T07:43:58.255Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/6e/ce/8c98e6604bb1ec9d03c8493c328185b69bb7533fe08f3640f0f3641bd9d1/autograd-1.9.1-py3-none-any.whl", hash = "sha256:b788bae3fa010cbffb4cfb7b8ba2a3f0daa6072a8506da6164c779fe9cf3e05a", size = 54387, upload-time = "2026-07-02T07:43:56.502Z" }, +] + +[[package]] +name = "autograd-gamma" +version = "0.5.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "autograd" }, + { name = "scipy" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/85/ae/7f2031ea76140444b2453fa139041e5afd4a09fc5300cfefeb1103291f80/autograd-gamma-0.5.0.tar.gz", hash = "sha256:f27abb7b8bb9cffc8badcbf59f3fe44a9db39e124ecacf1992b6d952934ac9c4", size = 3952, upload-time = "2020-10-15T16:51:06.785Z" } + [[package]] name = "beautifulsoup4" version = "4.15.0" @@ -154,6 +176,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/58/50/6c0d534c5f134586a8e1ba4e330569e32f057e33372ae556463212fb4cd3/click-8.5.0-py3-none-any.whl", hash = "sha256:255bc9599cf7748b4b1a446ccc735421bd08a2ae529a8b88597d3de5664ee360", size = 125251, upload-time = "2026-08-26T13:33:12.928Z" }, ] +[[package]] +name = "cloudpickle" +version = "3.1.2" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/27/fb/576f067976d320f5f0114a8d9fa1215425441bb35627b1993e5afd8111e5/cloudpickle-3.1.2.tar.gz", hash = "sha256:7fda9eb655c9c230dab534f1983763de5835249750e85fbcef43aaa30a9a2414", size = 22330, upload-time = "2025-11-03T09:25:26.604Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/88/39/799be3f2f0f38cc727ee3b4f1445fe6d5e4133064ec2e4115069418a5bb6/cloudpickle-3.1.2-py3-none-any.whl", hash = "sha256:9acb47f6afd73f60dc1df93bb801b472f05ff42fa6c84167d25cb206be1fbf4a", size = 22228, upload-time = "2025-11-03T09:25:25.534Z" }, +] + [[package]] name = "colorama" version = "0.4.6" @@ -163,6 +194,28 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/d1/d6/3965ed04c63042e047cb6a3e6ed1a63a35087b6a609aa3a15ed8ac56c221/colorama-0.4.6-py2.py3-none-any.whl", hash = "sha256:4f1d9991f5acc0ca119f9d443620b77f9d6b33703e51011c16baf57afb285fc6", size = 25335, upload-time = "2022-10-25T02:36:20.889Z" }, ] +[[package]] +name = "contourpy" +version = "1.4.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "numpy" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/83/5a/a55177dd22553a277388e8a1b3220e92de91bacb28356cdc73caa240121d/contourpy-1.4.0.tar.gz", hash = "sha256:20156f5a1ac4f8ce02656e39a61e82164a3d359796dc8026f75b062783d500e1", size = 13323726, upload-time = "2026-09-11T19:05:05.808Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/97/ba/01bde9753bdbba04a6da9d2bff3881f021bc708f654655e95fae26d3b3a3/contourpy-1.4.0-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:186ba929df36d61b6127da2e89cd1357e4cb79aca647381ab1a6feb0b152877b", size = 297302, upload-time = "2026-09-11T19:02:45.103Z" }, + { url = "https://files.pythonhosted.org/packages/dd/f5/c5bf67522d49a2222f3154fe46879451d26f0e35ece41770758a64ea78cb/contourpy-1.4.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:c76f3a5318164db7d9401132fc1b5364b784c613c93fa506e3b5e6d1bf353ec8", size = 284921, upload-time = "2026-09-11T19:02:48.049Z" }, + { url = "https://files.pythonhosted.org/packages/0f/cc/cb989599eec12fda312e127eb8e04a8b21a7a6c94bf7ff1bc68273334a42/contourpy-1.4.0-cp312-cp312-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:43c3ccbb32c6294b183dcc8e8c46dacf5ecef809497e3d48a5be298eef185ad0", size = 356332, upload-time = "2026-09-11T19:02:51.408Z" }, + { url = "https://files.pythonhosted.org/packages/47/fb/6f620d7602507817b0b4c21dd790ed68688f1f4ae9c29aacf21f08db5cf1/contourpy-1.4.0-cp312-cp312-manylinux_2_26_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:86d05cec773c9507a3950122e0e40ce77c23c75ceeb2fc189514e71893cbb34b", size = 405878, upload-time = "2026-09-11T19:02:54.093Z" }, + { url = "https://files.pythonhosted.org/packages/32/4e/693c6d6bdece679f0953775eef3a1b66d56be55478d936e7deb1f3004422/contourpy-1.4.0-cp312-cp312-manylinux_2_26_s390x.manylinux_2_28_s390x.whl", hash = "sha256:2deb580178ca19437a84bd77e4bc2cd91a8ccad212413a83c273900868b5978c", size = 408296, upload-time = "2026-09-11T19:02:56.528Z" }, + { url = "https://files.pythonhosted.org/packages/78/f9/b6831508960d559581c448532ba4df312217072975486dbf2b24fcc5b76f/contourpy-1.4.0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:875f42444c9cf48d56f724f2637e60d0f73b3b12c9041e1484580a233edf9591", size = 383243, upload-time = "2026-09-11T19:02:58.638Z" }, + { url = "https://files.pythonhosted.org/packages/f2/21/52903825a0ae7bb625e8bd30a09816d4c3c5d6174a469c10a616166cf780/contourpy-1.4.0-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:f863c6100bf926cf47d13f3cd75f9bb8ebd98aaab230eeb4468b91f98f39a6b3", size = 1357306, upload-time = "2026-09-11T19:03:01.036Z" }, + { url = "https://files.pythonhosted.org/packages/d8/79/d68b1ce8539e4071518fed9523f558398f34dcd078b8927b109c72dad2ef/contourpy-1.4.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:3863ef2e2b13fe93f8c0ebb08ee400cb07153b8b0e91c5acb26e4537f283634f", size = 1427243, upload-time = "2026-09-11T19:03:03.866Z" }, + { url = "https://files.pythonhosted.org/packages/ac/f0/b75a10e9d0616b97a30de277b6dfe83f1879880854c7330337309530eb48/contourpy-1.4.0-cp312-cp312-win32.whl", hash = "sha256:5450f091ac1be0be3ad3a2a3b3f23b5e443e78c670ced4fd347d626f92a28fd2", size = 347345, upload-time = "2026-09-11T19:03:06.192Z" }, + { url = "https://files.pythonhosted.org/packages/50/a9/dab08786bb4d77ef9a046b3c1be923be4e73c5d07d91f5a54e8ea9b09e41/contourpy-1.4.0-cp312-cp312-win_amd64.whl", hash = "sha256:6e697d94e69f499ff6bebb899cae97a58d5d14f0e1fe9568b43a0248d2f9af8c", size = 234083, upload-time = "2026-09-11T19:03:08.122Z" }, + { url = "https://files.pythonhosted.org/packages/47/b9/3ba509755a970dbde5948e7142ac21f266be72f38017cc4087a05fdceee1/contourpy-1.4.0-cp312-cp312-win_arm64.whl", hash = "sha256:0c7a4c2716a4e98342221954416a836cca77c996c14ddbf22b5f01d5d93ca09c", size = 559521, upload-time = "2026-09-11T19:03:10.047Z" }, +] + [[package]] name = "curl-cffi" version = "0.16.3" @@ -186,6 +239,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/45/bb/67bec3132aeabac99dfe2f299a9b43dcb5de23ad96219ee98516d177fc9c/curl_cffi-0.16.3-cp310-abi3-win_arm64.whl", hash = "sha256:5a2ba880019f9e5a9e8f38ae22de6e4ea4c8d34a51ae4f1a2fce962c7b632006", size = 1713140, upload-time = "2026-09-02T11:58:01.558Z" }, ] +[[package]] +name = "cycler" +version = "0.12.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/a9/95/a3dbbb5028f35eafb79008e7522a75244477d2838f38cbb722248dabc2a8/cycler-0.12.1.tar.gz", hash = "sha256:88bb128f02ba341da8ef447245a9e138fae777f6a23943da4540077d3601eb1c", size = 7615, upload-time = "2023-10-07T05:32:18.335Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e7/05/c19819d5e3d95294a6f5947fb9b9629efb316b96de511b418c53d245aae6/cycler-0.12.1-py3-none-any.whl", hash = "sha256:85cef7cff222d8644161529808465972e51340599459b8ac3ccbac5a854e0d30", size = 8321, upload-time = "2023-10-07T05:32:16.783Z" }, +] + [[package]] name = "duckdb" version = "1.5.5" @@ -234,6 +296,23 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/cb/03/10388a42375ee7e4ac9b94eb2c5c569c8b5795e377e701c9ac3ad63de890/fastapi-0.141.1-py3-none-any.whl", hash = "sha256:bfb91aa2d334c61cb35ba9a116fc123b3d3df31640b801cf57a7a78ec3f603b3", size = 131954, upload-time = "2026-07-29T17:18:04.364Z" }, ] +[[package]] +name = "fonttools" +version = "4.66.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/a2/24/86f9930b930b97fc82266083320f3c34643ef67261c32651e893db525aca/fonttools-4.66.0.tar.gz", hash = "sha256:ef0610dfe7bb5bf574d9bdad6f597403ebc9807d124ac6f148d7604b2609be98", size = 3692007, upload-time = "2026-09-23T17:59:14.345Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e7/c1/25821e3bddd0644df229cfdc2e8b4101d28e4da15ffbe6e6b563ea6f3acd/fonttools-4.66.0-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:67ea5af3ca60e1e5c9b2841b1f6dba5ef16aab581bf463abb68515447154ab61", size = 3096788, upload-time = "2026-09-23T17:57:08.302Z" }, + { url = "https://files.pythonhosted.org/packages/0c/3d/d16b4cb63710abd91f1aaa482f2848d8b0c402502de5b8a72ba9d2613cb0/fonttools-4.66.0-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:dbee639a23c4e067aedfbcf84d2e466a71a1a1156ad04d6d09023791ed167495", size = 2586926, upload-time = "2026-09-23T17:57:10.678Z" }, + { url = "https://files.pythonhosted.org/packages/e2/f0/dffa4fa83ec6780464e8484fa2f30ccc252d480137c8e4d3fc6ac510b3eb/fonttools-4.66.0-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:35684b562df7154d7a0ebfa512db9199c7fb0a93b5abfb59dac9d022dbee7aaa", size = 5415821, upload-time = "2026-09-23T17:57:13.174Z" }, + { url = "https://files.pythonhosted.org/packages/6d/56/3c862bdd297a272a37e2ba1149da01446fd3f387ee80855699bb939bf5b2/fonttools-4.66.0-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:aea0cc5609a5f2a2500f91bd6e1989fdd22da0f225be00dc1c4cec775838a7d0", size = 5396689, upload-time = "2026-09-23T17:57:15.987Z" }, + { url = "https://files.pythonhosted.org/packages/93/d7/f4b7c81a7a6b316b7226d11f8e8659f023babf1e0957f8e7ddcc4e155a82/fonttools-4.66.0-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:8af85dfc2ad564f95b3e650024578bb2268bc4d827d418aab9650bbde4b6926a", size = 5353218, upload-time = "2026-09-23T17:57:18.803Z" }, + { url = "https://files.pythonhosted.org/packages/70/1e/f12bc535f3da359a4d29d5e7442ab2d7d67579d1516b9c7e1a38890168bd/fonttools-4.66.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:2b278e2abe596872b3c9fe611f956ded85588537974e48e2adbe2d6ee159e099", size = 5516965, upload-time = "2026-09-23T17:57:21.077Z" }, + { url = "https://files.pythonhosted.org/packages/14/25/a1df66888c8b5f63c14e089400f2353da0c1ff9d09683cc0bf85afea348e/fonttools-4.66.0-cp312-cp312-win32.whl", hash = "sha256:ee7b9f6c835ea1a9beb5a35c8eb0c62b6a3290f8105d3d0c29a0a18c90f9b8ba", size = 2434425, upload-time = "2026-09-23T17:57:23.463Z" }, + { url = "https://files.pythonhosted.org/packages/b5/f2/fd490f1ddf2e430289c13059982771e6b02a2bf37504401d2a5c53307892/fonttools-4.66.0-cp312-cp312-win_amd64.whl", hash = "sha256:1d120ea0f5260b04e9b5ac0d9239efde4c23d990347d5d5889cdd37221726fec", size = 2486335, upload-time = "2026-09-23T17:57:25.561Z" }, + { url = "https://files.pythonhosted.org/packages/83/65/826290863c9df6041f2e36a5ae5d604cd8247fffc8ec7b45581fd2473e4d/fonttools-4.66.0-py3-none-any.whl", hash = "sha256:bc7b7ddc1a1f46c363354304e9a8dd93722e4a6a24f785015650898dacf40df9", size = 1201583, upload-time = "2026-09-23T17:59:12.104Z" }, +] + [[package]] name = "formulaic" version = "1.2.2" @@ -406,6 +485,44 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/a1/d3/20a61a248feb1249cd81f83ce731f0bdf5170ed96a08a298f7081cda9e90/interface_meta-2.0.1-py3-none-any.whl", hash = "sha256:f38016bef9a4429b6d0792d809be7b65e9781820c674bf7f463999086b6e6323", size = 15330, upload-time = "2026-04-30T22:35:35.201Z" }, ] +[[package]] +name = "joblib" +version = "1.6.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "cloudpickle" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/d5/1d/537ab090f302b838943a1b56497dd53059b9a9b46a074936470173a2e207/joblib-1.6.0.tar.gz", hash = "sha256:2ccc96785b12046c08fd6d55839c12857831b54a3c1673ffadd2f04bfc4eda03", size = 327903, upload-time = "2026-08-31T09:39:04.122Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/18/53/84099323c2ec4be98d935f63c033ac4151ee83836ca1050ede3b3aadf155/joblib-1.6.0-py3-none-any.whl", hash = "sha256:3dbbf9f6e4b592a2357b854608e980fe6390d131d7a82f011a377ef2ebef7aba", size = 306115, upload-time = "2026-08-31T09:39:02.298Z" }, +] + +[[package]] +name = "kiwisolver" +version = "1.5.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/ba/07/bd78e6a8fae171ea041ef5bba3ed21a003522fa088834b069b1909981f30/kiwisolver-1.5.1.tar.gz", hash = "sha256:f1303ef2eec81262a4b708c3e858afe58d7c75ad91c1c05266eda7673369859a", size = 104395, upload-time = "2026-08-28T10:28:27.153Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/6b/9b/65b302742389c6f96f2956bef5decf26011309feb2fc5d79613af18adea4/kiwisolver-1.5.1-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:63fb7294b768f444eb4b068965f2662f28c2fd4161e23bd60fcf3ff27b74c046", size = 123876, upload-time = "2026-08-28T10:25:27.44Z" }, + { url = "https://files.pythonhosted.org/packages/71/74/c21f339956f6f691b2ed7e31d5f3ae767304df6c460192739fc830853051/kiwisolver-1.5.1-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:0ebdef3eae5336568147c39a55be6a2036ffde53faa9ca2d978989ae7c2da12c", size = 66487, upload-time = "2026-08-28T10:25:28.728Z" }, + { url = "https://files.pythonhosted.org/packages/84/e5/bdb34e21523e01dceda064d63713f3bdec91388af24fba1eca7ea5e85864/kiwisolver-1.5.1-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:1798e83840c3f627246104c4d8a9639c60fa068adf9ce92b61791781fa8a68c1", size = 64660, upload-time = "2026-08-28T10:25:30.071Z" }, + { url = "https://files.pythonhosted.org/packages/fc/f4/dadfec469313c7f428efa7e84b4aba9732f813c13ea7131a24b7b008ef57/kiwisolver-1.5.1-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:34633ecf50d16187ab8e5528b7a2530f2feb4e23f300db4672538b51cfc5cd38", size = 1477929, upload-time = "2026-08-28T10:25:31.495Z" }, + { url = "https://files.pythonhosted.org/packages/6f/35/09c58daac34e6f6ea5c6dee0094b422118e5a7c265586008a95fd135ac5f/kiwisolver-1.5.1-cp312-cp312-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d27c2123977cb9269c30a49ba45f03a4323017ef693e19db4ec9dbe1299a3002", size = 1278499, upload-time = "2026-08-28T10:25:33.375Z" }, + { url = "https://files.pythonhosted.org/packages/19/32/739765e24fbad29d13f83e546ea4abc215a78cea9d677ca09025b027724d/kiwisolver-1.5.1-cp312-cp312-manylinux_2_24_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:6a797a1cefc8b9c93170db580337e1fe3d011ad18b1299943231279406342048", size = 1296677, upload-time = "2026-08-28T10:25:35.059Z" }, + { url = "https://files.pythonhosted.org/packages/df/32/03304d1010e2cc45e5b3b52cef7e43fed3a2a5cd6c87a89b4a88e1d85b5d/kiwisolver-1.5.1-cp312-cp312-manylinux_2_24_s390x.manylinux_2_28_s390x.whl", hash = "sha256:2551cf9917af48ee7c4b29cc82320489508cf96fd26a51f6fc124de661cd44c7", size = 1346037, upload-time = "2026-08-28T10:25:36.705Z" }, + { url = "https://files.pythonhosted.org/packages/3e/57/4c49377bfd274450dd72ecaa13eaac32ea804a03363e4d1db0c5aa999ceb/kiwisolver-1.5.1-cp312-cp312-manylinux_2_39_riscv64.whl", hash = "sha256:38f6e0deb4d0a4615efe0c4efc5990b06ae450ab50a0b321c0b078b6d238c083", size = 988248, upload-time = "2026-08-28T10:25:38.299Z" }, + { url = "https://files.pythonhosted.org/packages/cb/c3/38df144a08b6c5d75ca4504e5cc3141bb3bfef64c04f4ef48204f42711b6/kiwisolver-1.5.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:bfd1de989b3330420e29de39352f5c049905c9e3ee67233a50d550e3d652c148", size = 2228722, upload-time = "2026-08-28T10:25:40.038Z" }, + { url = "https://files.pythonhosted.org/packages/e7/11/3221838a89cd64d9b386353e000cd8a296069a20fbe3584507fdfd5bebae/kiwisolver-1.5.1-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:1209042a623ddfda5497e4066c7b77651dde8e1d3a9dd97599dc7e97f3b9b78c", size = 2325216, upload-time = "2026-08-28T10:25:41.699Z" }, + { url = "https://files.pythonhosted.org/packages/83/d4/075c219230697bb5db910d37262b9bacf880f92b4811a02ab81ed073a253/kiwisolver-1.5.1-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:26e8268480be5061d509e29669d59103c067a26377a56491630ece11762e3858", size = 1977689, upload-time = "2026-08-28T10:25:43.559Z" }, + { url = "https://files.pythonhosted.org/packages/bb/08/1d219c3c2dd960983d0d4da623d916e9de6385df2b0bab3d1af0e9b8fccc/kiwisolver-1.5.1-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:d79308fa689fac89cbcfbd4dbfc80b5f95c54c5a7fd4d194be221f9d33d026e6", size = 2491443, upload-time = "2026-08-28T10:25:45.242Z" }, + { url = "https://files.pythonhosted.org/packages/ba/d3/024208ec1079d273f1047468d1bdffbf38bb75b7b268090fd3a0301b9d9a/kiwisolver-1.5.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:b03af77d77e50edba2030fd5f7c352ff209314b09030a3cba7c14edf9a09a444", size = 2295200, upload-time = "2026-08-28T10:25:46.984Z" }, + { url = "https://files.pythonhosted.org/packages/6e/7c/7b210498f9f92e1cd7855f260fa69ef056881087b199ee20c208f0e4189a/kiwisolver-1.5.1-cp312-cp312-win_amd64.whl", hash = "sha256:06a6917674de9e0fe3f66f5430787f59a9f2ddb64af9b714eaec547e29ef5c19", size = 70748, upload-time = "2026-08-28T10:25:48.444Z" }, + { url = "https://files.pythonhosted.org/packages/94/61/ef0daa157c8bb23672f7423e0d14c39db1dc6ef8ed47e6bc54c9c1bef3bf/kiwisolver-1.5.1-cp312-cp312-win_arm64.whl", hash = "sha256:ad8b9671348d7c8716715652ae11f85ed0eb99e265a2df2ca490577d69860b2c", size = 68324, upload-time = "2026-08-28T10:25:49.81Z" }, + { url = "https://files.pythonhosted.org/packages/a9/c4/1407df7512a5b36cc79840e01710dc575733c461b13ab866cae77eaf87f3/kiwisolver-1.5.1-graalpy312-graalpy250_312_native-macosx_11_0_arm64.whl", hash = "sha256:482676e5bd48d70ac99d9fc78863469845421e01184fa83f1f9366dc49f7e974", size = 134002, upload-time = "2026-08-28T10:28:08.881Z" }, + { url = "https://files.pythonhosted.org/packages/16/45/c37a21ad5c0ab581a93c55ad544721aaa1f0ae94edb29c6a678a23d013e6/kiwisolver-1.5.1-graalpy312-graalpy250_312_native-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:072bdb15a3c19a5b5dbc8f8fb1f4e1884bf4f3507eeb4cc6334401274d37a5c0", size = 194292, upload-time = "2026-08-28T10:28:11.06Z" }, + { url = "https://files.pythonhosted.org/packages/a6/c1/69f00d627949580e43d57af0aa465df46868d7c29801c137a55374101294/kiwisolver-1.5.1-graalpy312-graalpy250_312_native-win_amd64.whl", hash = "sha256:a5a00665d1a0e26763a7338d7e911d4598fbc1d50dd0d6b7919b7dc6c5d6569f", size = 73362, upload-time = "2026-08-28T10:28:12.449Z" }, +] + [[package]] name = "korean-lunar-calendar" version = "0.4.0" @@ -415,6 +532,24 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/c1/88/f56033b09fdca5c8c5e5fc2f3513950a05cd62caa9bd8e6d2750a887c86e/korean_lunar_calendar-0.4.0-py3-none-any.whl", hash = "sha256:c042e20de0bb702add6bec8d0f6da1ea8d3b170838e63846f70420cf341fe4e7", size = 11323, upload-time = "2026-06-15T14:19:56.089Z" }, ] +[[package]] +name = "lifelines" +version = "0.30.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "autograd" }, + { name = "autograd-gamma" }, + { name = "formulaic" }, + { name = "matplotlib" }, + { name = "numpy" }, + { name = "pandas" }, + { name = "scipy" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/ea/4f/f0b363278d40baf7d7a03217bee839cb880946c62109f243391c8754bb09/lifelines-0.30.0.tar.gz", hash = "sha256:f7f6f6275fcb167fe0f5b1ef98f868993f9c074cb74b1dd6e92736efa854be18", size = 383221, upload-time = "2024-10-29T12:00:43.513Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/14/f7/379e185a75ac8166ac70756d0ba68d9a2b02b555c7fde4983246752396bd/lifelines-0.30.0-py3-none-any.whl", hash = "sha256:ac7c602c8aceced9770d3977817c9d99c250ed8cd86f2567fa0d23e4e8014bf9", size = 349319, upload-time = "2024-10-29T12:00:41.749Z" }, +] + [[package]] name = "lxml" version = "6.1.3" @@ -441,6 +576,32 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/e4/1b/7bcebb7b6332cb3ae85e9c13b139adb6f23f75c71d84041c56a5005d9a29/lxml-6.1.3-cp312-cp312-win_arm64.whl", hash = "sha256:1aeca87830c4fe649dcf93fe2b059525b71c72587f21be4ae4af7103082a79fa", size = 3666631, upload-time = "2026-09-02T14:48:14.567Z" }, ] +[[package]] +name = "matplotlib" +version = "3.11.2" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "contourpy" }, + { name = "cycler" }, + { name = "fonttools" }, + { name = "kiwisolver" }, + { name = "numpy" }, + { name = "packaging" }, + { name = "pillow" }, + { name = "pyparsing" }, + { name = "python-dateutil" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/e7/c8/9aa712a0afb882649424dd8de8ad9aa6235e796e84c6052e8f6dc1598d0d/matplotlib-3.11.2.tar.gz", hash = "sha256:cec596316640f2b394b8f0daa0ea61a8eae82d017b620b9f202befb972a59ea4", size = 32660610, upload-time = "2026-09-11T19:05:31.214Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/cb/ce/1bfcc4873b121597791ad74032943b123218c0613af0b97e8dd05e916fb1/matplotlib-3.11.2-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:ef752769cd962f39ea0b6ffc82d1ea43a0012c5a6157c7a075212fa509cfcff2", size = 9476466, upload-time = "2026-09-11T19:03:19.45Z" }, + { url = "https://files.pythonhosted.org/packages/a6/c4/7f5f3601ee69baf072c0c7d3ce60c03e0618621c5a56c34a62a460e29d11/matplotlib-3.11.2-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:ef31985c4dedb5f1424e1aec6849a47dd37689cb7fa3c20b1b82187f26806261", size = 9305583, upload-time = "2026-09-11T19:03:22.39Z" }, + { url = "https://files.pythonhosted.org/packages/b8/90/2b3fd67ee273163faeda6d514be70b5596eaa0fd77b60ffc294ad0b34f5f/matplotlib-3.11.2-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:df4f7784aca81a94f254c0a2767d592ee25f407e488f5fa7203e51093fb6ca27", size = 9860353, upload-time = "2026-09-11T19:03:25.049Z" }, + { url = "https://files.pythonhosted.org/packages/f4/84/32549e7a462dc311aed2ab62e5d2538028840b5c53a8be0195c789937c3d/matplotlib-3.11.2-cp312-cp312-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:1b9a7ad579856284135e401ecc918c5f8a017ee30539298862a109f51b971710", size = 10670348, upload-time = "2026-09-11T19:03:28.108Z" }, + { url = "https://files.pythonhosted.org/packages/84/39/02e21b74f7439bd643d717ea846006d46e309e6d29b632b57751265311d6/matplotlib-3.11.2-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:3aa4b8516fd26659e4363abbf317c703d9116496c5db2e9d0609a2866dd39dd2", size = 10810580, upload-time = "2026-09-11T19:03:31.402Z" }, + { url = "https://files.pythonhosted.org/packages/04/93/0d239bde12b308262265c2d98909b2f0ecad2b5deeae96241642e822e9c7/matplotlib-3.11.2-cp312-cp312-win_amd64.whl", hash = "sha256:c5c1c68ee401fc98271263410f0e5ce88285abacf7627132914e8adf3d70ff43", size = 9349409, upload-time = "2026-09-11T19:03:34.195Z" }, + { url = "https://files.pythonhosted.org/packages/49/a8/06baf901c02246c8a222b655cc4540ef9c15e1549a04e07927f9cde716c3/matplotlib-3.11.2-cp312-cp312-win_arm64.whl", hash = "sha256:643ff850d8e0f5b8319337f87ed3cb59506afb3df3cc48de777d85871233be7b", size = 9028084, upload-time = "2026-09-11T19:03:37.032Z" }, +] + [[package]] name = "mpmath" version = "1.3.0" @@ -468,6 +629,24 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/40/b5/1b84b2c784db76d69442334bc8b8748c840f13ca53be086f4f250ad4a0bc/narwhals-2.26.0-py3-none-any.whl", hash = "sha256:29326d74f107c347fd1009bd58e38d9f7c7c5b51e6de97bc93dbc325d9038b54", size = 474034, upload-time = "2026-09-08T13:32:07.159Z" }, ] +[[package]] +name = "ngboost" +version = "0.5.11" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "lifelines" }, + { name = "matplotlib" }, + { name = "numpy" }, + { name = "scikit-learn" }, + { name = "scipy" }, + { name = "sympy" }, + { name = "tqdm" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/68/66/119fe55e3b0ab091101778b9b7d228ca27b75e5da88c0424f93594a2f236/ngboost-0.5.11.tar.gz", hash = "sha256:873326442f00632829a521209b1030e150e0ef0a9de09952e044f8f3b8839940", size = 42369, upload-time = "2026-06-26T03:57:03.355Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/52/f6/c3ede5cc357b2c1c70d626c68ece0c352bac5d5314652b4ad35215ed6bc8/ngboost-0.5.11-py3-none-any.whl", hash = "sha256:c3683334ab6ad58d79bc50aa11d8f090f4d452010c73dc2f2304affc448b08fa", size = 53051, upload-time = "2026-06-26T03:57:02.271Z" }, +] + [[package]] name = "numpy" version = "2.5.3" @@ -504,6 +683,7 @@ dependencies = [ { name = "pydantic" }, { name = "regimelib" }, { name = "scipy" }, + { name = "statsmodels" }, { name = "uvicorn", extra = ["standard"] }, { name = "yfinance" }, ] @@ -515,7 +695,10 @@ dev = [ { name = "pytest-asyncio" }, { name = "ruff" }, ] -research = [] +research = [ + { name = "ngboost" }, + { name = "scikit-learn" }, +] [package.metadata] requires-dist = [ @@ -531,6 +714,7 @@ requires-dist = [ { name = "pydantic", specifier = ">=2.13" }, { name = "regimelib", specifier = "==0.1.0" }, { name = "scipy", specifier = ">=1.18,<2" }, + { name = "statsmodels", specifier = "==0.15.0" }, { name = "uvicorn", extras = ["standard"], specifier = ">=0.32" }, { name = "yfinance", specifier = ">=1.7" }, ] @@ -542,7 +726,10 @@ dev = [ { name = "pytest-asyncio", specifier = ">=0.25" }, { name = "ruff", specifier = ">=0.15" }, ] -research = [] +research = [ + { name = "ngboost", specifier = "==0.5.11" }, + { name = "scikit-learn", specifier = "==1.6.1" }, +] [[package]] name = "packaging" @@ -597,6 +784,23 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/13/90/3ddf6ef69090c63b6fe0d42b41e5dd3323d95dfe01eda6766305337be862/peewee-4.5.1-py3-none-any.whl", hash = "sha256:dbbdc93e9be08d1df49ceed48dc5d609bfeda6112848086f1f6c1f1debe70975", size = 193883, upload-time = "2026-09-08T01:20:09.027Z" }, ] +[[package]] +name = "pillow" +version = "12.3.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/1c/3d/bb7fca845737cf9d7dbde16ed1843984665ff2e0a518f5db43e77ec540b9/pillow-12.3.0.tar.gz", hash = "sha256:3b8182a766685eaa002637e28b4ec8d6b18819a0c71f579bf0dbaa5830297cce", size = 47025035, upload-time = "2026-07-01T11:56:38.965Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/37/bf/fb3ebff8ddcb76aac5a01389251bbbb9519922a9b520d8247c1ca864a25d/pillow-12.3.0-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:ba09209fbe443b4acccebe845d8a138b89a8f4fbaeedd44953490b5315d5e965", size = 5345969, upload-time = "2026-07-01T11:54:06.397Z" }, + { url = "https://files.pythonhosted.org/packages/d8/66/9a386a92561f402389a4fc70c18838bf6d35eb5eb5c6850b4b2dc64f5048/pillow-12.3.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:ffd0c5368496f41b0944be820fcb7a838aa6e623d250b01acf2643939c3f99d7", size = 4780323, upload-time = "2026-07-01T11:54:09.351Z" }, + { url = "https://files.pythonhosted.org/packages/25/27/ac8f99618ffd3dde21db0f4d4b1d2ab00c0880595bfd17df103f7f39fd0c/pillow-12.3.0-cp312-cp312-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d9c7f76c0673154f044e9d78c8655fb4213f6ca31a836df48b40fe5d187717b9", size = 6266838, upload-time = "2026-07-01T11:54:11.71Z" }, + { url = "https://files.pythonhosted.org/packages/84/21/a35af28dcc61f37ed850a2d64c65c701321dfbf25085e469d5559360cbbf/pillow-12.3.0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:78cb2c6865a35ab8ff8b75fd122f6033b92a62c82801110e48ddd6c936a45d91", size = 6940830, upload-time = "2026-07-01T11:54:13.732Z" }, + { url = "https://files.pythonhosted.org/packages/eb/51/8b08617af3ad95e33ce6d7dd2c99ed6c8298f7fb131636303956be022e25/pillow-12.3.0-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:e491916b378fba47242221bb9ead245211b70d504f495d105d17b14a24b4907c", size = 6344383, upload-time = "2026-07-01T11:54:15.756Z" }, + { url = "https://files.pythonhosted.org/packages/1d/72/cf78ac9780bb93c28328f408973845a309d4d145041665f734572ced1b52/pillow-12.3.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:0dd2064cbc55aaec028ef5fbb60fa47bb6c3e7918e07ff17935284b227a9d2df", size = 7052934, upload-time = "2026-07-01T11:54:17.721Z" }, + { url = "https://files.pythonhosted.org/packages/20/20/25e0f4dc178a6bc0696793720055519a0de89e7661dae886992decbd2f81/pillow-12.3.0-cp312-cp312-win32.whl", hash = "sha256:dbce0b29841537a2fa4a214c2bbf14de3587c9680caa9b4e217568472490b28f", size = 6472684, upload-time = "2026-07-01T11:54:19.839Z" }, + { url = "https://files.pythonhosted.org/packages/45/89/da2f7971a317f83d807fdd4065c0af40208e59e692cc43d315a71a0e96d1/pillow-12.3.0-cp312-cp312-win_amd64.whl", hash = "sha256:a2b55dd6b2a4c4b7d87ffa56bdb33fdc5fdb9a462173861a7bc097f17d91cb09", size = 7227137, upload-time = "2026-07-01T11:54:22.025Z" }, + { url = "https://files.pythonhosted.org/packages/de/47/4845a0a6c0dbf1db8456bd9fc791f13c5ced7ced20606d08a0aacfd25b49/pillow-12.3.0-cp312-cp312-win_arm64.whl", hash = "sha256:331b624368d4f1d069149002f25f44bc61c8919ce8ddb3c45bdad8f6e2d89510", size = 2568267, upload-time = "2026-07-01T11:54:24.051Z" }, +] + [[package]] name = "platformdirs" version = "4.11.14" @@ -745,6 +949,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/8a/c8/f96208ade3ca4c23b372497d0788bcf0f2e0ff4310e5ee693366bc33fdf0/pyluach-2.3.0-py3-none-any.whl", hash = "sha256:4497b731aef59508b079dbf5f00bc5bf4329ac45090a6cd37b5a83756f0e69ab", size = 25914, upload-time = "2025-09-09T20:24:37.831Z" }, ] +[[package]] +name = "pyparsing" +version = "3.3.3" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/e4/11/b213bebff182584360cb8d17c72c1677fec5c5c228de439e63bcf8ab1c8f/pyparsing-3.3.3.tar.gz", hash = "sha256:928ae7e20211f3b6f3915a72f06a0cfd29ab9d24279dd6346b6b1a7146397d36", size = 1050487, upload-time = "2026-09-20T20:59:05.609Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/38/bb/d215ee7c73b61497b28a5503f9f53523f294fcc936762b7caf90e0c1c2b5/pyparsing-3.3.3-py3-none-any.whl", hash = "sha256:ece8c00a69cf01b45d0b1dedabb469c90d8caf996d4fda40f147627a122849a4", size = 126420, upload-time = "2026-09-20T20:59:04.025Z" }, +] + [[package]] name = "pytest" version = "9.1.1" @@ -877,6 +1090,25 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/51/60/5fb1a39dbb5ae314d5f59bc7348a63c1d5c20f3cd83914c4b5cb0be31d2d/ruff-0.16.9-py3-none-win_arm64.whl", hash = "sha256:ed1a252039200f57a59eebc063b54beabea67bfbaaca0eeaa7f54b5fbcda2284", size = 10458649, upload-time = "2026-09-24T20:37:46.882Z" }, ] +[[package]] +name = "scikit-learn" +version = "1.6.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "joblib" }, + { name = "numpy" }, + { name = "scipy" }, + { name = "threadpoolctl" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/9e/a5/4ae3b3a0755f7b35a280ac90b28817d1f380318973cff14075ab41ef50d9/scikit_learn-1.6.1.tar.gz", hash = "sha256:b4fc2525eca2c69a59260f583c56a7557c6ccdf8deafdba6e060f94c1c59738e", size = 7068312, upload-time = "2025-01-10T08:07:55.348Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/0a/18/c797c9b8c10380d05616db3bfb48e2a3358c767affd0857d56c2eb501caa/scikit_learn-1.6.1-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:926f207c804104677af4857b2c609940b743d04c4c35ce0ddc8ff4f053cddc1b", size = 12104516, upload-time = "2025-01-10T08:06:40.009Z" }, + { url = "https://files.pythonhosted.org/packages/c4/b7/2e35f8e289ab70108f8cbb2e7a2208f0575dc704749721286519dcf35f6f/scikit_learn-1.6.1-cp312-cp312-macosx_12_0_arm64.whl", hash = "sha256:2c2cae262064e6a9b77eee1c8e768fc46aa0b8338c6a8297b9b6759720ec0ff2", size = 11167837, upload-time = "2025-01-10T08:06:43.305Z" }, + { url = "https://files.pythonhosted.org/packages/a4/f6/ff7beaeb644bcad72bcfd5a03ff36d32ee4e53a8b29a639f11bcb65d06cd/scikit_learn-1.6.1-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:1061b7c028a8663fb9a1a1baf9317b64a257fcb036dae5c8752b2abef31d136f", size = 12253728, upload-time = "2025-01-10T08:06:47.618Z" }, + { url = "https://files.pythonhosted.org/packages/29/7a/8bce8968883e9465de20be15542f4c7e221952441727c4dad24d534c6d99/scikit_learn-1.6.1-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:2e69fab4ebfc9c9b580a7a80111b43d214ab06250f8a7ef590a4edf72464dd86", size = 13147700, upload-time = "2025-01-10T08:06:50.888Z" }, + { url = "https://files.pythonhosted.org/packages/62/27/585859e72e117fe861c2079bcba35591a84f801e21bc1ab85bce6ce60305/scikit_learn-1.6.1-cp312-cp312-win_amd64.whl", hash = "sha256:70b1d7e85b1c96383f872a519b3375f92f14731e279a7b4c6cfd650cf5dffc52", size = 11110613, upload-time = "2025-01-10T08:06:54.115Z" }, +] + [[package]] name = "scipy" version = "1.18.1" @@ -963,6 +1195,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/a2/09/77d55d46fd61b4a135c444fc97158ef34a095e5681d0a6c10b75bf356191/sympy-1.14.0-py3-none-any.whl", hash = "sha256:e091cc3e99d2141a0ba2847328f5479b05d94a6635cb96148ccb3f34671bd8f5", size = 6299353, upload-time = "2025-04-27T18:04:59.103Z" }, ] +[[package]] +name = "threadpoolctl" +version = "3.7.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/00/dc/6c58154c1c65f758ea979e7139cb76993a9cfc662d14e9be3c4a667cfb77/threadpoolctl-3.7.0.tar.gz", hash = "sha256:61348cfb77d53b9242e0017029244b559b810c142ced65b4e21eeca1843959a7", size = 31961, upload-time = "2026-09-15T15:46:20.263Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/43/3f/f88a53f60a472b46f4023f56d204dd7de33d34c5d2acbfa0d70a674e639e/threadpoolctl-3.7.0-py3-none-any.whl", hash = "sha256:cd8b60b5641b45c67bbf73c64c843235fc2d8a480c87389f52f5dbee893b86be", size = 26362, upload-time = "2026-09-15T15:46:19.168Z" }, +] + [[package]] name = "toolz" version = "1.1.0" @@ -972,6 +1213,18 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/fb/12/5911ae3eeec47800503a238d971e51722ccea5feb8569b735184d5fcdbc0/toolz-1.1.0-py3-none-any.whl", hash = "sha256:15ccc861ac51c53696de0a5d6d4607f99c210739caf987b5d2054f3efed429d8", size = 58093, upload-time = "2025-10-17T04:03:20.435Z" }, ] +[[package]] +name = "tqdm" +version = "4.70.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "colorama", marker = "sys_platform == 'win32'" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/0d/ea/b2a5bd54b28a324dae8211928b2d730b6547500342c7e6c6dea08bd0a485/tqdm-4.70.1.tar.gz", hash = "sha256:cefd0eca11b2a37a3aee776544d4f4ae913f02688135b5556b8788dfa474afc4", size = 171846, upload-time = "2026-09-11T07:25:16.601Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a7/03/921a3d3c75785aca9ebfbfcabfbc3a1be12e2ab5265deb026d55a5a3f83e/tqdm-4.70.1-py3-none-any.whl", hash = "sha256:c293e525e6fef9c20e8728fd4612df02a0aa31bb5fe91ecd93e123b1b7bffa73", size = 80199, upload-time = "2026-09-11T07:25:14.599Z" }, +] + [[package]] name = "truststore" version = "0.10.4" diff --git a/docs/source-rights.md b/docs/source-rights.md new file mode 100644 index 0000000..9824121 --- /dev/null +++ b/docs/source-rights.md @@ -0,0 +1,27 @@ +# Forecast input rights and provenance + +Checked 2026-09-28. This is an engineering gate, not a license determination. + +| Source | What the public source provides | Decision for this release | +| --- | --- | --- | +| [Nasdaq website terms](https://www.nasdaq.com/legal) | Section 7 requires express written permission to capture or extract site content for machine learning or data-analysis software. | Do not create a Nasdaq quote archive or train models on captured site quotes. Existing live display behavior is outside this expansion. | +| [Yahoo terms](https://legal.yahoo.com/ca/en/yahoo/terms/otos/) | Automated collection requires prior permission; a competing database or archive is restricted. [Yahoo Finance help](https://in.help.yahoo.com/kb/SLN2311.html) also says download rights vary by instrument. | Do not expand automated history downloads for model training. The new vintage helper only freezes already-local verified caches; it never fetches. Before a production-scale backfill or redistribution, obtain a suitable data license or substitute a rights-cleared source. | +| [Cboe historical archive](https://www.cboe.com/us/options/market_statistics/historical_data/) | Free historical option **volume**; bid/ask history is a separate [DataShop product](https://datashop.cboe.com/option-eod-summary). | Cannot support IV-informed training or SSVI quote backtesting. No new option-quote capture. | +| [SEC EDGAR APIs](https://www.sec.gov/search-filings/edgar-application-programming-interfaces) | Public filing history and documents, not a normalized forward earnings calendar. [SEC fair-access guidance](https://www.sec.gov/about/webmaster-frequently-asked-questions) allows declared scripted access at no more than 10 requests/s. | Capture only explicit *future* earnings schedule announcements in 8-K exhibits. Preserve accession, acceptance, retrieval and event times, source hash and revisions. An actual earnings filing alone never establishes prior knowledge of its date. | + +The `forecast/training/cohort.json` artifact is a hash-verified convenience sample frozen from local prices. It is separate from the 50-ticker audit cohort and carries `immutable_current_vintage_training` provenance. Replaying past dates from it is retrospective current-vintage screening, never an as-issued historical forecast. A missing or changed member fails closed. The helper is deliberately offline; its presence does not grant rights to fetch or retain new vendor data. + +Yahoo-derived cohort manifests carry `training_rights_unverified`. The normal cohort loader refuses to supply frames to pooled-model training while that status remains. An explicit inspection-only read can verify integrity; it does not qualify the data for model training. A separately provisioned, rights-cleared cohort can use `approved_for_training` only when its cohort and every immutable vintage carry the same explicit source name and license reference, each bound to the price data by SHA-256 manifest hashes. This metadata records an operator's source qualification; it is not itself a license grant. + +Live pooled forecasts also remain unavailable for a qualified artifact until issuance and evidence records include that artifact's SHA-256. Otherwise a later retraining run could be scored as if it were the same model input. + +For deliberate SEC capture, identify the issuer's CIK from SEC's [ticker/CIK mapping](https://www.sec.gov/files/company_tickers.json), then run from `backend/`: + +```bash +HYPEROPTIONS_SEC_USER_AGENT='HyperOptions YourContact@example.com' \ + uv run --frozen python scripts/capture_sec_events.py AAPL 320193 +``` + +The command inspects at most five recent 8-Ks for that exact ticker/CIK and, for each filing, at most one unambiguous HTML/text EX-99.1 attachment. It does not run at app startup, scrape a calendar, or infer an event from a filing date. Schedule it only with a real declared SEC contact and a deliberate ticker list; repeated runs are idempotent. An earnings-jump model requires an explicit prior future schedule **and** a separately dated actual results release matching the exact event date. A schedule alone is insufficient. Even both dated filings do not establish the actual release **clock time**; a numeric jump model remains unavailable until that time and enough verified, split-safe event-return outcomes support the overnight/intraday alignment. + +Expected intraday capture windows are recorded independently of successful forecast issuances, so missed windows remain in the denominator. Already-recorded expectations continue to resolve after watch deletion. A watch deleted while the app was offline, before its expectation was recorded, cannot be reconstructed from the current watch table; such gaps must be reported separately rather than counted as successful coverage. Read-only capture counts are limited to a 35-day interval and separate from matched forecast-score coverage. diff --git a/frontend/e2e/watchlist.spec.ts b/frontend/e2e/watchlist.spec.ts index dd80896..baeb563 100644 --- a/frontend/e2e/watchlist.spec.ts +++ b/frontend/e2e/watchlist.spec.ts @@ -122,12 +122,22 @@ test("compares all models in a bounded dialog on desktop and mobile", async ({ p ...item, physical_models: [ { method: "lognormal_ewma", status: "available", itm_pct_tenths: 612, otm_pct_tenths: 388, atm_pct_tenths: 0, price_basis: "completed_close", support: 60 }, + { method: "empirical_scaled", status: "available", itm_pct_tenths: 630, otm_pct_tenths: 370, atm_pct_tenths: 0, price_basis: "completed_close", support: 210 }, { method: "student_t_ewma", status: "pending", reason: "candidate_not_prepared" }, { method: "gjr_garch_t", status: "unavailable", reason: "gjr_parameters_invalid" }, + { method: "ohlc_har", status: "available", itm_pct_tenths: 590, otm_pct_tenths: 410, atm_pct_tenths: 0, price_basis: "completed_close", support: 4096 }, + { method: "skew_t_ewma", status: "pending", reason: "candidate_not_prepared" }, + { method: "egarch_skew_t", status: "unavailable", reason: "egarch_parameters_invalid" }, + { method: "markov_switching", status: "pending", reason: "candidate_not_prepared" }, + { method: "ngboost_pooled", status: "pending", reason: "candidate_not_prepared" }, + { method: "earnings_jump", status: "unavailable", reason: "verified_release_time_history_unavailable" }, + { method: "iv_physical", status: "unavailable", reason: "rights_cleared_option_history_unavailable" }, + { method: "intraday_shadow", status: "unavailable", reason: "underlying_quote_unavailable" }, ], market_models: [ { method: "regimelib", status: "available", itm_pct_tenths: 583, otm_pct_tenths: 417, bound_low_pct_tenths: 452, bound_high_pct_tenths: 753 }, { method: "constrained_call_curve", status: "unavailable", reason: "clean_strikes_do_not_bracket_contract" }, + { method: "ssvi", status: "unavailable", reason: "rights_cleared_option_history_unavailable" }, ], } await page.route("**/api/watchlist**", (route) => route.fulfill({ json: { items: [comparisonItem] } })) @@ -137,9 +147,11 @@ test("compares all models in a bounded dialog on desktop and mobile", async ({ p await page.getByRole("button", { name: "Compare models" }).click() const dialog = page.getByRole("dialog", { name: "Compare models" }) await expect(dialog).toBeVisible() - await expect(dialog.locator(".model-result")).toHaveCount(5) + await expect(dialog.locator(".model-result")).toHaveCount(15) await expect(dialog.locator(".model-result details[open]")).toHaveCount(0) await expect(dialog).toContainText("gjr parameters invalid") + await expect(dialog.getByText("Daily OHLC range/HAR proxy")).toBeVisible() + await expect(dialog.getByText("SSVI volatility surface")).toBeVisible() const bounds = await dialog.evaluate((element) => { const rect = element.getBoundingClientRect() return { width: rect.width, height: rect.height, scrollWidth: element.scrollWidth, clientWidth: element.clientWidth } diff --git a/frontend/openapi.json b/frontend/openapi.json index 4942184..d4b988a 100644 --- a/frontend/openapi.json +++ b/frontend/openapi.json @@ -123,6 +123,13 @@ "empirical_scaled", "student_t_ewma", "gjr_garch_t", + "ohlc_har", + "skew_t_ewma", + "egarch_skew_t", + "markov_switching", + "ngboost_pooled", + "earnings_jump", + "iv_physical", "intraday_shadow" ], "type": "string", @@ -202,6 +209,13 @@ "empirical_scaled", "student_t_ewma", "gjr_garch_t", + "ohlc_har", + "skew_t_ewma", + "egarch_skew_t", + "markov_switching", + "ngboost_pooled", + "earnings_jump", + "iv_physical", "intraday_shadow" ], "type": "string", @@ -335,6 +349,13 @@ "empirical_scaled", "student_t_ewma", "gjr_garch_t", + "ohlc_har", + "skew_t_ewma", + "egarch_skew_t", + "markov_switching", + "ngboost_pooled", + "earnings_jump", + "iv_physical", "intraday_shadow" ], "type": "string", @@ -380,6 +401,13 @@ "empirical_scaled", "student_t_ewma", "gjr_garch_t", + "ohlc_har", + "skew_t_ewma", + "egarch_skew_t", + "markov_switching", + "ngboost_pooled", + "earnings_jump", + "iv_physical", "intraday_shadow" ], "type": "string", @@ -1817,6 +1845,13 @@ "empirical_scaled", "student_t_ewma", "gjr_garch_t", + "ohlc_har", + "skew_t_ewma", + "egarch_skew_t", + "markov_switching", + "ngboost_pooled", + "earnings_jump", + "iv_physical", "intraday_shadow" ] }, @@ -2019,7 +2054,8 @@ "type": "string", "enum": [ "regimelib", - "constrained_call_curve" + "constrained_call_curve", + "ssvi" ] }, { @@ -2348,6 +2384,13 @@ "empirical_scaled", "student_t_ewma", "gjr_garch_t", + "ohlc_har", + "skew_t_ewma", + "egarch_skew_t", + "markov_switching", + "ngboost_pooled", + "earnings_jump", + "iv_physical", "intraday_shadow" ] }, diff --git a/frontend/src/ItmChain.browser.test.tsx b/frontend/src/ItmChain.browser.test.tsx index 0647464..b995835 100644 --- a/frontend/src/ItmChain.browser.test.tsx +++ b/frontend/src/ItmChain.browser.test.tsx @@ -1006,7 +1006,7 @@ describe("chain interactions", () => { fireEvent.change(screen.getByLabelText("Min DTE"), { target: { value: "1" } }) expect(screen.getByText(/Displaying 250 of 300 rows/)).toBeTruthy() expect(document.querySelectorAll("tbody tr")).toHaveLength(250) - }) + }, 15_000) it("refetches puts with OTM default and put filters", async () => { fetchMock.mockImplementation(async (_ticker, side) => ( diff --git a/frontend/src/ModelComparison.browser.test.tsx b/frontend/src/ModelComparison.browser.test.tsx index 8cb4ac0..777baaa 100644 --- a/frontend/src/ModelComparison.browser.test.tsx +++ b/frontend/src/ModelComparison.browser.test.tsx @@ -44,13 +44,14 @@ it("shows all physical and market methods, including pending and unavailable res it("reports simulation precision as sampling error rather than forecast confidence", () => { render() fireEvent.click(screen.getByText("Compare models")) fireEvent.click(screen.getByText("Student-t EWMA")) const physical = screen.getByRole("region", { name: "Physical forecast models" }).textContent - expect(physical).toContain("4,096 simulated paths") + expect(physical).toContain("4,096 terminal-price scenarios") expect(physical).toContain("Maximum 95% simulation error ±1.6 percentage points; model uncertainty excluded") expect(physical).toContain("Model spread: N/A") expect(physical).toContain("data abc123456789") @@ -89,3 +90,87 @@ it("keeps retrospective and as-issued evidence distinct and reports N=0 without expect(replayPanel.textContent).toContain("Call calibration: N=0") expect(replayPanel.textContent).toContain("Put calibration: N=0") }) + +it("keeps the selected physical model first without hiding valid alternatives or risk-neutral methods", () => { + const physical: PredictiveOddsView[] = [ + { method: "iv_physical", status: "unavailable", reason: "rights_cleared_option_history_unavailable" }, + { method: "ngboost_pooled", status: "pending", reason: "candidate_not_prepared" }, + { method: "markov_switching", status: "available", itm_pct_tenths: 570, otm_pct_tenths: 430, atm_pct_tenths: 0, price_basis: "completed_close" }, + { method: "earnings_jump", status: "unavailable", reason: "verified_release_time_history_unavailable" }, + { method: "ohlc_har", status: "available", itm_pct_tenths: 610, otm_pct_tenths: 390, atm_pct_tenths: 0, price_basis: "completed_close" }, + { method: "skew_t_ewma", status: "unavailable", reason: "skew_t_fit_failed" }, + { method: "egarch_skew_t", status: "unavailable", reason: "egarch_parameters_invalid" }, + ] + const market: MarketOddsView[] = [ + { method: "ssvi", status: "unavailable", reason: "rights_cleared_option_history_unavailable" }, + { method: "constrained_call_curve", status: "pending", reason: "curve_fit_pending" }, + { method: "regimelib", status: "available", itm_pct_tenths: 540, otm_pct_tenths: 460 }, + ] + render() + fireEvent.click(screen.getByRole("button", { name: "Compare models" })) + const dialog = screen.getByRole("dialog", { name: "Compare models" }) + const physicalRows = Array.from(dialog.querySelectorAll('[aria-label="Physical forecast models"] .model-result-heading')) + expect(physicalRows.map((row) => row.querySelector("strong")?.textContent)).toEqual([ + "EGARCH skewed-tSelected", "Two-regime switching variance", "Daily OHLC range/HAR proxy", + "Pooled NGBoost", "IV-informed forecast", "Earnings jump", "Skewed-t EWMA", + ]) + expect(physicalRows[0].textContent).toContain("Unavailable · egarch parameters invalid") + expect(physicalRows[1].textContent).toContain("57.0% ITM · 43.0% OTM") + const marketRows = Array.from(dialog.querySelectorAll('[aria-label="Risk-neutral market models"] .model-result-heading')) + expect(marketRows.map((row) => row.querySelector("strong")?.textContent)).toEqual([ + "RegimelibBenchmark", "Constrained call curve", "SSVI volatility surface", + ]) + expect(marketRows[2].textContent).toContain("rights cleared option history unavailable") + expect(dialog.textContent).not.toContain("confidence score: ") +}) + +it("shows adjusted matched evidence and CRPS only as evidence, not as odds", () => { + const report = { + generated_at: "2026-09-27T12:00:00Z", model_version: "ohlc-har-v1", input_version: "frozen-v1", + tickers: 20, independent_date_blocks: 20, ticker_origin_horizon_units: 500, + contract_forecasts_available: 500, contract_cells_attempted: 525, + brier: { baseline: 0.24, candidate: 0.20, paired_delta: -0.04, + bootstrap_95: [-0.06, -0.02], bootstrap_familywise_95: [-0.08, -0.01], comparison_count: 10 }, + log_loss: { baseline: 0.66, candidate: 0.61, paired_delta: -0.05 }, + crps: { baseline: 3.4, candidate: 3.1, scored_units: 480 }, + calibration_by_side: { call: [], put: [] }, latency_ms: {}, rejection_reasons: { missing_bar: 25 }, + } + render() + fireEvent.click(screen.getByRole("button", { name: "Compare models" })) + fireEvent.click(screen.getByText("Daily OHLC range/HAR proxy")) + fireEvent.click(screen.getByText(/Prospective as-issued · N=500/)) + fireEvent.click(screen.getByText(/Retrospective replay · N=500/)) + const prospective = screen.getByRole("region", { name: "Prospective as-issued evidence" }) + const retrospective = screen.getByRole("region", { name: "Retrospective replay evidence" }) + expect(prospective.textContent).toContain("Bonferroni-adjusted 95% within-band calendar-block interval (10 methods) [-0.080, -0.010]") + expect(prospective.textContent).toContain("CRPS 480 scored units · model 3.100 · EWMA 3.400") + expect(prospective.textContent).toContain("Coverage 500/525 recorded contract cells (95.2%)") + expect(prospective.textContent).toContain("missing_bar: 25") + expect(retrospective.textContent).toContain("Current-vintage screening; not an as-issued accuracy claim") + expect(retrospective.textContent).toContain("significance not estimable") +}) + +it("shows missed capture windows separately from scored intraday coverage", () => { + render() + fireEvent.click(screen.getByRole("button", { name: "Compare models" })) + fireEvent.click(screen.getByText("Intraday conditioned")) + fireEvent.click(screen.getByText(/Prospective as-issued · N=0/)) + const details = screen.getByRole("region", { name: "Prospective as-issued evidence" }) + expect(details.textContent).toContain("Capture windows, all horizons (last 35 days): 3/5 captured · 1 missed · 1 pending") + expect(screen.getByText(/60 completed returns/)).toBeTruthy() +}) diff --git a/frontend/src/ModelComparison.tsx b/frontend/src/ModelComparison.tsx index a397deb..7b7eeaf 100644 --- a/frontend/src/ModelComparison.tsx +++ b/frontend/src/ModelComparison.tsx @@ -51,11 +51,16 @@ function evidencePanel(title: string, value: unknown) { if (!report) return

{title}: no report available

const brier = data(report.brier) const logLoss = data(report.log_loss) + const crps = data(report.crps) const blocks = number(report.independent_date_blocks) ?? 0 const units = number(report.ticker_origin_horizon_units) ?? 0 const attempted = number(report.contract_cells_attempted) ?? 0 const available = number(report.contract_forecasts_available) ?? 0 - const interval = brier?.bootstrap_95 + const adjustedInterval = brier?.bootstrap_familywise_95 + const interval = Array.isArray(adjustedInterval) ? adjustedInterval : brier?.bootstrap_95 + const intervalLabel = Array.isArray(adjustedInterval) + ? `Bonferroni-adjusted 95% within-band calendar-block interval (${count(brier?.comparison_count)} methods)` + : "exploratory 95% calendar-block interval" const intervalText = report.significance === "not_applicable" ? "not applicable for the baseline" : blocks >= 20 && Array.isArray(interval) && interval.length === 2 ? `[${score(interval[0])}, ${score(interval[1])}]` : "significance not estimable" @@ -72,10 +77,12 @@ function evidencePanel(title: string, value: unknown) { {report.coverage_basis === "recorded_contract_cells_with_baseline_issuance" ?

This coverage is conditional on recorded cells with an EWMA issuance; it does not cover every scheduled origin.

: null} {report.coverage_basis === "recorded_current_version_baseline_attempts" ?

Coverage counts recorded current-version EWMA attempts; older unversioned attempts and missed origins are excluded.

: null} {report.coverage_basis === "recorded_current_version_candidate_cells" ?

Availability counts recorded current-version candidate cells, including cells without EWMA as failures. Older unversioned attempts and missed origins are excluded.

: null} - {typeof report.coverage_basis === "string" && report.coverage_basis.includes("window") ?

This coverage counts recorded timestamped windows only; missed windows are not in the ledger.

: null} + {typeof report.coverage_basis === "string" && report.coverage_basis.includes("window") ?

Matched-score coverage counts recorded timestamped windows only.

: null} + {data(report.capture_windows_all_horizons) ?

Capture windows, all horizons (last 35 days): {count(data(report.capture_windows_all_horizons)?.captured)}/{count(data(report.capture_windows_all_horizons)?.expected)} captured · {count(data(report.capture_windows_all_horizons)?.missed)} missed · {count(data(report.capture_windows_all_horizons)?.pending)} pending.

: null} {report.replay_scheduled_units != null ?

Replay fit coverage {count(report.replay_baseline_available_units)}/{count(report.replay_scheduled_units)} scheduled units ({percent(report.replay_fit_coverage)}) · skipped before strikes: {reasonList(report.replay_rejection_reasons)}

: null} -

Brier {units ? score(brier?.candidate) : "N/A"} (EWMA {units ? score(brier?.baseline) : "N/A"}); paired change {units ? score(brier?.paired_delta) : "N/A"}; 95% calendar-block interval {intervalText}

+

Brier {units ? score(brier?.candidate) : "N/A"} (EWMA {units ? score(brier?.baseline) : "N/A"}); paired change {units ? score(brier?.paired_delta) : "N/A"}; {intervalLabel} {intervalText}

Log loss {units ? score(logLoss?.candidate) : "N/A"} (EWMA {units ? score(logLoss?.baseline) : "N/A"}); paired change {units ? score(logLoss?.paired_delta) : "N/A"}

+

CRPS {count(crps?.scored_units)} scored units · model {score(crps?.candidate)} · EWMA {score(crps?.baseline)}

{data(report.quote_reanchored_comparator) ?

Matched quote-reanchored comparator: Brier {score(data(report.quote_reanchored_comparator)?.brier)} · log loss {score(data(report.quote_reanchored_comparator)?.log_loss)}

: null} {data(report.by_window) ?

Matched intraday windows: {Object.entries(data(report.by_window)!).map(([window, value]) => `${window} N=${count(data(value)?.scored_ticker_date_window_expiry_units)}`).join(" · ")}

: null} {(["call", "put"] as const).map((side) => { @@ -95,9 +102,9 @@ function PhysicalResult({ model, selected, evidenceIndex }: { model: PredictiveO const evidence = data(model.model_evidence) ?? data(model.evidence_key ? evidenceIndex?.[model.evidence_key] : null) const support = model.support == null ? "support N/A" : model.method === "empirical_scaled" ? `${count(model.support)} historical horizon samples (overlapping)` - : model.method === "student_t_ewma" || model.method === "gjr_garch_t" - ? `${count(model.support)} simulated paths` - : `${count(model.support)} completed returns` + : model.method === "lognormal_ewma" || model.method === "intraday_shadow" + ? `${count(model.support)} completed returns` + : `${count(model.support)} terminal-price scenarios` return
  • {physicalModelName(model.method)}{model.method === selected ? Selected : null}{valid ? `${unsignedPercentTenths(model.itm_pct_tenths)} ITM · ${unsignedPercentTenths(model.otm_pct_tenths)} OTM` : `${model.status === "pending" ? "Pending" : "Unavailable"} · ${reasonLabel(model.reason)}`}
    @@ -145,6 +152,16 @@ export default function ModelComparison({ physical, market, selected, evidenceIn const disagreement = comparable.length >= 2 ? ((Math.max(...comparable.map((model) => model.itm_pct_tenths!)) - Math.min(...comparable.map((model) => model.itm_pct_tenths!))) / 10).toFixed(1) : null + const orderedPhysical = [...(physical ?? [])].sort((a, b) => { + const rank = (model: PredictiveOddsView) => model.method === selected ? 0 + : predictiveAvailable(model) ? 1 : model.status === "pending" ? 2 : 3 + return rank(a) - rank(b) + }) + const orderedMarket = [...(market ?? [])].sort((a, b) => { + const rank = (model: MarketOddsView) => oddsAvailable(model) ? 0 + : model.status === "pending" ? 1 : 2 + return rank(a) - rank(b) + }) return <> {open ? createPortal( { setOpen(false); trigger.current?.focus() }}> @@ -152,10 +169,10 @@ export default function ModelComparison({ physical, market, selected, evidenceIn

    Stock-close forecasts Real-world expiry-close odds

    Your selection sets the compact odds and hypothetical risk.

    Model spread: {disagreement == null ? "N/A (fewer than two comparable estimates)" : `${disagreement} percentage points across ${comparable.length} completed-close models`}

    -
      {physical?.length ? physical.map((model, index) => ) :
    • No physical model results yet.
    • }
    +
      {orderedPhysical.length ? orderedPhysical.map((model, index) => ) :
    • No physical model results yet.
    • }

    Option-price estimates Risk-neutral odds

    Quote fit does not measure realized forecast accuracy.

    -
      {market?.length ? market.map((model, index) => ) :
    • No market model results yet.
    • }
    +
      {orderedMarket.length ? orderedMarket.map((model, index) => ) :
    • No market model results yet.
    • }
    , document.body) : null} diff --git a/frontend/src/forecastModels.ts b/frontend/src/forecastModels.ts index a2dc9e4..fba65cd 100644 --- a/frontend/src/forecastModels.ts +++ b/frontend/src/forecastModels.ts @@ -3,6 +3,13 @@ export const PHYSICAL_MODELS = [ "empirical_scaled", "student_t_ewma", "gjr_garch_t", + "ohlc_har", + "skew_t_ewma", + "egarch_skew_t", + "markov_switching", + "ngboost_pooled", + "earnings_jump", + "iv_physical", "intraday_shadow", ] as const @@ -15,6 +22,13 @@ export const PHYSICAL_MODEL_NAMES: Record = { empirical_scaled: "Scaled empirical", student_t_ewma: "Student-t EWMA", gjr_garch_t: "GJR-GARCH Student-t", + ohlc_har: "Daily OHLC range/HAR proxy", + skew_t_ewma: "Skewed-t EWMA", + egarch_skew_t: "EGARCH skewed-t", + markov_switching: "Two-regime switching variance", + ngboost_pooled: "Pooled NGBoost", + earnings_jump: "Earnings jump", + iv_physical: "IV-informed forecast", intraday_shadow: "Intraday conditioned", } @@ -29,5 +43,5 @@ export function physicalModelName(value: string | null | undefined): string { export function marketModelName(value: string | null | undefined): string { return value === "regimelib" ? "Regimelib" : value === "constrained_call_curve" - ? "Constrained call curve" : "Unknown market model" + ? "Constrained call curve" : value === "ssvi" ? "SSVI volatility surface" : "Unknown market model" } diff --git a/frontend/src/generated/types.gen.ts b/frontend/src/generated/types.gen.ts index c437d1d..2c4884c 100644 --- a/frontend/src/generated/types.gen.ts +++ b/frontend/src/generated/types.gen.ts @@ -597,7 +597,7 @@ export type HypotheticalRiskView = { /** * Forecast Method */ - forecast_method?: 'lognormal_ewma' | 'empirical_scaled' | 'student_t_ewma' | 'gjr_garch_t' | 'intraday_shadow' | null; + forecast_method?: 'lognormal_ewma' | 'empirical_scaled' | 'student_t_ewma' | 'gjr_garch_t' | 'ohlc_har' | 'skew_t_ewma' | 'egarch_skew_t' | 'markov_switching' | 'ngboost_pooled' | 'earnings_jump' | 'iv_physical' | 'intraday_shadow' | null; /** * Reason */ @@ -681,7 +681,7 @@ export type MarketOddsView = { /** * Method */ - method?: 'regimelib' | 'constrained_call_curve' | null; + method?: 'regimelib' | 'constrained_call_curve' | 'ssvi' | null; /** * Status */ @@ -811,7 +811,7 @@ export type PredictiveOddsView = { /** * Method */ - method?: 'lognormal_ewma' | 'empirical_scaled' | 'student_t_ewma' | 'gjr_garch_t' | 'intraday_shadow' | null; + method?: 'lognormal_ewma' | 'empirical_scaled' | 'student_t_ewma' | 'gjr_garch_t' | 'ohlc_har' | 'skew_t_ewma' | 'egarch_skew_t' | 'markov_switching' | 'ngboost_pooled' | 'earnings_jump' | 'iv_physical' | 'intraday_shadow' | null; /** * Reason */ @@ -1196,7 +1196,7 @@ export type GetCoveredCallsApiCoveredCallsTickerGetData = { /** * Forecast Model */ - forecast_model?: 'lognormal_ewma' | 'empirical_scaled' | 'student_t_ewma' | 'gjr_garch_t' | 'intraday_shadow'; + forecast_model?: 'lognormal_ewma' | 'empirical_scaled' | 'student_t_ewma' | 'gjr_garch_t' | 'ohlc_har' | 'skew_t_ewma' | 'egarch_skew_t' | 'markov_switching' | 'ngboost_pooled' | 'earnings_jump' | 'iv_physical' | 'intraday_shadow'; }; url: '/api/covered-calls/{ticker}'; }; @@ -1235,7 +1235,7 @@ export type GetCashSecuredPutsApiCashSecuredPutsTickerGetData = { /** * Forecast Model */ - forecast_model?: 'lognormal_ewma' | 'empirical_scaled' | 'student_t_ewma' | 'gjr_garch_t' | 'intraday_shadow'; + forecast_model?: 'lognormal_ewma' | 'empirical_scaled' | 'student_t_ewma' | 'gjr_garch_t' | 'ohlc_har' | 'skew_t_ewma' | 'egarch_skew_t' | 'markov_switching' | 'ngboost_pooled' | 'earnings_jump' | 'iv_physical' | 'intraday_shadow'; }; url: '/api/cash-secured-puts/{ticker}'; }; @@ -1325,7 +1325,7 @@ export type GetWatchlistApiWatchlistGetData = { /** * Forecast Model */ - forecast_model?: 'lognormal_ewma' | 'empirical_scaled' | 'student_t_ewma' | 'gjr_garch_t' | 'intraday_shadow'; + forecast_model?: 'lognormal_ewma' | 'empirical_scaled' | 'student_t_ewma' | 'gjr_garch_t' | 'ohlc_har' | 'skew_t_ewma' | 'egarch_skew_t' | 'markov_switching' | 'ngboost_pooled' | 'earnings_jump' | 'iv_physical' | 'intraday_shadow'; }; url: '/api/watchlist'; }; @@ -1355,7 +1355,7 @@ export type AddWatchApiWatchlistPostData = { /** * Forecast Model */ - forecast_model?: 'lognormal_ewma' | 'empirical_scaled' | 'student_t_ewma' | 'gjr_garch_t' | 'intraday_shadow'; + forecast_model?: 'lognormal_ewma' | 'empirical_scaled' | 'student_t_ewma' | 'gjr_garch_t' | 'ohlc_har' | 'skew_t_ewma' | 'egarch_skew_t' | 'markov_switching' | 'ngboost_pooled' | 'earnings_jump' | 'iv_physical' | 'intraday_shadow'; }; url: '/api/watchlist'; }; diff --git a/frontend/src/generated/zod.gen.ts b/frontend/src/generated/zod.gen.ts index a4b216a..b353059 100644 --- a/frontend/src/generated/zod.gen.ts +++ b/frontend/src/generated/zod.gen.ts @@ -21,6 +21,13 @@ export const zHypotheticalRiskView = z.object({ 'empirical_scaled', 'student_t_ewma', 'gjr_garch_t', + 'ohlc_har', + 'skew_t_ewma', + 'egarch_skew_t', + 'markov_switching', + 'ngboost_pooled', + 'earnings_jump', + 'iv_physical', 'intraday_shadow' ]).nullish(), reason: z.string().nullish(), @@ -57,7 +64,11 @@ export const zJob = z.object({ * MarketOddsView */ export const zMarketOddsView = z.object({ - method: z.enum(['regimelib', 'constrained_call_curve']).nullish(), + method: z.enum([ + 'regimelib', + 'constrained_call_curve', + 'ssvi' + ]).nullish(), status: z.enum([ 'pending', 'available', @@ -142,6 +153,13 @@ export const zPredictiveOddsView = z.object({ 'empirical_scaled', 'student_t_ewma', 'gjr_garch_t', + 'ohlc_har', + 'skew_t_ewma', + 'egarch_skew_t', + 'markov_switching', + 'ngboost_pooled', + 'earnings_jump', + 'iv_physical', 'intraday_shadow' ]).nullish(), reason: z.string().nullish(),