From 890bd6ca0562b73da7d65fd455a8962f318cf3ed Mon Sep 17 00:00:00 2001 From: Pigbibi <20649888+Pigbibi@users.noreply.github.com> Date: Tue, 6 Oct 2026 06:08:21 +0800 Subject: [PATCH] feat: add causal semiconductor regime research observation --- .../semiconductor-regime-observer-research.md | 143 +++++ ...onductor-regime-observer-research.zh-CN.md | 143 +++++ .../semiconductor_regime_observer_research.py | 509 ++++++++++++++++++ ..._semiconductor_regime_observer_research.py | 435 +++++++++++++++ 4 files changed, 1230 insertions(+) create mode 100644 docs/semiconductor-regime-observer-research.md create mode 100644 docs/semiconductor-regime-observer-research.zh-CN.md create mode 100644 src/quant_strategy_plugins/semiconductor_regime_observer_research.py create mode 100644 tests/test_semiconductor_regime_observer_research.py diff --git a/docs/semiconductor-regime-observer-research.md b/docs/semiconductor-regime-observer-research.md new file mode 100644 index 0000000..3e8bd8b --- /dev/null +++ b/docs/semiconductor-regime-observer-research.md @@ -0,0 +1,143 @@ +# Semiconductor regime observation research + +Status: `UNVALIDATED_RESEARCH`. This is a callable pure research module, not an enabled plugin, strategy router, or trading permission. No production calibration or economic improvement is claimed. + +[中文完整说明](semiconductor-regime-observer-research.zh-CN.md) + +## Boundary and existing interfaces + +`quant_strategy_plugins.semiconductor_regime_observer_research` adds close-only SOXX/SOXL evidence beside the existing research modules. It reuses the current `plugin_signal_envelope_v2.canonical_json_bytes` and `build_signal_envelope` without changing their contract. + +Shared price/SMA, population realized-volatility, and trailing-drawdown formulas follow the existing QQQ observer. Synthetic tests compare these four facts with that observer on identical artificial numbers. The QQQ observer remains unchanged and is never relabeled as SOXX. + +The earlier QSP `3416da47580be1537183e378f3ca9efacc6f9b8c` is a fixed source anchor declared by the UES optional research dependency. It is not proof of active installation, enablement, or production producer/consumer pins. + +Package-level __init__ exports are unchanged, and no runtime entry is registered. No new CLI, runner entry, catalog entry, dependency, default/live configuration, service, or scheduler is added. Ordinary import does not enable or run the research entrypoint. The module itself has no provider, filesystem, clock, AI, broker, or strategy dependency; the existing package-level imports are unchanged. + +Outputs contain no position, capital, order, control, or authorization fields. Keep observation, approved strategy targets, risk reductions/vetoes, and execution separate. Restoring data or lifting risk restrictions must not buy from an old signal. New exposure requires a current valid approved strategy signal, the shared budget, existing holdings/pending-order netting, and existing account/release permissions. These are downstream constraints, not actions implemented here. + +## Callable API + +- `build_semiconductor_regime_observation(snapshot, config)`: deterministic observation; incomplete or inconsistent data returns unknown +- `semiconductor_regime_observer_config_sha256(config)`: canonical frozen research-configuration identity +- `semiconductor_regime_observation_usable_at(observation, at)`: explicit-time qualification/freshness check; never permission to trade +- `build_semiconductor_regime_signal_v2(*, snapshot, config, producer, input_provenance)`: computes the observation and calls the existing pure V2 builder + +All calculation configuration is mandatory. Invalid configuration or V2 binding raises the sanitized `ContractError`. Data defects are separate quality reasons and never become a market risk label. + +The V2 research entrypoint is exactly: + +`quant_strategy_plugins.semiconductor_regime_observer_research:build_semiconductor_regime_signal_v2` + +The wrapper requires `producer.repo=QuantStrategyLab/QuantStrategyPlugins`, this exact entrypoint, and the matching configuration hash. The unchanged V2 helper validates immutable producer revision/code/config digest formats and rejects forbidden control fields and mutable latest references. A caller's code/revision declaration is not installed-byte proof; independent P3 evidence remains necessary. + +`input_provenance` must contain exactly `p1_manifest_sha256`, `input_root_sha256`, and `date_cutoff`. The cutoff must match the observation; the root must equal its causal `input_sha256`. The P1 hash must have valid digest syntax. This verifies declared consistency, not P1 content, provider truth, or research qualification. A full raw-file manifest containing invisible future rows cannot substitute for the causal projection's qualified manifest. + +## Symbols and proxy limits + +Only the following exact pairs are supported: + +| symbol | series_role | Meaning | +| --- | --- | --- | +| SOXX | SIGNAL_PROXY | Semiconductor signal proxy for an SOXL research candidate | +| SOXL | EXECUTION_BENCHMARK | Observation of SOXL's own price path | + +Each series has independent input identity. SOXX signals do not imply identical SOXL performance: leverage, daily reset, tracking differences, volatility drag, overnight gaps, and execution/cost definitions need separate validation. No execution prices or realized strategy returns are computed. + +QQQ belongs to the separate QQQ/TQQQ background. This module rejects QQQ and never copies a QQQ state to semiconductor evidence. + +## Exact input contract + +Top-level keys must be exactly: + +`schema_version`, `symbol`, `series_role`, `as_of`, `available_at`, `decision_at`, `calendar`, `adjustment`, `components`, `bars`. + +- `schema_version`: `qsl.semiconductor-regime-input.research.v1` +- `as_of`: strict YYYY-MM-DD ending trading session +- `available_at`: first complete availability of this causal input snapshot; no later than decision_at and no earlier than used rows/components +- `decision_at`: actual observation evaluation/completion/publication time; a real producer must not backdate it before its completed publication +- All timestamps require explicit ISO timezone information and are normalized to UTC in output + +`calendar` has exactly `id`, `version`, `available_at`, `sessions`. The caller supplies the complete official session list and an immutable calendar version, including holidays/early closes; the module does not infer them. Each session has exactly `date`, `close_at`, `complete`. Selected sessions must be strictly increasing, complete=true, closed by decision_at, end at as_of, and cover the required windows. The UTC date of close_at must match the session label for this SOXX/SOXL contract. Extra prices outside the declared calendar fail qualification. + +`adjustment` has exactly `basis`, `version`, `available_at`, `point_in_time_attested`. Basis is explicitly `raw`, `split_adjusted`, or `split_and_distribution_adjusted`; availability must be no later than decision_at and point_in_time_attested must be true. Versions/IDs cannot be empty or mutable latest values. Different adjustment bases must not be treated as one comparable experiment; raw split jumps remain a data-validity risk. No factor recalculation occurs. + +`components` has exactly `prices`, `calendar`, `adjustment`. Each component has exactly `as_of`, `available_at`, `version`. All end at the same as_of and are available by decision_at. Calendar/adjustment versions and availability match their respective declarations; price-component availability must not predate its used rows. Version identifiers describe frozen producer/interpretation versions, not a rewritten entire-file latest identity. + +Each bar has exactly `symbol`, `date`, `close`, `closed`, `available_at`, `adjustment_available_at`, `adjustment_version`. Selected rows must match the symbol and factor version, have finite positive non-boolean closes, closed=true, and publication no earlier than session close and factor availability. Duplicate or unsorted selected dates are rejected; the module does not sort or choose between multiple already-visible revisions. + +Qualification is explicitly `qualified_by_declaration`: caller attestations and internal consistency only. The module cannot prove provider historical truth, official-calendar completeness, or factor availability. If both calendar and prices omit a session while declaring completeness, this check alone cannot detect it. Real P0 evidence and immutable PIT snapshots remain required. + +## Causal projection and hash + +The chosen policy slices at as_of/decision_at: + +1. Rows dated after as_of are invisible before any other columns are inspected +2. For historical dates, price availability after decision_at makes the row invisible, including unrelated or malformed remaining columns +3. Factor availability after decision_at also excludes that adjusted row +4. Required unavailable sessions are missing coverage, not imputed prices or market pressure +5. Calendar rows after as_of are likewise invisible + +An unparseable selector cannot prove future status and produces `ROW_SELECTOR_INVALID`. Top-level/component declarations must remain frozen for that historical decision; rewriting them is not a hidden future-row append. + +`input_sha256` hashes only the projected input using the existing QSP canonical JSON helper. Invisible rows do not enter features, validation, counts, reasons, or identity. Appending them leaves the entire historical observation and hash unchanged. Invalid selected values receive diagnostic error-type identity and remain unknown, never qualified evidence. Configuration hash and V2 payload hash are separate identities. + +## Explicit features and configuration + +Configuration schema is `qsl.semiconductor-regime-config.research.v1`. Its exact additional fields are: + +- `sma_window_sessions` ≥2; `sma_slope_lag_sessions` ≥1 +- `path_window_returns` ≥2, requiring one more close than return steps +- `short_vol_window_returns` and `long_vol_window_returns` ≥2, short≤long +- `drawdown_window_sessions` ≥2; `annualization_sessions` ≥1 +- `ttl_seconds` ≥1; `classification` explicitly null or the complete hypothesis below + +There are no numerical defaults. Freeze windows, TTL, thresholds, and comparison policy before evaluation; never choose them from future outcomes. Very short windows and values in tests are `SYNTHETIC_FIXTURE_ONLY`, not deployment recommendations. + +Minimum required history is the maximum of SMA window+slope lag, path returns+1, short returns+1, long returns+1, and drawdown sessions. + +| Feature | Definition | +| --- | --- | +| close_to_sma_ratio | last close/current trailing SMA −1 | +| sma_slope_per_session | (current SMA/lagged SMA −1)/lag sessions | +| path_efficiency | absolute endpoint change/sum of absolute path steps | +| short/long_realized_volatility_annualized | population standard deviation of arithmetic returns × square root of annualization sessions | +| volatility_ratio | short volatility/long volatility | +| trailing_drawdown_ratio | last close/trailing-window peak −1 | + +Finite features are rounded to 12 decimals; classification uses that same precision. Zero path variation yields null path efficiency; zero long volatility yields null volatility ratio. These are undefined market features, not automatic range or safety labels. Non-finite computation fails qualification and clears all features. + +## Independent axes and research hypotheses + +`classification=null` is normal feature-only mode. Qualified input still yields features, while all axes return unknown/`CLASSIFICATION_NOT_CONFIGURED`. + +A classifier must have exactly `hypothesis_id`, `status=UNVALIDATED_RESEARCH_HYPOTHESIS`, `direction`, `trendiness`, `pressure`. No calibrated status or probabilities are accepted. + +- Direction: explicit positive `price_distance_min` and `sma_slope_min`; both facts exceed positive thresholds for up, negative thresholds for down. Significant opposite signs or threshold gaps yield unknown +- Trendiness: 0≤`range_efficiency_max`<`trend_efficiency_min`≤1; low efficiency is range_like, high is trend_like, the gap is unknown. Range_like does not prove profitable mean reversion or authorize RSI2 +- Pressure: 0≤`drawdown_normal_max`<`drawdown_stress_min`≤1 and 0≤`volatility_ratio_normal_max`<`volatility_ratio_stress_min`; both market features must align for normal or stressed. Significant disagreement and threshold gaps yield unknown + +Pressure uses market facts only. Missing/inconsistent input sets quality unknown and every axis unknown/`INPUT_QUALIFICATION_FAILED`, never stressed or risk_off. Zero denominator reasons remain independent market-feature unknowns when declarations are qualified. + +No fitted model, HMM, parameter optimizer, state memory, dwell time, or hysteresis is hidden here. Any later switching policy needs its own preregistered causal candidate/version and validation. + +## Availability, expiry, and outputs + +Output records schema/feature version, configuration/input digest, symbol/role, as_of, input calendar/factor/producer context, features, axes, and separate quality. + +- `input_available_at`: declared complete input availability +- `decision_at`: actual evaluation/completed publication instant +- `available_at`: no earlier than either; equals decision_at for qualified observations, never the earlier raw-price publication time +- `valid_until`: as_of's official close_at + explicit ttl_seconds, exclusive + +At decision_at≥valid_until the observation is expired and unknown. Replaying old closes or changing evaluation time never renews expiry. The usability helper returns true only for declared-qualified evidence at or after availability/decision and strictly before expiry; axes may still be unknown. True means usable evidence, not transaction permission. + +Consumers must independently verify actual publication time, immutable V2 payload, P1/P2/P3 evidence, source identity, and expiry. available_at/decision_at checks validate declarations only; actual computation/publication-time proof remains P0 work. Synthetic future-row invariance is not real-provider PIT evidence. Hashes and pure functions are not authentication or production adoption proof. + +## Validation and next admission + +Run `PYTHONPATH=src python -m pytest -q tests/test_semiconductor_regime_observer_research.py tests/test_qqq_price_regime_observer_v2.py tests/test_plugin_signal_envelope_v2.py`, then the existing full tests, Ruff, whitespace check, and package build. + +Synthetic tests cover future/late invisible-row invariance, late required bars, incomplete/unclosed sessions, unavailable factors, mismatched component dates/versions, zero variation/volatility, NaN/infinity/extreme finite values, duplicate/disordered dates, conflicts, TTL boundary/replay, feature-only mode, and original QQQ/V2 compatibility. Producer repo/entrypoint/config and causal-root/cutoff binding are separately tested. + +These are contract/interface results, not real market backtests, economic alpha, provider PIT verification, production deployment, or active consumption. Further research still needs qualified real P0 inputs, frozen comparable baselines, preregistration/all-trial records, causal out-of-sample results, common costs/risk controls, component ablation, and real no-order paired shadow before approved use. diff --git a/docs/semiconductor-regime-observer-research.zh-CN.md b/docs/semiconductor-regime-observer-research.zh-CN.md new file mode 100644 index 0000000..ad9839f --- /dev/null +++ b/docs/semiconductor-regime-observer-research.zh-CN.md @@ -0,0 +1,143 @@ +# 半导体三轴状态观察研究 + +状态:`UNVALIDATED_RESEARCH`。这是可直接调用的纯研究模块,不是已启用插件、策略路由或交易权限。没有生产校准或经济增量结论。 + +[Full English contract](semiconductor-regime-observer-research.md) + +## 边界与现有接口 + +`quant_strategy_plugins.semiconductor_regime_observer_research` 在现有研究模块旁新增 SOXX/SOXL 收盘观察证据,复用当前 `plugin_signal_envelope_v2.canonical_json_bytes` 和 `build_signal_envelope`,不修改原接口合同。 + +价格/SMA、总体方差实现波动、尾部峰值回撤沿用现有 QQQ observer 口径。合成测试用相同人工数列核对四个共有特征;原 QQQ observer 保持原样,绝不重标为 SOXX。 + +早期 QSP `3416da47580be1537183e378f3ca9efacc6f9b8c` 是 UES optional research dependency 声明的固定源码锚点,不证明现役安装、启用或生产 producer/consumer pin。 + +不修改包级 __init__ 导出,不注册运行入口。不新增 CLI、runner 入口、catalog、依赖、默认/live 配置、服务或调度。普通 import 不启用或运行该研究入口。新模块本身无 provider、文件系统、时钟、AI、券商或策略依赖;原包级 import 不变。 + +输出没有仓位、资金、订单、控制或授权字段。观察、获批策略目标、风险缩减/否决和执行分别负责。数据恢复或风险解除不能用旧信号补仓;新风险须有当前有效的获批策略信号、共同预算、已有持仓/挂单净额及原账户/恢复权限。这是后续消费约束,本模块没有实现对应动作。 + +## 可调用 API + +- `build_semiconductor_regime_observation(snapshot, config)`:确定性观察;输入缺失或不一致返回 unknown +- `semiconductor_regime_observer_config_sha256(config)`:冻结研究配置的规范身份 +- `semiconductor_regime_observation_usable_at(observation, at)`:显式时间的数据资格/新鲜度检查,不是交易许可 +- `build_semiconductor_regime_signal_v2(*, snapshot, config, producer, input_provenance)`:计算观察并调用原纯 V2 builder + +全部计算配置必须明确提供。非法配置或 V2 绑定抛出不回显数据的 `ContractError`。数据缺陷有独立资格原因,不会变成市场风险标签。 + +V2 研究入口准确为: + +`quant_strategy_plugins.semiconductor_regime_observer_research:build_semiconductor_regime_signal_v2` + +wrapper 要求 producer.repo=QuantStrategyLab/QuantStrategyPlugins、上述准确 entrypoint 及对应配置 hash。原 V2 helper 继续校验不可变 revision/code/config digest 的形式,并拒绝禁用控制字段和 mutable latest 引用。调用方的 code/revision 声明不是实际已安装字节证明,仍须独立 P3 证据。 + +input_provenance 恰有 p1_manifest_sha256、input_root_sha256、date_cutoff;cutoff 与观察一致,root 必须等于观察的因果 input_sha256,P1 hash 形式须合法。这只验证声明一致性,不证明 P1 内容、供应商真实性或研究资格。含隐藏未来行的全文件 manifest 不能替代合格因果投影 manifest。 + +## 标的与代理差异 + +只接受以下准确配对: + +| symbol | series_role | 含义 | +| --- | --- | --- | +| SOXX | SIGNAL_PROXY | SOXL 研究候选的半导体信号代理 | +| SOXL | EXECUTION_BENCHMARK | SOXL 自身价格路径的观察基准 | + +两个序列分别形成输入身份。SOXX 信号不能默认等于 SOXL 收益:杠杆、日重置、跟踪差异、波动耗损、隔夜跳空及成交/成本口径均须独立验证。不计算执行价格或真实策略收益。 + +QQQ 属于独立 QQQ/TQQQ 背景。本模块拒绝 QQQ,不复制其状态为半导体证据。 + +## 严格输入合同 + +顶层字段必须恰为: + +schema_version、symbol、series_role、as_of、available_at、decision_at、calendar、adjustment、components、bars。 + +- schema_version:qsl.semiconductor-regime-input.research.v1 +- as_of:严格 YYYY-MM-DD 结束交易日 +- available_at:此因果输入快照首次完整可得时间;不晚于 decision_at,不早于使用的行/组件 +- decision_at:观察实际评估/完成发布时点;真实 producer 不能回填为完成发布之前 +- 全部时间要求显式 ISO 时区,输出归一 UTC + +calendar 恰有 id、version、available_at、sessions。调用方提供完整正式交易日列表及不可变版本,包括节假日/提前收盘;模块不自行猜测。每个 session 恰有 date、close_at、complete。选中交易日严格递增,complete=true,decision_at 前已收盘,结束于 as_of 且覆盖计算窗口。此 SOXX/SOXL 合同要求 close_at 的 UTC 日期等于 session 标签。价格日期不在声明日历内则资格失败。 + +adjustment 恰有 basis、version、available_at、point_in_time_attested。basis 明确为 raw、split_adjusted 或 split_and_distribution_adjusted;可得时间不晚于 decision_at,point_in_time_attested=true。版本/ID 不得为空或使用 latest。不同复权口径不能混为同一可比实验;raw 拆分跳变仍是数据资格风险。不重新计算复权因子。 + +components 恰有 prices、calendar、adjustment,每个恰有 as_of、available_at、version。全部同一截止日并在 decision_at 前可得。日历/复权的版本与可得时间还须匹配对应声明;价格组件可得时间不早于其使用的行。版本指冻结 producer/解释口径,不是会重写的全文件 latest 身份。 + +每个 bar 恰有 symbol、date、close、closed、available_at、adjustment_available_at、adjustment_version。选中行标的/因子版本一致,价格有限且为正、非布尔,closed=true,发布不早于该日收盘与因子可得时点。重复或乱序日期拒绝;模块不自动重排或选择多个已可得修订。 + +资格明确标 qualified_by_declaration:只校验调用方声明及内部一致性。不证明供应商历史真实性、正式日历完整性或因子可得性。如果日历/价格同时遗漏一个交易日却声称完整,单靠本检查无法发现。真实 P0 证据与不可变 PIT 快照仍是必需条件。 + +## 因果投影与 hash + +采用按 as_of/decision_at 切片的政策: + +1. date>as_of 的行在读取其他列之前即不可见 +2. 历史日期的价格 available_at>decision_at 时行不可见,损坏或无关剩余列也不影响历史结果 +3. adjustment_available_at>decision_at 时该复权行也不进入切片 +4. 必需行不可得是覆盖缺失,不补价,不转成市场压力 +5. 日历中 as_of 后的交易日同样不可见 + +选择器本身无法解析就不能证明属于未来,产生 ROW_SELECTOR_INVALID。顶层/组件声明必须是该历史决策的冻结版本;重写声明不属于隐藏未来行追加。 + +input_sha256 使用原 QSP canonical JSON helper,只 hash 因果投影。隐藏行不进特征、校验、计数、理由或身份,追加后整个历史观察和 hash 不变。选中错误值只得到诊断错误类型身份,仍为 unknown,不成为合格证据。配置 hash 和 V2 payload hash 分别独立。 + +## 显式特征与配置 + +配置 schema 为 qsl.semiconductor-regime-config.research.v1。其余字段必须恰为: + +- sma_window_sessions≥2;sma_slope_lag_sessions≥1 +- path_window_returns≥2,需要步数+1 个收盘价 +- short_vol_window_returns/long_vol_window_returns≥2,短≤长 +- drawdown_window_sessions≥2;annualization_sessions≥1 +- ttl_seconds≥1;classification 明确为 null 或下述完整研究假设 + +没有数值默认值。窗口、TTL、阈值和比较政策必须先冻结再评估,不能从未来结果选参数。测试中的短窗口和数值仅是 SYNTHETIC_FIXTURE_ONLY,不是部署建议。 + +最少历史长度为 max(SMA窗口+斜率滞后、路径收益数+1、短收益数+1、长收益数+1、回撤交易日数)。 + +| 特征 | 定义 | +| --- | --- | +| close_to_sma_ratio | 最后收盘/当前尾部 SMA −1 | +| sma_slope_per_session | (当前 SMA/滞后 SMA −1)/滞后交易日数 | +| path_efficiency | 路径端点净变化绝对值/每步绝对变化之和 | +| short/long_realized_volatility_annualized | 算术收益总体标准差 × 年化交易日数平方根 | +| volatility_ratio | 短波动/长波动 | +| trailing_drawdown_ratio | 最后收盘/尾部窗口峰值 −1 | + +有限特征保留 12 位小数,分类用同一精度。零路径变化时 path_efficiency=null;长波动为零时 volatility_ratio=null。它们是市场特征定义域未知,不能自动判震荡或安全。计算非有限则数据资格失败并清空所有特征。 + +## 独立三轴与研究假设 + +classification=null 是正常的仅特征模式。数据声明合格仍给特征,三轴全为 unknown/CLASSIFICATION_NOT_CONFIGURED。 + +分类配置恰有 hypothesis_id、status=UNVALIDATED_RESEARCH_HYPOTHESIS、direction、trendiness、pressure。不接受 calibrated 或概率。 + +- 方向:显式正数 price_distance_min/sma_slope_min;价格位置与均线斜率一起过正阈值为 up,一起过负阈值为 down。显著反向冲突或阈值间隙均 unknown +- 趋势性:0≤range_efficiency_max bytes: + return canonical_json_bytes(value) + + +def _digest(value: Any) -> str: + return hashlib.sha256(_canonical(value)).hexdigest() + + +def _exact(value: Any, keys: set[str]) -> bool: + return isinstance(value, Mapping) and set(value) == keys + + +def _date(value: Any) -> str | None: + if not isinstance(value, str) or not _DATE.fullmatch(value): + return None + try: + return value if date.fromisoformat(value).isoformat() == value else None + except ValueError: + return None + + +def _time(value: Any) -> datetime | None: + if not isinstance(value, str) or "T" not in value: + return None + try: + parsed = datetime.fromisoformat(value.replace("Z", "+00:00")) + return parsed.astimezone(timezone.utc) if parsed.tzinfo is not None else None + except (ValueError, OverflowError): + return None + + +def _stamp(value: datetime | None) -> str | None: + return value.isoformat().replace("+00:00", "Z") if value is not None else None + + +def _version(value: Any) -> bool: + return isinstance(value, str) and bool(value.strip()) and "latest" not in value.casefold() + + +def _finite(value: Any, *, positive=False) -> bool: + if not isinstance(value, (int, float)) or isinstance(value, bool): + return False + try: + return math.isfinite(value) and (not positive or value > 0) + except OverflowError: + return False + + +def _validate_config(value: Any) -> dict: + if not _exact(value, _CONFIG_KEYS) or value["schema_version"] != CONFIG_VERSION: + raise ContractError("CONFIG_INVALID") + cfg = dict(value) + for name in _CONFIG_KEYS - {"schema_version", "classification"}: + if not isinstance(cfg[name], int) or isinstance(cfg[name], bool) or cfg[name] <= 0: + raise ContractError("CONFIG_INVALID") + if min(cfg["sma_window_sessions"], cfg["path_window_returns"], cfg["short_vol_window_returns"], + cfg["long_vol_window_returns"], cfg["drawdown_window_sessions"]) < 2: + raise ContractError("CONFIG_INVALID") + if cfg["short_vol_window_returns"] > cfg["long_vol_window_returns"]: + raise ContractError("CONFIG_INVALID") + classifier = cfg["classification"] + if classifier is not None: + if not _exact(classifier, {"hypothesis_id", "status", "direction", "trendiness", "pressure"}): + raise ContractError("CLASSIFICATION_CONFIG_INVALID") + if not _version(classifier["hypothesis_id"]) or classifier["status"] != "UNVALIDATED_RESEARCH_HYPOTHESIS": + raise ContractError("CLASSIFICATION_CONFIG_INVALID") + fields = { + "direction": {"price_distance_min", "sma_slope_min"}, + "trendiness": {"range_efficiency_max", "trend_efficiency_min"}, + "pressure": {"drawdown_normal_max", "drawdown_stress_min", "volatility_ratio_normal_max", "volatility_ratio_stress_min"}, + } + for axis, keys in fields.items(): + if not _exact(classifier[axis], keys) or not all(_finite(v) for v in classifier[axis].values()): + raise ContractError("CLASSIFICATION_CONFIG_INVALID") + d, t, p = (classifier[k] for k in ("direction", "trendiness", "pressure")) + if not (d["price_distance_min"] > 0 and d["sma_slope_min"] > 0 + and 0 <= t["range_efficiency_max"] < t["trend_efficiency_min"] <= 1 + and 0 <= p["drawdown_normal_max"] < p["drawdown_stress_min"] <= 1 + and 0 <= p["volatility_ratio_normal_max"] < p["volatility_ratio_stress_min"]): + raise ContractError("CLASSIFICATION_CONFIG_INVALID") + # Detached JSON-only config; no opaque values or undeclared knobs. + return json.loads(_canonical(cfg)) + + +def semiconductor_regime_observer_config_sha256(config: Mapping[str, Any]) -> str: + return _digest(_validate_config(config)) + + +def _safe_hash_value(value: Any) -> Any: + """Invalid data gets stable error identity; it never becomes usable evidence.""" + if value is None or isinstance(value, (str, bool)): + return value + if isinstance(value, (int, float)): + return value if _finite(value) else {"invalid_number": str(value)} + if isinstance(value, Mapping): + return {str(k): _safe_hash_value(v) for k, v in value.items()} + if isinstance(value, (list, tuple)): + return [_safe_hash_value(v) for v in value] + return {"invalid_type": type(value).__name__} + + +def _slice_rows(rows: Any, cutoff: str | None, decision: datetime | None, *, bars: bool) -> tuple[list, list[str]]: + if not isinstance(rows, Sequence) or isinstance(rows, (str, bytes, bytearray)): + return [], ["BARS_INVALID" if bars else "CALENDAR_SESSIONS_INVALID"] + selected, reasons = [], [] + for row in rows: + if not isinstance(row, Mapping) or _date(row.get("date")) is None or cutoff is None: + reasons.append("ROW_SELECTOR_INVALID") + continue + # Reject no payload fields before this causal date slice. Future payload + # can be malformed, contain NaN, or unknown keys and remains invisible. + if row["date"] > cutoff: + continue + if bars: + published = _time(row.get("available_at")) + if published is None or decision is None: + reasons.append("ROW_SELECTOR_INVALID") + continue + # A hidden late revision's other columns are irrelevant too. + if published > decision: + continue + adjusted = _time(row.get("adjustment_available_at")) + if adjusted is None: + reasons.append("ROW_SELECTOR_INVALID") + continue + if adjusted > decision: + continue + selected.append(dict(row)) + return selected, reasons + + +def _project(data: Any) -> tuple[dict, list[str]]: + if not isinstance(data, Mapping): + return {}, ["INPUT_FIELDS_INVALID"] + projected = dict(data) + cutoff, decision = _date(data.get("as_of")), _time(data.get("decision_at")) + selected, reasons = _slice_rows(data.get("bars"), cutoff, decision, bars=True) + projected["bars"] = selected + calendar = data.get("calendar") + if isinstance(calendar, Mapping): + calendar = dict(calendar) + sessions, errors = _slice_rows(calendar.get("sessions"), cutoff, decision, bars=False) + reasons.extend(errors) + calendar["sessions"] = sessions + projected["calendar"] = calendar + return projected, reasons + + +def _minimum(cfg: dict) -> int: + return max(cfg["sma_window_sessions"] + cfg["sma_slope_lag_sessions"], cfg["path_window_returns"] + 1, + cfg["short_vol_window_returns"] + 1, cfg["long_vol_window_returns"] + 1, + cfg["drawdown_window_sessions"]) + + +def _qualify(data: dict, cfg: dict, initial: list[str]) -> tuple[list[str], datetime | None]: + reasons = list(initial) + if not _exact(data, _INPUT_KEYS) or data.get("schema_version") != INPUT_VERSION: + reasons.append("INPUT_FIELDS_INVALID") + symbol, role = data.get("symbol"), data.get("series_role") + if not ((symbol == "SOXX" and role == "SIGNAL_PROXY") or (symbol == "SOXL" and role == "EXECUTION_BENCHMARK")): + reasons.append("SYMBOL_ROLE_UNSUPPORTED") + cutoff = _date(data.get("as_of")) + decision, available = _time(data.get("decision_at")), _time(data.get("available_at")) + if cutoff is None or decision is None or available is None: + reasons.append("OBSERVATION_TIMES_INVALID") + if decision is not None and available is not None and available > decision: + reasons.append("SNAPSHOT_UNAVAILABLE_AT_DECISION") + calendar, adjustment, components = (data.get(k) for k in ("calendar", "adjustment", "components")) + cal_keys = {"id", "version", "available_at", "sessions"} + adj_keys = {"basis", "version", "available_at", "point_in_time_attested"} + if not _exact(calendar, cal_keys): + reasons.append("CALENDAR_DECLARATION_INVALID") + calendar = calendar if isinstance(calendar, Mapping) else {} + if not _version(calendar.get("id")) or not _version(calendar.get("version")): + reasons.append("CALENDAR_ID_VERSION_REQUIRED") + cal_avail = _time(calendar.get("available_at")) + if cal_avail is None or decision is None or cal_avail > decision: + reasons.append("CALENDAR_UNAVAILABLE_AT_DECISION") + if not _exact(adjustment, adj_keys): + reasons.append("ADJUSTMENT_DECLARATION_INVALID") + adjustment = adjustment if isinstance(adjustment, Mapping) else {} + if adjustment.get("basis") not in {"raw", "split_adjusted", "split_and_distribution_adjusted"}: + reasons.append("ADJUSTMENT_BASIS_UNSUPPORTED") + if not _version(adjustment.get("version")): + reasons.append("ADJUSTMENT_VERSION_REQUIRED") + if adjustment.get("point_in_time_attested") is not True: + reasons.append("ADJUSTMENT_PIT_NOT_ATTESTED") + adj_avail = _time(adjustment.get("available_at")) + if adj_avail is None or decision is None or adj_avail > decision: + reasons.append("ADJUSTMENT_UNAVAILABLE_AT_DECISION") + if not _exact(components, {"prices", "calendar", "adjustment"}): + reasons.append("COMPONENTS_INVALID") + components = components if isinstance(components, Mapping) else {} + declared_availabilities = [cal_avail, adj_avail] + for name in ("prices", "calendar", "adjustment"): + component = components.get(name) + if not _exact(component, {"as_of", "available_at", "version"}): + reasons.append("COMPONENTS_INVALID") + continue + if component["as_of"] != cutoff: + reasons.append("COMPONENT_AS_OF_MISMATCH") + if not _version(component["version"]): + reasons.append("COMPONENT_VERSION_INVALID") + comp_avail = _time(component["available_at"]) + declared_availabilities.append(comp_avail) + if comp_avail is None or decision is None or comp_avail > decision: + reasons.append("COMPONENT_UNAVAILABLE_AT_DECISION") + if name != "prices": + metadata = calendar if name == "calendar" else adjustment + if component["version"] != metadata.get("version") or comp_avail != _time(metadata.get("available_at")): + reasons.append("COMPONENT_VERSION_MISMATCH") + sessions = calendar.get("sessions", []) + sessions = sessions if isinstance(sessions, list) else [] + dates, closes = [], {} + for session in sessions: + if not _exact(session, {"date", "close_at", "complete"}): + reasons.append("CALENDAR_SESSION_FIELDS_INVALID") + session_date = _date(session.get("date")) + if session_date is None: + reasons.append("CALENDAR_DATE_INVALID") + continue + if dates and session_date <= dates[-1]: + reasons.append("CALENDAR_DATES_NOT_STRICTLY_INCREASING") + dates.append(session_date) + close = _time(session.get("close_at")) + closes[session_date] = close + if close is None or close.date().isoformat() != session_date: + reasons.append("SESSION_CLOSE_TIME_INVALID") + if close is not None and (decision is None or close > decision): + reasons.append("SESSION_NOT_CLOSED_AT_DECISION") + if session.get("complete") is not True: + reasons.append("SESSION_NOT_COMPLETE") + if len(dates) < _minimum(cfg): + reasons.append("INSUFFICIENT_CALENDAR_HISTORY") + if not dates or dates[-1] != cutoff: + reasons.append("CALENDAR_AS_OF_MISMATCH") + bars, bar_dates = data.get("bars", []), [] + bars = bars if isinstance(bars, list) else [] + row_availabilities = [] + for bar in bars: + if not _exact(bar, {"symbol", "date", "close", "closed", "available_at", "adjustment_available_at", "adjustment_version"}): + reasons.append("BAR_FIELDS_INVALID") + session_date = _date(bar.get("date")) + if bar_dates and session_date <= bar_dates[-1]: + reasons.append("BAR_DATES_NOT_STRICTLY_INCREASING") + bar_dates.append(session_date) + if bar.get("symbol") != symbol: + reasons.append("BAR_SYMBOL_MISMATCH") + if not _finite(bar.get("close"), positive=True): + reasons.append("CLOSE_INVALID") + if bar.get("closed") is not True: + reasons.append("BAR_NOT_CLOSED") + if bar.get("adjustment_version") != adjustment.get("version"): + reasons.append("BAR_ADJUSTMENT_VERSION_MISMATCH") + pub, adj = _time(bar.get("available_at")), _time(bar.get("adjustment_available_at")) + row_availabilities.extend((pub, adj)) + close = closes.get(session_date) + if session_date not in closes: + reasons.append("BAR_NOT_IN_CALENDAR") + if pub is not None and close is not None and pub < close: + reasons.append("BAR_AVAILABLE_BEFORE_CLOSE") + if pub is not None and adj is not None and pub < adj: + reasons.append("BAR_AVAILABLE_BEFORE_ADJUSTMENT") + needed_dates = dates[-_minimum(cfg):] + if len(bars) < _minimum(cfg) or not set(needed_dates).issubset(bar_dates) or (bar_dates and bar_dates[-1] != cutoff): + reasons.append("SESSION_COVERAGE_INCOMPLETE") + if available is not None and any(item is not None and item > available for item in declared_availabilities + row_availabilities): + reasons.append("SNAPSHOT_AVAILABILITY_PRECEDES_INPUT") + prices_component = components.get("prices", {}) + prices_avail = _time(prices_component.get("available_at")) if isinstance(prices_component, Mapping) else None + if prices_avail is not None and any(item is not None and item > prices_avail for item in row_availabilities): + reasons.append("PRICE_COMPONENT_AVAILABILITY_PRECEDES_BARS") + latest_close = closes.get(cutoff) + try: + valid_until = latest_close + timedelta(seconds=cfg["ttl_seconds"]) if latest_close is not None else None + except OverflowError: + valid_until = None + reasons.append("TTL_INVALID_FOR_SESSION") + if decision is not None and valid_until is not None and decision >= valid_until: + reasons.append("OBSERVATION_EXPIRED") + return sorted(set(reasons)), valid_until + + +def _features(bars: list, cfg: dict) -> dict: + # Shared ratios/volatility/drawdown follow pinned QSP QQQ observer formulas. + close = [float(row["close"]) for row in bars] + m, lag = cfg["sma_window_sessions"], cfg["sma_slope_lag_sessions"] + mean = math.fsum(close[-m:]) / m + past_mean = math.fsum(close[-m-lag:-lag]) / m + returns = [later / earlier - 1 for earlier, later in zip(close, close[1:])] + + def volatility(window): + values = returns[-window:] + average = math.fsum(values) / window + variance = math.fsum((v - average) ** 2 for v in values) / window + return math.sqrt(variance * cfg["annualization_sessions"]) + + short, long = volatility(cfg["short_vol_window_returns"]), volatility(cfg["long_vol_window_returns"]) + path = close[-cfg["path_window_returns"]-1:] + distance = math.fsum(abs(later - earlier) for earlier, later in zip(path, path[1:])) + values = { + "close_to_sma_ratio": close[-1] / mean - 1, + "sma_slope_per_session": (mean / past_mean - 1) / lag, + "path_efficiency": abs(path[-1] - path[0]) / distance if distance else None, + "short_realized_volatility_annualized": short, + "long_realized_volatility_annualized": long, + "volatility_ratio": short / long if long else None, + "trailing_drawdown_ratio": close[-1] / max(close[-cfg["drawdown_window_sessions"]:]) - 1, + } + if any(v is not None and not math.isfinite(v) for v in values.values()): + raise ArithmeticError("FEATURE_NONFINITE") + return {key: (None if value is None else (0.0 if round(value, 12) == 0 else round(value, 12))) for key, value in values.items()} + + +def _axis(state="unknown", reason="CLASSIFICATION_NOT_CONFIGURED") -> dict: + return {"state": state, "reason_codes": [reason]} + + +def _classify(facts: dict, classifier: dict | None) -> dict: + axes = {key: _axis() for key in ("direction", "trendiness", "pressure")} + if classifier is None: + return axes + d, t, p = (classifier[k] for k in ("direction", "trendiness", "pressure")) + price, slope = facts["close_to_sma_ratio"], facts["sma_slope_per_session"] + if price >= d["price_distance_min"] and slope >= d["sma_slope_min"]: + axes["direction"] = _axis("up", "DIRECTION_FEATURES_ALIGNED") + elif price <= -d["price_distance_min"] and slope <= -d["sma_slope_min"]: + axes["direction"] = _axis("down", "DIRECTION_FEATURES_ALIGNED") + else: + conflict = ((price >= d["price_distance_min"] and slope <= -d["sma_slope_min"]) + or (price <= -d["price_distance_min"] and slope >= d["sma_slope_min"])) + axes["direction"] = _axis(reason="DIRECTION_FEATURE_CONFLICT" if conflict else "DIRECTION_THRESHOLD_GAP") + efficiency = facts["path_efficiency"] + if efficiency is None: + axes["trendiness"] = _axis(reason="ZERO_PATH_VARIATION") + elif efficiency >= t["trend_efficiency_min"]: + axes["trendiness"] = _axis("trend_like", "PATH_EFFICIENCY_ABOVE_RESEARCH_THRESHOLD") + elif efficiency <= t["range_efficiency_max"]: + axes["trendiness"] = _axis("range_like", "PATH_EFFICIENCY_BELOW_RESEARCH_THRESHOLD") + else: + axes["trendiness"] = _axis(reason="TRENDINESS_THRESHOLD_GAP") + drawdown, ratio = -facts["trailing_drawdown_ratio"], facts["volatility_ratio"] + if ratio is None: + axes["pressure"] = _axis(reason="ZERO_LONG_VOLATILITY") + elif drawdown >= p["drawdown_stress_min"] and ratio >= p["volatility_ratio_stress_min"]: + axes["pressure"] = _axis("stressed", "PRESSURE_FEATURES_ALIGNED") + elif drawdown <= p["drawdown_normal_max"] and ratio <= p["volatility_ratio_normal_max"]: + axes["pressure"] = _axis("normal", "PRESSURE_FEATURES_ALIGNED") + else: + conflict = ((drawdown >= p["drawdown_stress_min"] and ratio <= p["volatility_ratio_normal_max"]) + or (drawdown <= p["drawdown_normal_max"] and ratio >= p["volatility_ratio_stress_min"])) + axes["pressure"] = _axis(reason="PRESSURE_FEATURE_CONFLICT" if conflict else "PRESSURE_THRESHOLD_GAP") + return axes + + +def build_semiconductor_regime_observation(snapshot: Mapping[str, Any], config: Mapping[str, Any]) -> dict: + """Describe one causal observation; missing/invalid input becomes unknown. + + Config is mandatory and has no numeric defaults. Thresholds may be None. + Hidden rows never enter features, validation, counters, or input identity. + """ + cfg = _validate_config(config) + data, selectors = _project(snapshot) + reasons, expiry = _qualify(data, cfg, selectors) + facts = dict.fromkeys(_FEATURE_KEYS) + if not reasons: + try: + facts = _features(data["bars"], cfg) + except (ArithmeticError, ValueError): + reasons = ["FEATURE_COMPUTATION_UNDEFINED"] + axes = ({key: _axis(reason="INPUT_QUALIFICATION_FAILED") for key in ("direction", "trendiness", "pressure")} + if reasons else _classify(facts, cfg["classification"])) + decision, input_available = _time(data.get("decision_at")), _time(data.get("available_at")) + available_candidates = [v for v in (decision, input_available) if v is not None] + calendar, adjustment, components = (data.get(k) for k in ("calendar", "adjustment", "components")) + calendar = calendar if isinstance(calendar, Mapping) else {} + adjustment = adjustment if isinstance(adjustment, Mapping) else {} + components = components if isinstance(components, Mapping) else {} + prices = components.get("prices", {}) + prices = prices if isinstance(prices, Mapping) else {} + + def declared_text(value): + return value if isinstance(value, str) else None + + return { + "schema_version": OBSERVATION_VERSION, + "research_status": "UNVALIDATED_RESEARCH", + "symbol": data.get("symbol") if isinstance(data.get("symbol"), str) else None, + "series_role": data.get("series_role") if isinstance(data.get("series_role"), str) else None, + "as_of": _date(data.get("as_of")), + "input_available_at": _stamp(input_available), + # This reference is observed at decision_at. A real producer must set + # that instant no earlier than its actual completed publication time. + "available_at": _stamp(max(available_candidates)) if available_candidates else None, + "decision_at": _stamp(decision), + "valid_until": _stamp(expiry), + "versions": {"feature": FEATURE_VERSION, "configuration_sha256": _digest(cfg)}, + "input_context": { + "calendar_id": declared_text(calendar.get("id")), + "calendar_version": declared_text(calendar.get("version")), + "adjustment_basis": declared_text(adjustment.get("basis")), + "adjustment_version": declared_text(adjustment.get("version")), + "prices_producer_version": declared_text(prices.get("version")), + }, + "input_sha256": _digest(_safe_hash_value(data)), + "features": facts, + "axes": axes, + "quality": { + "status": "unknown" if reasons else "qualified_by_declaration", + "reason_codes": reasons, + "minimum_required_sessions": _minimum(cfg), + "observed_sessions": len(data.get("bars", [])), + "assurance": "CALLER_DECLARATIONS_AND_CONSISTENCY_ONLY", + }, + } + + +def semiconductor_regime_observation_usable_at(observation: Mapping[str, Any], at: str) -> bool: + """Fresh evidence check only, never an instruction or trading permission.""" + if not isinstance(observation, Mapping) or not isinstance(observation.get("quality"), Mapping): + return False + when, decision, available, expiry = (_time(v) for v in ( + at, observation.get("decision_at"), observation.get("available_at"), observation.get("valid_until"))) + return bool(observation.get("schema_version") == OBSERVATION_VERSION + and observation.get("quality", {}).get("status") == "qualified_by_declaration" + and all(v is not None for v in (when, decision, available, expiry)) + and max(decision, available) <= when < expiry) + + +def build_semiconductor_regime_signal_v2( + *, + snapshot: Mapping[str, Any], + config: Mapping[str, Any], + producer: Mapping[str, Any], + input_provenance: Mapping[str, Any], +) -> dict: + """Build a research-only V2 observation with the existing pure helper. + + The P1 manifest must already describe this causal projection. This checks + declared identity and consistency, never provider truth or P1 qualification. + Nothing registers, enables, routes, publishes, or authorizes this signal. + """ + cfg = _validate_config(config) + observation = build_semiconductor_regime_observation(snapshot, cfg) + if (not isinstance(producer, Mapping) + or producer.get("repo") != REPOSITORY + or producer.get("entrypoint") != ENTRYPOINT + or producer.get("config_sha256") != _digest(cfg) + or not _exact(input_provenance, {"p1_manifest_sha256", "input_root_sha256", "date_cutoff"}) + or not isinstance(input_provenance["p1_manifest_sha256"], str) + or not _SHA256.fullmatch(input_provenance["p1_manifest_sha256"]) + or input_provenance["input_root_sha256"] != observation["input_sha256"] + or input_provenance["date_cutoff"] != observation["as_of"] + or observation["as_of"] is None): + raise ContractError("QSP_BINDING_INVALID") + return build_signal_envelope( + plugin_id=PLUGIN_ID, + producer=producer, + input_provenance=input_provenance, + payload=observation, + ) diff --git a/tests/test_semiconductor_regime_observer_research.py b/tests/test_semiconductor_regime_observer_research.py new file mode 100644 index 0000000..3d701fa --- /dev/null +++ b/tests/test_semiconductor_regime_observer_research.py @@ -0,0 +1,435 @@ +"""Standard-library synthetic contract tests, never market-performance tests.""" + +from __future__ import annotations + +import copy +import hashlib +import json +from datetime import date, timedelta +from pathlib import Path +import unittest + +from quant_strategy_plugins.semiconductor_regime_observer_research import ( + CONFIG_VERSION, + INPUT_VERSION, + ContractError, + build_semiconductor_regime_observation as build_observation, + build_semiconductor_regime_signal_v2 as build_qsp_signal, + semiconductor_regime_observer_config_sha256 as config_sha256, + semiconductor_regime_observation_usable_at as usable_at, + ENTRYPOINT, + REPOSITORY, +) + + +from quant_strategy_plugins import semiconductor_regime_observer_research as candidate +from quant_strategy_plugins import qqq_price_regime_observer_v2 as qqq_observer +from quant_strategy_plugins.plugin_signal_envelope_v2 import validate_signal_envelope + + +HERE = Path(__file__).resolve().parent + + +def config(*, classify: bool = False) -> dict: + """These small values are test fixtures; none is an operational default.""" + value = { + "schema_version": CONFIG_VERSION, + "sma_window_sessions": 3, + "sma_slope_lag_sessions": 2, + "path_window_returns": 5, + "short_vol_window_returns": 2, + "long_vol_window_returns": 5, + "drawdown_window_sessions": 6, + "annualization_sessions": 252, + "ttl_seconds": 86400, + "classification": None, + } + if classify: + value["classification"] = { + "hypothesis_id": "SYNTHETIC_FIXTURE_ONLY", + "status": "UNVALIDATED_RESEARCH_HYPOTHESIS", + "direction": {"price_distance_min": 0.001, "sma_slope_min": 0.001}, + "trendiness": {"range_efficiency_max": 0.2, "trend_efficiency_min": 0.8}, + "pressure": { + "drawdown_normal_max": 0.03, + "drawdown_stress_min": 0.1, + "volatility_ratio_normal_max": 1.1, + "volatility_ratio_stress_min": 1.5, + }, + } + return value + + +def stamp(session: str, hour: int = 20, minute: int = 1) -> str: + return f"{session}T{hour:02}:{minute:02}:00Z" + + +def snapshot(prices=None, *, symbol="SOXX") -> dict: + prices = prices if prices is not None else [100, 102, 104, 107, 109, 112] + start = date(2026, 9, 21) + sessions = [(start + timedelta(days=i)).isoformat() for i in (0, 1, 2, 3, 4, 7)] + as_of = sessions[-1] + availability = stamp(as_of) + adjustment_availability = stamp(sessions[0], 12, 0) + return { + "schema_version": INPUT_VERSION, + "symbol": symbol, + "series_role": "SIGNAL_PROXY" if symbol == "SOXX" else "EXECUTION_BENCHMARK", + "as_of": as_of, + "available_at": availability, + "decision_at": stamp(as_of, 21, 0), + "calendar": { + "id": "SYNTHETIC_XNYS", + "version": "synthetic-calendar-v1", + "available_at": stamp(sessions[0], 12, 0), + "sessions": [{"date": d, "close_at": stamp(d, 20, 0), "complete": True} for d in sessions], + }, + "adjustment": { + "basis": "split_and_distribution_adjusted", + "version": "synthetic-adjustment-v1", + "available_at": adjustment_availability, + "point_in_time_attested": True, + }, + "components": { + name: {"as_of": as_of, "available_at": available, "version": version} + for name, available, version in ( + ("prices", availability, "synthetic-prices-producer-v1"), + ("calendar", stamp(sessions[0], 12, 0), "synthetic-calendar-v1"), + ("adjustment", adjustment_availability, "synthetic-adjustment-v1"), + ) + }, + "bars": [ + { + "symbol": symbol, + "date": d, + "close": p, + "closed": True, + "available_at": stamp(d), + "adjustment_available_at": adjustment_availability, + "adjustment_version": "synthetic-adjustment-v1", + } + for d, p in zip(sessions, prices) + ], + } + + +class ContractTests(unittest.TestCase): + def assert_unknown_quality(self, data, reason): + observed = build_observation(data, config(classify=True)) + self.assertEqual(observed["quality"]["status"], "unknown") + self.assertIn(reason, observed["quality"]["reason_codes"]) + self.assertTrue(all(axis["state"] == "unknown" for axis in observed["axes"].values())) + self.assertTrue(all(v is None for v in observed["features"].values())) + self.assertNotIn("risk_off", json.dumps(observed)) + return observed + + def test_no_classification_is_the_normal_feature_only_mode(self): + observed = build_observation(snapshot(), config()) + self.assertEqual(observed["quality"]["status"], "qualified_by_declaration") + self.assertGreater(observed["features"]["sma_slope_per_session"], 0) + self.assertEqual(observed["features"]["path_efficiency"], 1) + for axis in observed["axes"].values(): + self.assertEqual(axis, {"state": "unknown", "reason_codes": ["CLASSIFICATION_NOT_CONFIGURED"]}) + + def test_synthetic_direction_and_trendiness_are_independent_axes(self): + up = build_observation(snapshot(), config(classify=True)) + down = build_observation(snapshot([112, 109, 107, 104, 102, 100]), config(classify=True)) + self.assertEqual(up["axes"]["direction"]["state"], "up") + self.assertEqual(down["axes"]["direction"]["state"], "down") + self.assertEqual(up["axes"]["trendiness"]["state"], "trend_like") + alternating = build_observation(snapshot([100, 103, 100, 103, 100, 101]), config(classify=True)) + self.assertEqual(alternating["axes"]["trendiness"]["state"], "range_like") + + def test_market_pressure_classifications_use_only_market_features(self): + normal = build_observation(snapshot([100, 100, 103, 103, 103, 103]), config(classify=True)) + stressed = build_observation(snapshot([100, 100, 100, 100, 130, 80]), config(classify=True)) + self.assertEqual(normal["axes"]["pressure"]["state"], "normal") + self.assertEqual(stressed["axes"]["pressure"]["state"], "stressed") + self.assertEqual(stressed["quality"]["status"], "qualified_by_declaration") + + def test_future_append_with_unusable_payload_is_completely_invisible(self): + data = snapshot() + before = build_observation(data, config(classify=True)) + after_data = copy.deepcopy(data) + future_date = "2026-09-29" + after_data["bars"].extend([ + {"date": future_date, "available_at": stamp(future_date), "close": float("nan"), "order": "ignored"}, + {"date": "2030-01-01", "available_at": "malformed", "close": object()}, + ]) + after_data["calendar"]["sessions"].append({"date": future_date, "close_at": "not-a-time"}) + after = build_observation(after_data, config(classify=True)) + self.assertEqual(before, after) + self.assertEqual(before["input_sha256"], after["input_sha256"]) + + def test_unavailable_historical_correction_does_not_change_identity(self): + data = snapshot() + before = build_observation(data, config()) + correction = copy.deepcopy(data["bars"][-1]) + correction["available_at"] = "2026-09-29T20:01:00Z" + correction["close"] = 1e9 + data["bars"].append(correction) + self.assertEqual(before, build_observation(data, config())) + # Even absent/unrelated columns on that hidden revision must not leak. + data["bars"].append({"date": data["as_of"], "available_at": "2030-01-01T00:00:00Z", "garbage": object()}) + self.assertEqual(before, build_observation(data, config())) + + def test_observation_availability_is_not_backdated_to_raw_input(self): + observed = build_observation(snapshot(), config()) + self.assertEqual(observed["input_available_at"], "2026-09-28T20:01:00Z") + self.assertEqual(observed["available_at"], observed["decision_at"]) + self.assertFalse(usable_at(observed, "2026-09-28T20:01:00Z")) + self.assertTrue(usable_at(observed, observed["available_at"])) + + def test_late_required_bar_is_missing_at_historical_decision(self): + data = snapshot() + data["bars"][-1]["available_at"] = "2026-09-29T20:01:00Z" + self.assert_unknown_quality(data, "SESSION_COVERAGE_INCOMPLETE") + + def test_snapshot_itself_not_available_at_decision(self): + data = snapshot() + data["available_at"] = "2026-09-29T20:01:00Z" + self.assert_unknown_quality(data, "SNAPSHOT_UNAVAILABLE_AT_DECISION") + + def test_unclosed_and_incomplete_sessions_are_not_market_pressure(self): + data = snapshot() + data["bars"][-1]["closed"] = False + self.assert_unknown_quality(data, "BAR_NOT_CLOSED") + data = snapshot() + data["calendar"]["sessions"][-1]["complete"] = False + self.assert_unknown_quality(data, "SESSION_NOT_COMPLETE") + data = snapshot() + data["decision_at"] = "2026-09-28T19:00:00Z" + self.assert_unknown_quality(data, "SESSION_NOT_CLOSED_AT_DECISION") + + def test_required_calendar_gap_and_missing_bar(self): + data = snapshot() + del data["bars"][2] + self.assert_unknown_quality(data, "SESSION_COVERAGE_INCOMPLETE") + data = snapshot() + del data["calendar"]["sessions"][2] + self.assert_unknown_quality(data, "INSUFFICIENT_CALENDAR_HISTORY") + + def test_adjustment_attestation_and_availability_are_required(self): + for field, value, reason in ( + ("point_in_time_attested", False, "ADJUSTMENT_PIT_NOT_ATTESTED"), + ("basis", "unspecified", "ADJUSTMENT_BASIS_UNSUPPORTED"), + ("available_at", "2026-09-29T00:00:00Z", "ADJUSTMENT_UNAVAILABLE_AT_DECISION"), + ): + data = snapshot() + data["adjustment"][field] = value + self.assert_unknown_quality(data, reason) + data = snapshot() + data["bars"][-1]["adjustment_available_at"] = "2026-09-29T00:00:00Z" + self.assert_unknown_quality(data, "SESSION_COVERAGE_INCOMPLETE") + + def test_component_dates_and_versions_must_bind(self): + data = snapshot() + data["components"]["prices"]["as_of"] = "2026-09-25" + self.assert_unknown_quality(data, "COMPONENT_AS_OF_MISMATCH") + data = snapshot() + data["components"]["calendar"]["version"] = "different-calendar" + self.assert_unknown_quality(data, "COMPONENT_VERSION_MISMATCH") + data = snapshot() + data["components"]["adjustment"]["available_at"] = "2026-09-29T00:00:00Z" + self.assert_unknown_quality(data, "COMPONENT_UNAVAILABLE_AT_DECISION") + + def test_nonfinite_nonpositive_bool_and_missing_values_fail_closed(self): + for value in (float("nan"), float("inf"), -float("inf"), 0, -1, True, None, "100"): + with self.subTest(value=repr(value)): + data = snapshot() + data["bars"][2]["close"] = value + self.assert_unknown_quality(data, "CLOSE_INVALID") + + def test_duplicate_unsorted_dates_and_symbol_mismatch_fail_closed(self): + data = snapshot() + data["bars"].insert(2, copy.deepcopy(data["bars"][1])) + self.assert_unknown_quality(data, "BAR_DATES_NOT_STRICTLY_INCREASING") + data = snapshot() + data["bars"][1], data["bars"][2] = data["bars"][2], data["bars"][1] + self.assert_unknown_quality(data, "BAR_DATES_NOT_STRICTLY_INCREASING") + data = snapshot() + data["bars"][0]["symbol"] = "QQQ" + self.assert_unknown_quality(data, "BAR_SYMBOL_MISMATCH") + + def test_calendar_duplicates_and_disordered_dates_fail_closed(self): + data = snapshot() + data["calendar"]["sessions"].insert(1, copy.deepcopy(data["calendar"]["sessions"][0])) + self.assert_unknown_quality(data, "CALENDAR_DATES_NOT_STRICTLY_INCREASING") + data = snapshot() + data["calendar"]["sessions"].reverse() + self.assert_unknown_quality(data, "CALENDAR_DATES_NOT_STRICTLY_INCREASING") + + def test_calendar_identity_and_explicit_timezone_are_required(self): + data = snapshot() + data["calendar"]["version"] = "latest" + self.assert_unknown_quality(data, "CALENDAR_ID_VERSION_REQUIRED") + data = snapshot() + data["decision_at"] = "2026-09-28T21:00:00" + self.assert_unknown_quality(data, "OBSERVATION_TIMES_INVALID") + + def test_snapshot_cannot_claim_availability_before_a_used_component(self): + data = snapshot() + data["available_at"] = "2026-09-28T20:00:00Z" + self.assert_unknown_quality(data, "SNAPSHOT_AVAILABILITY_PRECEDES_INPUT") + data = snapshot() + data["bars"][-1]["available_at"] = "2026-09-28T19:00:00Z" + self.assert_unknown_quality(data, "BAR_AVAILABLE_BEFORE_CLOSE") + + def test_missing_nonnumeric_identity_does_not_crash_unknown_path(self): + self.assert_unknown_quality(None, "INPUT_FIELDS_INVALID") + for value in (None, [], {}, True): + data = snapshot() + data["symbol"] = value + self.assert_unknown_quality(data, "SYMBOL_ROLE_UNSUPPORTED") + self.assertFalse(usable_at({"quality": None}, "2026-09-28T21:00:00Z")) + + def test_finite_extreme_prices_cannot_emit_nonfinite_features(self): + self.assert_unknown_quality(snapshot([1e-300, 1e308, 1e-300, 1e308, 1e-300, 1e308]), "FEATURE_COMPUTATION_UNDEFINED") + + def test_zero_variation_is_qualified_data_with_undefined_market_features(self): + observed = build_observation(snapshot([100] * 6), config(classify=True)) + self.assertEqual(observed["quality"]["status"], "qualified_by_declaration") + self.assertEqual(observed["features"]["long_realized_volatility_annualized"], 0) + self.assertIsNone(observed["features"]["volatility_ratio"]) + self.assertIsNone(observed["features"]["path_efficiency"]) + self.assertEqual(observed["axes"]["pressure"]["reason_codes"], ["ZERO_LONG_VOLATILITY"]) + self.assertEqual(observed["axes"]["trendiness"]["reason_codes"], ["ZERO_PATH_VARIATION"]) + + def test_conflicting_direction_features_return_unknown_not_a_vote(self): + observed = build_observation(snapshot([100, 120, 120, 110, 110, 115]), config(classify=True)) + self.assertGreater(observed["features"]["close_to_sma_ratio"], 0) + self.assertLess(observed["features"]["sma_slope_per_session"], 0) + self.assertEqual(observed["axes"]["direction"], {"state": "unknown", "reason_codes": ["DIRECTION_FEATURE_CONFLICT"]}) + + def test_conflicting_pressure_features_return_unknown_without_data_failure(self): + observed = build_observation(snapshot([120, 115, 100, 100, 100, 100]), config(classify=True)) + self.assertEqual(observed["quality"]["status"], "qualified_by_declaration") + self.assertEqual(observed["axes"]["pressure"], {"state": "unknown", "reason_codes": ["PRESSURE_FEATURE_CONFLICT"]}) + + def test_ttl_boundary_and_late_replay_never_refresh_old_signal(self): + observed = build_observation(snapshot(), config(classify=True)) + self.assertEqual(observed["valid_until"], "2026-09-29T20:00:00Z") + self.assertTrue(usable_at(observed, "2026-09-29T19:59:59Z")) + self.assertFalse(usable_at(observed, "2026-09-29T20:00:00Z")) + data = snapshot() + data["decision_at"] = "2026-10-01T20:00:00Z" + replayed = self.assert_unknown_quality(data, "OBSERVATION_EXPIRED") + self.assertEqual(replayed["valid_until"], observed["valid_until"]) + + def test_no_lookback_or_classifier_production_defaults(self): + with self.assertRaises(ContractError): + build_observation(snapshot(), {}) + bad = config(classify=True) + bad["classification"]["status"] = "CALIBRATED" + with self.assertRaises(ContractError): + build_observation(snapshot(), bad) + bad = config() + bad["target_weight"] = 1.0 + with self.assertRaises(ContractError): + build_observation(snapshot(), bad) + for invalid in (float("nan"), float("inf"), True, "0.1"): + bad = config(classify=True) + bad["classification"]["direction"]["sma_slope_min"] = invalid + with self.assertRaises(ContractError): + build_observation(snapshot(), bad) + + def test_missing_input_evidence_and_unknown_control_fields_fail_closed(self): + for key in ("calendar", "adjustment", "available_at", "components", "decision_at"): + data = snapshot() + del data[key] + self.assert_unknown_quality(data, "INPUT_FIELDS_INVALID") + data = snapshot() + data["authorization"] = True + self.assert_unknown_quality(data, "INPUT_FIELDS_INVALID") + + def test_qqq_cannot_substitute_for_semiconductor_signal(self): + self.assert_unknown_quality(snapshot(symbol="QQQ"), "SYMBOL_ROLE_UNSUPPORTED") + soxl = build_observation(snapshot(symbol="SOXL"), config()) + self.assertEqual(soxl["series_role"], "EXECUTION_BENCHMARK") + data = snapshot(symbol="SOXX") + data["series_role"] = "EXECUTION_BENCHMARK" + self.assert_unknown_quality(data, "SYMBOL_ROLE_UNSUPPORTED") + + def test_feature_calculation_matches_existing_qqq_observer_facts(self): + observer = qqq_observer + data = snapshot() + existing = observer.build_qqq_price_regime_observation( + qqq_bars=[{"date": b["date"], "close": b["close"]} for b in data["bars"]], + as_of=data["as_of"], + config={ + "schema_version": observer.CONFIG_SCHEMA_VERSION, "symbol": "QQQ", + "trend_window_sessions": 3, "short_realized_volatility_window_sessions": 2, + "long_realized_volatility_window_sessions": 5, "drawdown_window_sessions": 6, + "annualization_sessions": 252, + }, + ) + features = build_observation(data, config())["features"] + for ours, theirs in ( + ("close_to_sma_ratio", "close_to_trend_mean_ratio"), + ("short_realized_volatility_annualized", "short_realized_volatility_annualized"), + ("long_realized_volatility_annualized", "long_realized_volatility_annualized"), + ("trailing_drawdown_ratio", "trailing_drawdown_ratio"), + ): + self.assertAlmostEqual(features[ours], existing["facts"][theirs], places=11) + + def test_existing_qsp_v2_builder_accepts_observation_without_control_fields(self): + observed = build_observation(snapshot(), config()) + producer = { + "repo": REPOSITORY, + # Synthetic immutable fixture, not a published candidate revision. + "revision": "a" * 40, + "entrypoint": ENTRYPOINT, + "code_sha256": hashlib.sha256(Path(candidate.__file__).read_bytes()).hexdigest(), + "config_sha256": config_sha256(config()), + } + provenance = {"p1_manifest_sha256": "b" * 64, "input_root_sha256": observed["input_sha256"], "date_cutoff": observed["as_of"]} + envelope = build_qsp_signal(snapshot=snapshot(), config=config(), producer=producer, input_provenance=provenance) + self.assertEqual(validate_signal_envelope(envelope), envelope) + self.assertEqual(envelope["input"]["input_root_sha256"], observed["input_sha256"]) + serialized = json.dumps(envelope) + for forbidden in ("target_weight", "capital", "order", "authorization", "existingV2"): + self.assertNotIn(forbidden, serialized) + bad_producer = {**producer, "config_sha256": "0" * 64} + with self.assertRaises(ContractError): + build_qsp_signal(snapshot=snapshot(), config=config(), producer=bad_producer, input_provenance=provenance) + + + def test_research_entrypoint_rejects_wrong_producer_or_causal_root(self): + observed = build_observation(snapshot(), config()) + producer = { + "repo": REPOSITORY, "revision": "a" * 40, "entrypoint": ENTRYPOINT, + "code_sha256": "c" * 64, "config_sha256": config_sha256(config()), + } + provenance = {"p1_manifest_sha256": "b" * 64, "input_root_sha256": observed["input_sha256"], "date_cutoff": observed["as_of"]} + for field, value in (("repo", "Example/OtherRepo"), ("entrypoint", "other_module:build_signal")): + with self.subTest(field=field), self.assertRaises(ContractError): + build_qsp_signal(snapshot=snapshot(), config=config(), producer={**producer, field: value}, input_provenance=provenance) + for field, value in (("input_root_sha256", "0" * 64), ("date_cutoff", "2026-09-25"), ("p1_manifest_sha256", "")): + with self.subTest(field=field), self.assertRaises(ContractError): + build_qsp_signal(snapshot=snapshot(), config=config(), producer=producer, input_provenance={**provenance, field: value}) + with self.assertRaises(ContractError): + build_qsp_signal(snapshot=snapshot(), config=config(), producer=producer, input_provenance={**provenance, "authorization": True}) + + def test_pure_repeatability_and_no_input_mutation(self): + data, cfg = snapshot(), config(classify=True) + original = copy.deepcopy((data, cfg)) + first = build_observation(data, cfg) + self.assertEqual(first, build_observation(data, cfg)) + self.assertEqual((data, cfg), original) + json.dumps(first, allow_nan=False) + + def test_price_scale_changes_input_identity_but_not_dimensionless_evidence(self): + original = snapshot() + observed = build_observation(original, config(classify=True)) + for multiplier in (0.1, 10.0, 1000.0): + data = copy.deepcopy(original) + for bar in data["bars"]: + bar["close"] *= multiplier + scaled = build_observation(data, config(classify=True)) + self.assertEqual(observed["features"], scaled["features"]) + self.assertEqual(observed["axes"], scaled["axes"]) + self.assertNotEqual(observed["input_sha256"], scaled["input_sha256"]) + + +if __name__ == "__main__": + unittest.main(verbosity=2)