From f8dc7e6ad6c37b9cabcd721d7d6345b5d7852725 Mon Sep 17 00:00:00 2001 From: Pigbibi <20649888+Pigbibi@users.noreply.github.com> Date: Tue, 6 Oct 2026 16:11:11 +0800 Subject: [PATCH] feat(research): add source-aware ten-year rates observations --- docs/rates-context-observer-research.md | 74 ++++ docs/rates-context-observer-research.zh-CN.md | 74 ++++ .../rates_context_observer_research.py | 235 ++++++++++++ tests/test_rates_context_observer_research.py | 337 ++++++++++++++++++ 4 files changed, 720 insertions(+) create mode 100644 docs/rates-context-observer-research.md create mode 100644 docs/rates-context-observer-research.zh-CN.md create mode 100644 src/quant_strategy_plugins/rates_context_observer_research.py create mode 100644 tests/test_rates_context_observer_research.py diff --git a/docs/rates-context-observer-research.md b/docs/rates-context-observer-research.md new file mode 100644 index 0000000..bc15ce1 --- /dev/null +++ b/docs/rates-context-observer-research.md @@ -0,0 +1,74 @@ +# Ten-year rates context observation research + +Status: `UNVALIDATED_RESEARCH`. This is one source-agnostic, standard-library pure observation function. It is not a downloader, enabled plugin, strategy policy, or production adoption. Existing modules, package exports, runner/catalog entries, wide external-context CSVs, defaults, dependencies, and runtime pins are unchanged. + +[中文合同](rates-context-observer-research.zh-CN.md) + +## Purpose and boundary + +`quant_strategy_plugins.rates_context_observer_research.build_rates_context_observation(snapshot, config)` describes ten-year nominal yield, real yield, independently reported breakeven, and an explicitly approximate nominal-minus-real yield spread. It returns facts and data-availability diagnostics only. There are no signals, votes, direction/risk classifications, positions, budgets, orders, notifications, files, clocks, models, or provider calls. + +The existing wide CSV merge coerces each column to numeric values. It cannot preserve this contract's source, revision, or publication metadata; do not insert these records into that merge and assume provenance survives. This module neither extends `DEFAULT_FRED_SERIES` nor modifies macro watch/actionable scoring. + +## Input and configuration + +The input has exactly `schema_version=qsl.rates-context-input.research.v1`, `decision_at`, and `series`. `decision_at` is the actual evaluation/receipt decision time declared by the caller, with an explicit ISO timezone. `series` contains only the optional roles `nominal_10y`, `real_10y`, and `breakeven_10y`; a missing role yields explicit unknown. All three roles are ten-year measurements, not ETF prices or another tenor relabeled as ten-year. + +Each series has exactly: + +- `source_id`: identity of the actual source collection; the nominal/real pair must match +- `series_id`: the actual measured series identity, not a shared display name +- `basis`: nominal/real accept `treasury_par_yield` or `treasury_constant_maturity_yield`; reported breakeven accepts only `reported_breakeven` +- `unit`: exactly `percent`, meaning percentage points per annum where relevant; `4.2` means 4.2%, not 0.042 +- `rows`: caller-frozen observations, in strictly increasing observation-date order + +The same `source_id + series_id` identity cannot occupy two roles. Such reuse, including a breakeven role repeating a yield series, makes both roles unknown with `ROLE_SOURCE_IDENTITY_AMBIGUOUS`; a renamed role does not turn one observation into independent evidence. + +Each row has exactly `observation_date`, `value`, `available_at`, `received_at`, and `revision_id`: + +- Dates are exact `YYYY-MM-DD` source observation labels; no fixed time is manufactured from the label +- Values are finite non-Boolean numbers; negative yields are valid +- `available_at` is the caller's evidence-backed time that this exact revision became available, not the import time or scheduled release time +- `received_at` is when this exact revision reached the caller; both times require explicit timezones and are normalized to UTC +- Missing/invalid times remain unknown; receipt cannot precede availability +- `revision_id` identifies the selected frozen revision; `latest` is rejected. Two visible revisions of one date are ambiguous and rejected rather than silently taking the last row + +Configuration has exactly `schema_version=qsl.rates-context-config.research.v1`, `window_start`, `window_end`, and `max_observation_age_days`. All are mandatory. Start must precede end. Maximum age is a nonnegative integer number of UTC calendar days from the latest visible observation date to decision time. It is an explicit research allowance, not a calibrated freshness recommendation. + +## Exact windows, units, and unknowns + +Each series requires observations on both exact common endpoint dates. The module never selects a nearest date, carries a value forward, interpolates, or changes a window to fit a source. Endpoint differences do not prove full trading-session coverage. A missing start/end is `WINDOW_ENDPOINT_UNAVAILABLE`. + +`change_bp = 100 × (end_percent − start_percent)`. It is a yield change in basis points, not a price return or relative percentage change. For example, synthetic 4.0% to 4.2% is +20 bp. This example is `SYNTHETIC_FIXTURE_ONLY` and is not downloaded data. + +For historical decisions, dates outside the explicit window and rows published or received after `decision_at` are invisible before value/revision validation. Hidden later revisions and future values therefore cannot alter the historical observation. Unparseable selectors cannot establish invisibility and produce unknown. Within the visible projection, duplicates, reversed dates, nonfinite values, missing metadata, unknown basis, wrong units, stale observations, or insufficient endpoints produce unknown and no numeric change. Invalid configuration raises `ContractError` without echoing input. + +Each series independently retains its source/series/basis, endpoint observation/revision/availability/receipt metadata, latest observation date, visible count, observation age, date-level availability delay, and receipt lag. Availability delay is a difference of UTC date labels, not a measured market-close-to-publication latency. Stale data do not become fresh because they were imported today. One missing series does not erase valid facts from another; aggregate quality remains unknown if the complete bundle or pair comparison is unavailable. + +## Approximate spread is separate from reported breakeven + +Only nominal and real observations with the same declared source collection, measurement basis, units, and exact endpoints form `approximate_yield_spread`. It is labeled `APPROXIMATE_NOMINAL_MINUS_REAL_YIELD_SPREAD`. It is not automatically an official published breakeven, a pure inflation expectation, or a second independent risk vote. + +Examples of identity distinctions for a future caller-owned adapter: + +- A direct Treasury nominal-par/real-par pair is identified as its Treasury collection and `treasury_par_yield`; its difference remains an approximate spread +- A caller-authorized H.15/FRED DGS10/DFII10 pair retains the H.15/FRED identities and `treasury_constant_maturity_yield`; it is not mixed with a separately sourced Treasury record merely because both say ten-year +- An independently acquired T10YIE retains its actual source/revision/publication metadata and `reported_breakeven`. It is returned separately and never filled from the nominal-real subtraction + +The H.15 nominal and inflation-indexed constant-maturity series are source-specific yield measurements. [H.15 definitions](https://www.federalreserve.gov/releases/h15/) and [Treasury interest-rate statistics](https://home.treasury.gov/policy-issues/financing-the-government/interest-rate-statistics) explain the underlying tenors and methods. The [DGS10](https://fred.stlouisfed.org/series/DGS10), [DFII10](https://fred.stlouisfed.org/series/DFII10), and [T10YIE](https://fred.stlouisfed.org/series/T10YIE) metadata describe percent/daily units; T10YIE's notes separately describe its derivation and the direct-Treasury source used since June 21, 2019. Shared numerical values do not establish shared release times or immutable historical revisions. + +## Source rights and historical evidence + +This phase contains no new real-data collection, provider credentials, copied provider dataset, or redistributed observations. Tests use labeled synthetic values. Source links are metadata references and do not grant permission, certify a feed, or establish provider authenticity. + +At the October 6, 2026 review, FRED labels DGS10/DFII10 `Public Domain: Citation Requested` and T10YIE `Copyrighted: Citation Required`. A future data collector must check the applicable source rights, attribution, service/API terms, and intended use, including the restrictions in the [current FRED terms](https://fred.stlouisfed.org/legal/). This module does not perform that authorization review or assert a license. A dataset's public-domain status does not override service access/use terms. No new FRED collector or extension of the existing CSV fetch mechanism is implemented here. + +Every output remains `UNVALIDATED_RESEARCH`, `assurance=CALLER_DECLARATIONS_AND_CONSISTENCY_ONLY`, `historical_pit_verified=false`, `backtest_eligible=false`, and `position_control_allowed=false`. `declared_available` means only that supplied records satisfy internal timing and consistency checks. An available/received declaration or revision string is not a verified capture. Genuine historical PIT use still needs immutable source snapshots, actual publication/receipt evidence, source rights, and independent producer/consumer verification. Loading an old historical CSV today must not invent its historical `available_at`. + +## Offline verification and next step + +Focused tests can run with `python -m unittest discover -s tests -p test_rates_context_observer_research.py` in an environment whose network/child-process/model/notification entrypoints are blocked before importing tests. They load the pure module by file specification, isolating existing package-level optional dependencies. The normal package import and complete repository suite remain separate checks. + +Synthetic coverage includes percent-to-bp arithmetic, independent breakeven, individual source delays, negative yields, future-row/revision invariance, missing publication/receipt evidence, today's import of old data, nonfinite/Boolean/string values, wrong units, mixed sources/methodologies, reversed dates, visible duplicate revisions, wrong/missing endpoints, stale observations, invalid configuration, strict JSON, input immutability, and absent trading fields. + +This is a preparation step. A future authorized collector/consumer integration must retain the row-level evidence rather than using the lossy wide CSV, then validate genuine source coverage/availability, strategy-specific economic usefulness, and approved consumption. Historical breadth membership/prices and NDX participation are separate gaps; this phase implements neither breadth nor an NDX proxy. No runtime adoption follows from a pure function or passing synthetic tests. diff --git a/docs/rates-context-observer-research.zh-CN.md b/docs/rates-context-observer-research.zh-CN.md new file mode 100644 index 0000000..ad9e44c --- /dev/null +++ b/docs/rates-context-observer-research.zh-CN.md @@ -0,0 +1,74 @@ +# 十年利率背景观察研究 + +状态:`UNVALIDATED_RESEARCH`。本期只有一个来源无关、标准库纯观察函数。不是下载器、已启用插件、策略政策或生产采用。既有模块、包级导出、runner/catalog、外部背景宽表CSV、默认配置、依赖和runtime pin保持原样。 + +[Full English contract](rates-context-observer-research.md) + +## 用途与边界 + +`quant_strategy_plugins.rates_context_observer_research.build_rates_context_observation(snapshot, config)` 描述十年名义收益率、真实收益率、独立公布的breakeven,以及明确标为近似的名义减真实收益率差。只输出事实及数据可用性诊断,没有信号、投票、方向/风险分类、仓位、预算、订单、通知、文件、时钟、模型或provider调用。 + +原宽表CSV合并会将各列转成数值,不能保存本合同的来源、修订和发布时间。不能把新记录塞进该合并后声称来源证据仍完整。本模块不扩充 `DEFAULT_FRED_SERIES`,不修改macro的watch/actionable评分。 + +## 输入与配置 + +输入恰有 `schema_version=qsl.rates-context-input.research.v1`、`decision_at`、`series`。`decision_at` 是调用方声明的实际评估/接收决策时点,须有明确ISO时区。`series` 仅允许可缺省的 `nominal_10y`、`real_10y`、`breakeven_10y` 三个角色;缺少角色明确unknown。三者必须是真正十年期测量,不能把ETF价格或其他期限重标为十年。 + +每个序列恰有: + +- `source_id`:实际来源集合身份,名义/真实配对须一致 +- `series_id`:实际测量序列身份,不能只用共同显示名 +- `basis`:名义/真实仅接受 `treasury_par_yield` 或 `treasury_constant_maturity_yield`;独立breakeven仅接受 `reported_breakeven` +- `unit`:准确为 `percent`,适用时为每年百分点口径;4.2表示4.2%,不是0.042 +- `rows`:调用方冻结的观察记录,按观察日严格递增 + +相同 `source_id + series_id` 不能占两个角色;包括breakeven复用收益率序列。这类复用使两个角色均unknown,理由为 `ROLE_SOURCE_IDENTITY_AMBIGUOUS`。改角色名不能让同一观察变成独立证据。 + +每行恰有 `observation_date`、`value`、`available_at`、`received_at`、`revision_id`: + +- 观察日为准确 `YYYY-MM-DD` 来源标签;不从日期拼接固定时钟 +- 数值须有限、非布尔;负收益率合法 +- `available_at` 是调用方有证据支持的该修订实际首次可得时点,不是导入时间或预定发布时间 +- `received_at` 是该修订到达调用方的时间;两者须有明确时区并归一UTC +- 缺失/非法时间保持unknown;接收时间不能早于可得时间 +- `revision_id` 标识所选冻结修订,拒绝 `latest`。同日两个已可见修订属于歧义,不能静默取最后一行 + +配置恰有 `schema_version=qsl.rates-context-config.research.v1`、`window_start`、`window_end`、`max_observation_age_days`,全部必填。起日须早于止日。最大年龄是最新可见观察日距决策时点的非负UTC日历日数;只是显式研究容忍度,不是校准后的新鲜度建议。 + +## 共同端点、单位及unknown + +各序列须在同一组准确起止日都有观察。模块不选邻近日、不前填、不插值,也不为迁就来源而改变窗口。端点差不证明全部交易日覆盖;缺少起点或终点为 `WINDOW_ENDPOINT_UNAVAILABLE`。 + +`change_bp = 100 × (end_percent − start_percent)`,表示收益率变化的基点,非价格收益或相对百分比变化。例如合成4.0%至4.2%为+20bp;该例为 `SYNTHETIC_FIXTURE_ONLY`,不是下载数据。 + +历史决策投影中,窗口外日期、decision_at以后公布或收到的行,在读取数值/修订前即不可见。隐藏的迟到修订和未来值不能改变历史观察。不能解析的选择器无法证明属于未来,产生unknown。可见投影中重复/倒序日期、非有限值、缺metadata、未知口径、错误单位、过时或端点不足,均unknown且不给数值变化。非法配置抛出不回显输入的 `ContractError`。 + +各序列独立保存source/series/basis、端点观察/修订/可得/接收metadata、最新观察日、可见数量、观察年龄、日期级可得延迟和接收延迟。可得延迟只是UTC日期标签差,不是经核证的市场收盘至发布时间差。今天导入不能让旧观察变新鲜。一个序列缺失不抹掉另一个序列的合格声明事实;完整组合或配对不可得时,总质量仍unknown。 + +## 近似差值与独立breakeven分开 + +只有名义/真实使用同一声明来源集合、测量口径、单位及准确端点,才计算 `approximate_yield_spread`,明确标为 `APPROXIMATE_NOMINAL_MINUS_REAL_YIELD_SPREAD`。不能自动当作官方公布的breakeven、纯通胀预期或第二个独立风险投票。 + +后续调用方自行负责的窄适配须区分: + +- 直接Treasury名义par/真实par配对保留Treasury集合身份及 `treasury_par_yield`;相减仍是近似spread +- 已获准H.15/FRED DGS10/DFII10配对保留H.15/FRED身份及 `treasury_constant_maturity_yield`;不能只因均写十年,就与另采Treasury记录混配 +- 独立取得的T10YIE保留其真实来源/修订/公布时间及 `reported_breakeven`,单独返回,绝不以名义减真实补值 + +H.15名义/通胀指数恒定期限序列是来源特定的收益率测量。[H.15定义](https://www.federalreserve.gov/releases/h15/)及[Treasury利率统计](https://home.treasury.gov/policy-issues/financing-the-government/interest-rate-statistics)说明期限/方法。[DGS10](https://fred.stlouisfed.org/series/DGS10)、[DFII10](https://fred.stlouisfed.org/series/DFII10)、[T10YIE](https://fred.stlouisfed.org/series/T10YIE) metadata为percent/daily;T10YIE说明独立列明其派生关系,以及自2019-06-21直接使用Treasury数据。数值相同不证明发布时间或历史修订相同。 + +## 来源许可与历史证据 + +本期不新增真实数据采集、provider凭据、provider数据复制或观察值再发布。测试仅用有标签的合成数值。来源链接仅作metadata参考,不授予许可、不认证feed或provider真实性。 + +2026-10-06核查时,FRED将DGS10/DFII10标为 `Public Domain: Citation Requested`,T10YIE标为 `Copyrighted: Citation Required`。未来采集方须核适用来源权利、署名、服务/API条款及具体用途,包括[当前FRED条款](https://fred.stlouisfed.org/legal/)的限制。本模块不替其审核授权或宣称已获许可。数据公有领域标签不覆盖服务访问/用途条款。本期没有新增FRED采集器,也未扩充既有CSV读取机制。 + +所有输出始终是 `UNVALIDATED_RESEARCH`、`assurance=CALLER_DECLARATIONS_AND_CONSISTENCY_ONLY`、`historical_pit_verified=false`、`backtest_eligible=false`、`position_control_allowed=false`。`declared_available` 仅表示输入满足内部时间/一致性检查。时间声明、revision字符串不是经核证的采集证据;真实历史PIT仍须不可变来源快照、实际公布/接收证据、来源许可和独立producer/consumer核验。今天导入旧CSV不能伪造其历史available_at。 + +## 离线验证及后续 + +focused测试可用 `python -m unittest discover -s tests -p test_rates_context_observer_research.py`,运行环境必须在import测试前禁止网络/子进程/模型/通知入口。测试用文件spec加载纯模块,避开原包级可选依赖;正常包级import和全仓suite仍须单独验证。 + +合成覆盖percent→bp、独立breakeven、各源延迟、负利率、未来行/修订不变、缺公布/接收证据、今天导入旧数据、NaN/无穷/布尔/字符串、错误单位、混源/混口径、倒序/可见重复修订、错误/缺失端点、过时、非法配置、严格JSON、输入不变及无交易字段。 + +这只是准备步骤。未来获准采集/consumer接线须保留逐行证据,不能经过丢失metadata的宽表;另验真实来源覆盖/可得性、策略经济价值及获批消费。breadth历史membership/prices和NDX参与结构是独立缺口,本期没有实现breadth或NDX代理。纯函数或合成测试通过均不证明runtime采用。 diff --git a/src/quant_strategy_plugins/rates_context_observer_research.py b/src/quant_strategy_plugins/rates_context_observer_research.py new file mode 100644 index 0000000..c6fdc43 --- /dev/null +++ b/src/quant_strategy_plugins/rates_context_observer_research.py @@ -0,0 +1,235 @@ +"""Pure ten-year rates observations from caller-frozen, source-agnostic records. + +No provider, clock, I/O, registration, thresholds, votes, or trading authority. +Availability and source identities are declarations, never historical PIT proof. +""" +from __future__ import annotations + +from collections.abc import Mapping, Sequence +from datetime import date, datetime, timezone +import math + + +INPUT_VERSION = "qsl.rates-context-input.research.v1" +CONFIG_VERSION = "qsl.rates-context-config.research.v1" +OBSERVATION_VERSION = "qsl.rates-context-observation.research.v1" +ROLES = ("nominal_10y", "real_10y", "breakeven_10y") +_CONFIG_KEYS = {"schema_version", "window_start", "window_end", "max_observation_age_days"} +_SERIES_KEYS = {"source_id", "series_id", "basis", "unit", "rows"} +_ROW_KEYS = {"observation_date", "value", "available_at", "received_at", "revision_id"} +_YIELD_BASES = {"treasury_par_yield", "treasury_constant_maturity_yield"} + + +class ContractError(ValueError): + """Invalid configuration without echoing caller data.""" + + +def _date(value): + if not isinstance(value, str): + return None + try: + parsed = date.fromisoformat(value) + return parsed if parsed.isoformat() == value else None + except ValueError: + return None + + +def _time(value): + if not isinstance(value, str): + return None + try: + parsed = datetime.fromisoformat(value.replace("Z", "+00:00")) + return parsed.astimezone(timezone.utc) if parsed.tzinfo is not None else None + except (ValueError, OverflowError): + return None + + +def _stamp(value): + return value.isoformat().replace("+00:00", "Z") if value is not None else None + + +def _text(value): + return value if isinstance(value, str) and value.strip() == value and 0 < len(value) <= 256 else None + + +def _number(value): + if not isinstance(value, (int, float)) or isinstance(value, bool): + return None + try: + number = float(value) + return number if math.isfinite(number) else None + except (ValueError, OverflowError): + return None + + +def _rounded(value): + if not math.isfinite(value): + return None + rounded = round(value, 12) + return 0.0 if rounded == 0 else rounded + + +def _config(value): + if not isinstance(value, Mapping) or set(value) != _CONFIG_KEYS or value.get("schema_version") != CONFIG_VERSION: + raise ContractError("invalid_config") + start, end, age = _date(value["window_start"]), _date(value["window_end"]), value["max_observation_age_days"] + if start is None or end is None or start >= end or not isinstance(age, int) or isinstance(age, bool) or age < 0: + raise ContractError("invalid_config") + return start, end, age + + +def _endpoint(row): + if row is None: + return None + return {"observation_date": row["observation_date"], + "available_at": _stamp(_time(row.get("available_at"))), + "received_at": _stamp(_time(row.get("received_at"))), + "revision_id": _text(row.get("revision_id"))} + + +def _observe_series(series, role, start, end, decision, max_age, top_valid): + series = series if isinstance(series, Mapping) else {} + result = {"status": "unknown", "reason_codes": [], "source_id": _text(series.get("source_id")), + "series_id": _text(series.get("series_id")), "basis": _text(series.get("basis")), + "declared_unit": _text(series.get("unit")), "unit": "percent", "start_percent": None, + "end_percent": None, "change_bp": None, "start_observation": None, "end_observation": None, + "latest_observation_date": None, "observation_age_days": None, + "availability_delay_calendar_days": None, "receipt_lag_seconds": None, + "visible_observations": 0} + reasons = result["reason_codes"] + if not top_valid: + reasons.append("INPUT_INVALID") + return result + if not series: + reasons.append("SERIES_MISSING") + return result + if set(series) != _SERIES_KEYS or result["source_id"] is None or result["series_id"] is None: + reasons.append("SOURCE_IDENTITY_INVALID") + allowed_basis = {"reported_breakeven"} if role == "breakeven_10y" else _YIELD_BASES + if result["basis"] not in allowed_basis: + reasons.append("BASIS_INVALID") + if series.get("unit") != "percent": + reasons.append("UNIT_NOT_PERCENT") + rows = series.get("rows") + if not isinstance(rows, Sequence) or isinstance(rows, (str, bytes, bytearray)): + reasons.append("ROWS_INVALID") + return result + visible = [] + previous = None + for row in rows: + if not isinstance(row, Mapping) or (observed := _date(row.get("observation_date"))) is None: + reasons.append("ROW_DATE_INVALID") + continue + # Historical projection precedes all other fields: no future values, + # delayed revisions, or dates outside the exact interval can affect it. + if observed < start or observed > end: + continue + available = _time(row.get("available_at")) + if available is not None and available > decision: + continue + received = _time(row.get("received_at")) + if received is not None and received > decision: + continue + if previous is not None and observed <= previous: + reasons.append("ROW_ORDER_OR_REVISION_AMBIGUOUS") + previous = observed + visible.append(row) + if set(row) != _ROW_KEYS: + reasons.append("ROW_FIELDS_INVALID") + if available is None: + reasons.append("AVAILABILITY_UNKNOWN") + elif available.date() < observed: + reasons.append("AVAILABILITY_BEFORE_OBSERVATION") + if received is None: + reasons.append("RECEIPT_UNKNOWN") + elif available is not None and received < available: + reasons.append("RECEIPT_BEFORE_AVAILABILITY") + if _text(row.get("revision_id")) is None or row.get("revision_id") == "latest": + reasons.append("REVISION_IDENTITY_INVALID") + if _number(row.get("value")) is None: + reasons.append("VALUE_INVALID") + result["visible_observations"] = len(visible) + by_date = {row["observation_date"]: row for row in visible} + first, last = by_date.get(start.isoformat()), by_date.get(end.isoformat()) + result["start_observation"], result["end_observation"] = _endpoint(first), _endpoint(last) + if first is None or last is None: + reasons.append("WINDOW_ENDPOINT_UNAVAILABLE") + if visible: + latest = max(_date(row["observation_date"]) for row in visible) + result["latest_observation_date"] = latest.isoformat() + result["observation_age_days"] = (decision.date() - latest).days + if result["observation_age_days"] > max_age: + reasons.append("OBSERVATION_STALE") + if latest > decision.date(): + reasons.append("OBSERVATION_AFTER_DECISION") + if last is not None: + available, received = _time(last.get("available_at")), _time(last.get("received_at")) + if available is not None: + result["availability_delay_calendar_days"] = (available.date() - end).days + if available is not None and received is not None: + result["receipt_lag_seconds"] = (received - available).total_seconds() + if not reasons: + first_value, last_value = _number(first["value"]), _number(last["value"]) + change = _rounded(100 * (last_value - first_value)) + if change is None: + reasons.append("CHANGE_NONFINITE") + else: + result.update(status="declared_available", start_percent=first_value, + end_percent=last_value, change_bp=change) + result["reason_codes"] = sorted(set(reasons)) + return result + + +def build_rates_context_observation(snapshot: Mapping, config: Mapping) -> dict: + """Observe exact common endpoint changes, never a release-time guess. + + Explicit timezone-aware available_at and received_at are required for both + endpoints. A receipt today cannot prove a historical publication time. + The result checks declarations only and remains unvalidated research. + """ + start, end, max_age = _config(config) + snapshot = snapshot if isinstance(snapshot, Mapping) else {} + decision = _time(snapshot.get("decision_at")) + series = snapshot.get("series") + top_valid = (set(snapshot) == {"schema_version", "decision_at", "series"} + and snapshot.get("schema_version") == INPUT_VERSION and decision is not None + and isinstance(series, Mapping) and not set(series).difference(ROLES)) + series = series if isinstance(series, Mapping) else {} + observations = {role: _observe_series(series.get(role), role, start, end, decision, max_age, top_valid) + for role in ROLES} + identities = {} + for role, item in observations.items(): + if item["source_id"] is not None and item["series_id"] is not None: + identities.setdefault((item["source_id"], item["series_id"]), []).append(role) + for repeated_roles in identities.values(): + if len(repeated_roles) > 1: + for role in repeated_roles: + item = observations[role] + item.update(status="unknown", start_percent=None, end_percent=None, change_bp=None) + item["reason_codes"] = sorted(set(item["reason_codes"] + ["ROLE_SOURCE_IDENTITY_AMBIGUOUS"])) + spread = {"status": "unknown", "reason_codes": [], "measurement": "APPROXIMATE_NOMINAL_MINUS_REAL_YIELD_SPREAD", + "source_id": None, "basis": None, "input_series_ids": [], "unit": "percent", + "start_percent": None, "end_percent": None, "change_bp": None} + nominal, real = observations["nominal_10y"], observations["real_10y"] + if any(item["status"] != "declared_available" for item in (nominal, real)): + spread["reason_codes"] = ["COMMON_WINDOW_UNAVAILABLE"] + elif nominal["source_id"] != real["source_id"] or nominal["basis"] != real["basis"]: + spread["reason_codes"] = ["SOURCE_OR_BASIS_MISMATCH"] + else: + first = _rounded(nominal["start_percent"] - real["start_percent"]) + last = _rounded(nominal["end_percent"] - real["end_percent"]) + change = None if first is None or last is None else _rounded(100 * (last - first)) + if change is None: + spread["reason_codes"] = ["SPREAD_NONFINITE"] + else: + spread.update(status="declared_available", source_id=nominal["source_id"], basis=nominal["basis"], + input_series_ids=[nominal["series_id"], real["series_id"]], + start_percent=first, end_percent=last, change_bp=change) + return {"schema_version": OBSERVATION_VERSION, "research_status": "UNVALIDATED_RESEARCH", + "assurance": "CALLER_DECLARATIONS_AND_CONSISTENCY_ONLY", "historical_pit_verified": False, + "backtest_eligible": False, "position_control_allowed": False, + "decision_at": _stamp(decision), "window_start": start.isoformat(), "window_end": end.isoformat(), + "window_method": "EXACT_ENDPOINT_DIFFERENCE", "coverage": "ENDPOINTS_ONLY_NOT_SESSION_COVERAGE", + "max_observation_age_days": max_age, "series": observations, "approximate_yield_spread": spread, + "quality": {"status": "declared_available" if all(item["status"] == "declared_available" + for item in observations.values()) and spread["status"] == "declared_available" else "unknown"}} diff --git a/tests/test_rates_context_observer_research.py b/tests/test_rates_context_observer_research.py new file mode 100644 index 0000000..4edb501 --- /dev/null +++ b/tests/test_rates_context_observer_research.py @@ -0,0 +1,337 @@ +"""SYNTHETIC_FIXTURE_ONLY: no downloaded values or provider calls.""" +from __future__ import annotations + +import copy +import importlib.util +import json +from pathlib import Path +import unittest + + +_SOURCE = Path(__file__).parents[1] / "src/quant_strategy_plugins/rates_context_observer_research.py" +_SPEC = importlib.util.spec_from_file_location("rates_context_observer_research", _SOURCE) +observer = importlib.util.module_from_spec(_SPEC) +_SPEC.loader.exec_module(observer) + + +def fixture(): + # Arbitrary synthetic percentages, deliberately unrelated to provider data. + series = {} + for role, values in (("nominal_10y", (4.0, 4.2)), ("real_10y", (1.0, 1.1)), + ("breakeven_10y", (2.8, 2.9))): + rows = [] + for day, value, published in (("2024-01-02", values[0], "2024-01-03T13:00:00Z"), + ("2024-01-03", values[1], "2024-01-04T13:00:00Z")): + rows.append({"observation_date": day, "value": value, "available_at": published, + "received_at": published.replace(":00:00", ":01:00"), "revision_id": "synthetic-v1"}) + series[role] = {"source_id": "SYNTHETIC_FIXTURE_ONLY:rates", "series_id": role, + "basis": "reported_breakeven" if role == "breakeven_10y" else "treasury_par_yield", + "unit": "percent", "rows": rows} + return {"schema_version": observer.INPUT_VERSION, "decision_at": "2024-01-04T15:00:00Z", + "series": series} + + +def config(): + return {"schema_version": observer.CONFIG_VERSION, "window_start": "2024-01-02", + "window_end": "2024-01-03", "max_observation_age_days": 2} + + +class RatesContextObserverTests(unittest.TestCase): + def build(self, snapshot=None, policy=None): + return observer.build_rates_context_observation(fixture() if snapshot is None else snapshot, + config() if policy is None else policy) + + def assert_unknown(self, snapshot, role="nominal_10y"): + result = self.build(snapshot) + self.assertEqual(result["series"][role]["status"], "unknown") + self.assertIsNone(result["series"][role]["change_bp"]) + return result + + def test_percent_change_is_basis_points_not_relative_return(self): + result = self.build() + self.assertEqual(result["series"]["nominal_10y"]["change_bp"], 20.0) + self.assertEqual(result["series"]["real_10y"]["change_bp"], 10.0) + self.assertEqual(result["approximate_yield_spread"]["end_percent"], 3.1) + self.assertEqual(result["approximate_yield_spread"]["change_bp"], 10.0) + + def test_independent_reported_breakeven_is_not_replaced_by_difference(self): + result = self.build() + self.assertEqual(result["series"]["breakeven_10y"]["end_percent"], 2.9) + self.assertEqual(result["approximate_yield_spread"]["end_percent"], 3.1) + self.assertEqual(result["research_status"], "UNVALIDATED_RESEARCH") + self.assertFalse(result["historical_pit_verified"]) + + def test_source_dates_and_per_series_update_lag_preserved(self): + snapshot = fixture() + snapshot["series"]["real_10y"]["rows"][-1]["available_at"] = "2024-01-03T17:00:00Z" + result = self.build(snapshot) + nominal, real = (result["series"][role] for role in ("nominal_10y", "real_10y")) + self.assertEqual(nominal["observation_age_days"], 1) + self.assertEqual(nominal["availability_delay_calendar_days"], 1) + self.assertEqual(real["availability_delay_calendar_days"], 0) + self.assertEqual(nominal["end_observation"]["observation_date"], "2024-01-03") + self.assertEqual(nominal["end_observation"]["revision_id"], "synthetic-v1") + + def test_negative_real_rates_are_valid(self): + snapshot = fixture() + snapshot["series"]["real_10y"]["rows"][0]["value"] = -1.0 + snapshot["series"]["real_10y"]["rows"][1]["value"] = -0.5 + result = self.build(snapshot) + self.assertEqual(result["series"]["real_10y"]["change_bp"], 50.0) + + def test_missing_series_is_explicit_unknown(self): + snapshot = fixture() + del snapshot["series"]["nominal_10y"] + self.assert_unknown(snapshot) + + def test_missing_available_at_is_not_replaced_by_received_today(self): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["rows"][0]["available_at"] = None + result = self.assert_unknown(snapshot) + self.assertIn("AVAILABILITY_UNKNOWN", result["series"]["nominal_10y"]["reason_codes"]) + + def test_importing_old_history_today_does_not_establish_pit(self): + snapshot = fixture() + for row in snapshot["series"]["nominal_10y"]["rows"]: + row["available_at"] = None + row["received_at"] = "2024-01-04T14:00:00Z" + self.assert_unknown(snapshot) + + def test_future_availability_is_invisible_to_historical_decision(self): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["rows"][-1]["available_at"] = "2024-01-05T13:00:00Z" + self.assert_unknown(snapshot) + + def test_future_receipt_is_not_consumable(self): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["rows"][-1]["received_at"] = "2024-01-05T13:01:00Z" + self.assert_unknown(snapshot) + + def test_missing_receipt_is_unknown(self): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["rows"][-1]["received_at"] = None + self.assert_unknown(snapshot) + + def test_receipt_before_publication_is_unknown(self): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["rows"][-1]["received_at"] = "2024-01-04T12:00:00Z" + self.assert_unknown(snapshot) + + def test_timezone_is_required(self): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["rows"][-1]["available_at"] = "2024-01-04T13:00:00" + self.assert_unknown(snapshot) + + def test_inverted_dates_are_not_silently_sorted(self): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["rows"].reverse() + self.assert_unknown(snapshot) + + def test_duplicate_revisions_are_not_latest_wins(self): + snapshot = fixture() + duplicate = copy.deepcopy(snapshot["series"]["nominal_10y"]["rows"][-1]) + duplicate["revision_id"] = "synthetic-v2" + snapshot["series"]["nominal_10y"]["rows"].append(duplicate) + self.assert_unknown(snapshot) + + def test_unavailable_later_revision_does_not_change_historical_result(self): + snapshot = fixture() + expected = self.build(snapshot) + future = {"observation_date": "2024-01-03", "available_at": "2024-01-05T13:00:00Z", + "received_at": "invalid", "value": float("nan"), "revision_id": "synthetic-future"} + snapshot["series"]["nominal_10y"]["rows"].append(future) + self.assertEqual(self.build(snapshot), expected) + + def test_future_observation_is_hidden_before_reading_value(self): + snapshot = fixture() + expected = self.build(snapshot) + snapshot["series"]["nominal_10y"]["rows"].append({"observation_date": "2024-01-10", "value": object()}) + self.assertEqual(self.build(snapshot), expected) + + def test_nonfinite_boolean_and_string_values_are_unknown(self): + for value in (float("nan"), float("inf"), -float("inf"), True, "4.2"): + with self.subTest(value=repr(value)): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["rows"][-1]["value"] = value + self.assert_unknown(snapshot) + + def test_overflowed_change_is_unknown_not_nonfinite_json(self): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["rows"][0]["value"] = -1e308 + snapshot["series"]["nominal_10y"]["rows"][1]["value"] = 1e308 + self.assert_unknown(snapshot) + + def test_unit_fraction_or_bp_is_not_assumed_percent(self): + for unit in ("fraction", "basis_points", "Percent"): + with self.subTest(unit=unit): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["unit"] = unit + self.assert_unknown(snapshot) + + def test_mixed_sources_cannot_form_approximate_spread(self): + snapshot = fixture() + snapshot["series"]["real_10y"]["source_id"] = "SYNTHETIC_FIXTURE_ONLY:other" + result = self.build(snapshot) + self.assertEqual(result["approximate_yield_spread"]["status"], "unknown") + self.assertIsNone(result["approximate_yield_spread"]["change_bp"]) + + def test_same_source_series_in_nominal_and_real_is_ambiguous(self): + snapshot = fixture() + snapshot["series"]["real_10y"]["series_id"] = snapshot["series"]["nominal_10y"]["series_id"] + result = self.build(snapshot) + for role in ("nominal_10y", "real_10y"): + self.assertEqual(result["series"][role]["status"], "unknown") + self.assertIn("ROLE_SOURCE_IDENTITY_AMBIGUOUS", result["series"][role]["reason_codes"]) + self.assertEqual(result["approximate_yield_spread"]["status"], "unknown") + + def test_breakeven_reusing_nominal_source_series_is_ambiguous(self): + snapshot = fixture() + snapshot["series"]["breakeven_10y"]["series_id"] = snapshot["series"]["nominal_10y"]["series_id"] + result = self.build(snapshot) + for role in ("nominal_10y", "breakeven_10y"): + self.assertEqual(result["series"][role]["status"], "unknown") + + def test_declarations_cannot_enable_backtest_or_position_control(self): + result = self.build() + self.assertIs(result["backtest_eligible"], False) + self.assertIs(result["position_control_allowed"], False) + + def test_mixed_methodologies_cannot_form_approximate_spread(self): + snapshot = fixture() + snapshot["series"]["real_10y"]["basis"] = "treasury_constant_maturity_yield" + self.assertEqual(self.build(snapshot)["approximate_yield_spread"]["status"], "unknown") + + def test_unknown_basis_is_unknown(self): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["basis"] = "intraday_bond_price" + self.assert_unknown(snapshot) + + def test_source_dates_must_match_exact_common_window(self): + snapshot = fixture() + snapshot["series"]["real_10y"]["rows"][-1]["observation_date"] = "2024-01-04" + result = self.assert_unknown(snapshot, "real_10y") + self.assertEqual(result["approximate_yield_spread"]["status"], "unknown") + + def test_window_missing_start_is_insufficient_not_forward_filled(self): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["rows"].pop(0) + self.assert_unknown(snapshot) + + def test_stale_observation_stays_stale_even_if_received_now(self): + snapshot = fixture() + snapshot["decision_at"] = "2024-01-10T15:00:00Z" + for row in snapshot["series"]["nominal_10y"]["rows"]: + row["received_at"] = "2024-01-10T14:00:00Z" + result = self.assert_unknown(snapshot) + self.assertIn("OBSERVATION_STALE", result["series"]["nominal_10y"]["reason_codes"]) + + def test_independent_series_failure_does_not_erase_valid_nominal(self): + snapshot = fixture() + del snapshot["series"]["breakeven_10y"] + result = self.build(snapshot) + self.assertEqual(result["series"]["nominal_10y"]["change_bp"], 20.0) + self.assertEqual(result["quality"]["status"], "unknown") + + def test_malformed_snapshot_is_unknown(self): + self.assertEqual(self.build({})["quality"]["status"], "unknown") + + def test_invalid_config_raises_without_input_echo(self): + policy = config() + policy["window_start"] = "2024-01-04" + with self.assertRaisesRegex(observer.ContractError, "invalid_config"): + self.build(policy=policy) + + def test_extra_fields_are_not_hidden_authority(self): + snapshot = fixture() + snapshot["target_weight"] = 1.0 + self.assertEqual(self.build(snapshot)["quality"]["status"], "unknown") + + def test_output_is_strict_json_and_has_no_trading_fields(self): + result = self.build() + json.dumps(result, allow_nan=False) + forbidden = {"target_weight", "position_control", "order", "capital", "risk_off", "risk_on", "vote"} + def walk(value): + if isinstance(value, dict): + self.assertFalse(forbidden.intersection(value)) + for child in value.values(): + walk(child) + elif isinstance(value, list): + for child in value: + walk(child) + walk(result) + + def test_inputs_are_not_mutated(self): + snapshot, policy = fixture(), config() + before = copy.deepcopy((snapshot, policy)) + self.build(snapshot, policy) + self.assertEqual((snapshot, policy), before) + + def test_invalid_decision_timezone_is_unknown(self): + snapshot = fixture() + snapshot["decision_at"] = "2024-01-04T15:00:00" + self.assert_unknown(snapshot) + + def test_invalid_config_age_or_fields_is_rejected(self): + for value in (-1, True, 1.5): + with self.subTest(value=value): + policy = config() + policy["max_observation_age_days"] = value + with self.assertRaises(observer.ContractError): + self.build(policy=policy) + policy = config() + policy["hidden_threshold"] = 1 + with self.assertRaises(observer.ContractError): + self.build(policy=policy) + + def test_visible_extra_row_fields_are_invalid(self): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["rows"][-1]["unit"] = "basis_points" + self.assert_unknown(snapshot) + + def test_invalid_date_selector_is_unknown(self): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["rows"][-1]["observation_date"] = "20240103" + self.assert_unknown(snapshot) + + def test_latest_revision_is_not_a_frozen_identity(self): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["rows"][-1]["revision_id"] = "latest" + self.assert_unknown(snapshot) + + def test_outside_start_window_rows_do_not_alter_exact_endpoints(self): + snapshot = fixture() + expected = self.build(snapshot) + snapshot["series"]["nominal_10y"]["rows"].insert(0, {"observation_date": "2024-01-01", "value": object()}) + self.assertEqual(self.build(snapshot), expected) + + def test_freshness_boundary_is_explicit_calendar_days(self): + snapshot = fixture() + snapshot["decision_at"] = "2024-01-05T15:00:00Z" + self.assertEqual(self.build(snapshot)["series"]["nominal_10y"]["status"], "declared_available") + snapshot["decision_at"] = "2024-01-06T00:00:00Z" + self.assert_unknown(snapshot) + + def test_times_with_offsets_are_normalized_to_utc(self): + snapshot = fixture() + row = snapshot["series"]["nominal_10y"]["rows"][-1] + row["available_at"], row["received_at"] = "2024-01-04T08:00:00-05:00", "2024-01-04T08:01:00-05:00" + result = self.build(snapshot) + self.assertEqual(result["series"]["nominal_10y"]["end_observation"]["available_at"], "2024-01-04T13:00:00Z") + + def test_publication_before_observation_date_is_unknown(self): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["rows"][-1]["available_at"] = "2024-01-02T13:00:00Z" + self.assert_unknown(snapshot) + + def test_unknown_preserves_available_endpoint_metadata(self): + snapshot = fixture() + snapshot["series"]["nominal_10y"]["rows"][-1]["available_at"] = None + result = self.assert_unknown(snapshot) + end = result["series"]["nominal_10y"]["end_observation"] + self.assertEqual(end["observation_date"], "2024-01-03") + self.assertIsNone(end["available_at"]) + self.assertEqual(end["received_at"], "2024-01-04T13:01:00Z") + + +if __name__ == "__main__": + unittest.main()