From 6884ad02b72c39c29ec29fa7ab5ebe5817e88839 Mon Sep 17 00:00:00 2001 From: Anatoly Maltsev Date: Fri, 25 Sep 2026 12:16:35 +0400 Subject: [PATCH 01/10] updv1 --- CHANGELOG.md | 1553 +---------------- docs/errors/NR-B003.md | 66 - docs/errors/README.md | 1 - pyproject.toml | 22 +- scripts/smoke_prod.py | 169 ++ src/nullrun/__init__.py | 150 +- src/nullrun/__version__.py | 2 +- src/nullrun/_handle.py | 4 - src/nullrun/actions.py | 5 +- src/nullrun/audit.py | 2 +- src/nullrun/breaker/circuit_breaker.py | 2 - src/nullrun/breaker/exceptions.py | 117 +- src/nullrun/business_impact.py | 427 +---- src/nullrun/capabilities.py | 5 - src/nullrun/context.py | 37 +- src/nullrun/decorators.py | 661 ++----- src/nullrun/extractor.py | 1157 ------------ src/nullrun/instrumentation/auto.py | 27 - src/nullrun/instrumentation/auto_requests.py | 6 +- src/nullrun/instrumentation/autogen.py | 3 - src/nullrun/instrumentation/langgraph.py | 15 - src/nullrun/instrumentation/llama_index.py | 1 - src/nullrun/integrations/fastapi.py | 81 +- src/nullrun/messages.py | 10 +- src/nullrun/observability/__init__.py | 2 - src/nullrun/observability/error_hooks.py | 10 +- src/nullrun/observability/status.py | 5 +- src/nullrun/toolbox/__init__.py | 3 +- src/nullrun/toolbox/langgraph.py | 118 +- src/nullrun/toolbox/mcp.py | 19 +- src/nullrun/transport.py | 158 +- src/nullrun/transport_websocket.py | 11 +- src/nullrun/uuid7.py | 1 - tests/test_2026_08_11_fixes.py | 15 +- tests/test_2026_09_08_gate_first_execute.py | 296 ---- .../test_2026_09_10_catchfanin_passthrough.py | 30 +- ...t_2026_09_10_decision_infra_passthrough.py | 448 ----- tests/test_2026_09_10_nr_ex01_passthrough.py | 312 ---- tests/test_2026_09_10_r001_passthrough.py | 391 ----- tests/test_2026_09_10_toolblocked_parser.py | 4 +- ...9_11_execute_capture_wires_execution_id.py | 221 --- tests/test_actions.py | 66 +- tests/test_approval_money_flow.py | 395 ----- tests/test_args_pii_masked.py | 137 -- tests/test_business_impact.py | 465 ----- tests/test_error_hooks.py | 23 +- tests/test_exception_hierarchy.py | 30 +- tests/test_execute_approval_flow.py | 127 -- tests/test_execute_tools_propagation.py | 506 ------ tests/test_integrations_fastapi.py | 42 +- tests/test_mcp_adapter.py | 8 +- tests/test_messages.py | 16 +- tests/test_money_hardening.py | 427 ----- tests/test_no_local_policy.py | 5 +- tests/test_observability.py | 413 ----- tests/test_preflight_fail_policy.py | 994 ----------- tests/test_protect.py | 971 ----------- tests/test_protect_branches.py | 564 ------ tests/test_protect_cancel_on_exception.py | 400 ----- tests/test_protect_only_public_api.py | 306 ---- tests/test_runtime.py | 155 +- tests/test_runtime_branches.py | 136 -- tests/test_sensitive_extractor.py | 289 --- tests/test_tool_params_extractor.py | 676 ------- tests/test_toolbox_langgraph.py | 110 -- tests/test_track_span_context.py | 71 +- tests/test_transport.py | 69 +- tests/test_transport_branches.py | 15 +- tests/test_typed_exceptions_full_audit.py | 5 - tests/test_units_discriminator.py | 495 ------ 70 files changed, 635 insertions(+), 13848 deletions(-) delete mode 100644 docs/errors/NR-B003.md create mode 100644 scripts/smoke_prod.py delete mode 100644 src/nullrun/extractor.py delete mode 100644 tests/test_2026_09_08_gate_first_execute.py delete mode 100644 tests/test_2026_09_10_decision_infra_passthrough.py delete mode 100644 tests/test_2026_09_10_nr_ex01_passthrough.py delete mode 100644 tests/test_2026_09_10_r001_passthrough.py delete mode 100644 tests/test_2026_09_11_execute_capture_wires_execution_id.py delete mode 100644 tests/test_approval_money_flow.py delete mode 100644 tests/test_args_pii_masked.py delete mode 100644 tests/test_business_impact.py delete mode 100644 tests/test_execute_approval_flow.py delete mode 100644 tests/test_execute_tools_propagation.py delete mode 100644 tests/test_money_hardening.py delete mode 100644 tests/test_observability.py delete mode 100644 tests/test_preflight_fail_policy.py delete mode 100644 tests/test_protect.py delete mode 100644 tests/test_protect_branches.py delete mode 100644 tests/test_protect_cancel_on_exception.py delete mode 100644 tests/test_protect_only_public_api.py delete mode 100644 tests/test_sensitive_extractor.py delete mode 100644 tests/test_tool_params_extractor.py delete mode 100644 tests/test_toolbox_langgraph.py delete mode 100644 tests/test_units_discriminator.py diff --git a/CHANGELOG.md b/CHANGELOG.md index b18955c..3e828c6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,1542 +1,21 @@ -## [0.18.1] - 2026-09-22 +## [0.18.2] - 2026-09-22 -Patch release — **lower-friction UX**: the user gets enforcement + observability from `@protect` alone, without having to call `init_or_die()` first or pick a framework extra. Closes four silent-failure modes at once: (1) `@protect` now auto-attaches a default tool-params extractor (no separate `@sensitive` needed for the common case; bare `@sensitive` is deprecated in favour of `@protect`); (2) `@protect` lazy-triggers `auto_instrument()` on first invocation so the user can write `@protect` before `init_or_die()` (or skip `init` entirely if `NULLRUN_API_KEY` is set); (3) a zero-activity diagnostic emits a one-time WARNING when `@protect` fires 50+ times without a single LLM event, naming the three most likely root causes; (4) `handle()` / `guarded()` / `init_or_die()` print a four-line developer report (what / where / why / how-to-fix) instead of just the catalog user-message. Dead `pip` extras (`[openai]`, `[anthropic]`, `[mistral]`, `[gemini]`, `[cohere]`, `[bedrock]`, `[all]`, `[fastapi]`) are removed — `pip install nullrun` alone is now sufficient for the HTTP-level + `@protect` flow. `toolbox.langgraph.wrapper()` is marked DEPRECATED in its docstring (auto-patch is the canonical path). Wire-format unchanged. SDK_MIN_VERSION unchanged. +### Surface -### Added +`@protect` is the single user-facing entry point. Every call routes +through `/api/v1/execute` unconditionally — no opt-outs, no per-tool +registry to manage, no alternate decorators. -- **`@protect` auto-attaches a default tool-params extractor** (`src/nullrun/decorators.py`, `src/nullrun/extractor.py`, `fcd623c`). The first call from `@protect` stamps `ToolParamsExtractor(include_all=True)` on the decorated function so the wire payload carries `tool_name + params` for every protected call — no second decorator required for the common case. The auto-attached extractor carries `_nullrun_auto_attached=True` so `_enforce_sensitive_tool` can distinguish "developer opted into the policy path" from "SDK auto-derived for tooling reasons"; the policy gate short-circuits on auto-attached extractors so bare `@protect` stays cheap (no extra `/execute` round-trip). Bounded extraction: oversized string values (`>=1024 bytes`) get a deterministic `...[truncated:N bytes]` suffix; circular references in nested dict/list structures return the partial walk instead of raising `RecursionError`; dropped values (float, bytes, custom) emit a single aggregate DEBUG log line with count + type names, never one log line per dropped field. +Top-level `dir(nullrun)` exposes only: -- **`@protect` lazy-triggers `auto_instrument()`** on first invocation (`src/nullrun/decorators.py`). The user's first decorated function call installs the runtime and patches `httpx` + framework adapters in a single process-wide idempotent step, behind a lock so concurrent `@protect` calls cannot double-fire. Never raises — a vendor SDK breaking change must not block the enforcement gate. Closes the "I added `@protect` but nothing tracks tokens" silent-failure mode. - -- **Zero-activity diagnostic on `NullRunRuntime`** (`src/nullrun/runtime.py`, `DEF-ZERO-ACTIVITY-DIAG`). `_bump_protect_count()` is invoked by `@protect` on every call; `_llm_call_event_count` is bumped by `track_llm()` on every successful call. When `@protect` fires 50+ times without a single LLM event, the runtime emits a one-time WARNING at `logging.WARNING` naming the three most likely root causes (raw httpx outside the patchable surface, custom transport / gRPC, framework not on the auto-detection table). Warn-once invariant under concurrent `@protect` calls is preserved by `_zero_activity_lock`. - -- **Four-line developer error report from `handle()` / `guarded()` / `init_or_die()`** (`src/nullrun/_handle.py`, `DEF-DEV-REPORT-EMPTY`). The catch-all exit path previously printed only the catalog user-message ("There's a configuration issue. Please contact support.") — end-user wording that gave a developer running an example with a missing `NULLRUN_API_KEY` zero actionable detail. The new `_render_dev_error_report()` helper emits a structured report answering the four questions a developer actually asks: (1) `what` — the stage that failed (`auth` / `gate` / `track` / `execute` / `approval` / …), (2) `where` — the wire endpoint + status code + transport source, (3) `why` — the underlying exception message + the machine `error_code`, (4) `how to fix` — the `user_action` from the typed class. The catalog headline is preserved as the first line so end-user-facing deployments still get a clean single sentence. Defensive: `handle()` / `init_or_die()` wrap the helper in `try/except` so a buggy report builder cannot freeze a script that would otherwise exit (falls back to the legacy single-line behaviour). - -### Changed - -- **Removed dead provider extras from `pyproject.toml`**: `[openai]`, `[anthropic]`, `[mistral]`, `[gemini]`, `[cohere]`, `[bedrock]`, `[all]`, `[fastapi]`. NullRun never imported these vendor SDKs — HTTP-level instrumentation (`patch_httpx` + 5 URL-keyed extractors) covers OpenAI, Azure, Anthropic, Mistral, Gemini, Cohere, and Bedrock without them. Kept framework extras (`[opentelemetry]`, `[langgraph]`, `[agents]`, `[langchain]`, `[llama-index]`, `[crewai]`, `[autogen]`) — those still install a vendor SDK that NullRun subscribes to via an event hook. `pip install nullrun` alone is now sufficient for the HTTP-level + `@protect` flow. - -- **`toolbox.langgraph.wrapper()` marked DEPRECATED** in its docstring (`src/nullrun/toolbox/langgraph.py`). `init_or_die()` / `@protect` auto-patches `langgraph.pregel.Pregel` via `patch_langgraph_compiled` — same callback injection the wrapper does manually, but process-wide and idempotent. `wrapper()` remains as an escape hatch for three narrow cases (tests with custom runtimes, Pregel imported before init, manual callback control). No removal — the public symbol stays. - -- **Bare `@sensitive` emits `DeprecationWarning`** (`src/nullrun/decorators.py`). `@sensitive(impact=...)` remains the advanced API for typed `BusinessImpact` + SHA-256 `action_digest`. Bare `@sensitive` is removed in `0.19.x`; emit is `DeprecationWarning` in `0.18.x` only. - -### Fixed - -- **DEF-DEV-REPORT-EMPTY** — `handle()` / `guarded()` / `init_or_die()` now print the four-line developer report (catalog headline + `[error_code]` + what + where + why + how-to-fix + docs URL) instead of just the catalog user-message. Closes the silent-failure mode where a developer hit a config failure at the first gate call and saw only end-user wording with no actionable detail. Defensive: the helper is wrapped in `try/except` so a buggy report builder cannot freeze a script that would otherwise exit. - -### Tests - -- `tests/test_zero_activity_diagnostic.py` (new, 6 tests) — pins the warn-once / threshold / concurrent-bump / message-content invariants on the zero-activity diagnostic. -- `tests/test_dev_error_report.py` (new, 11 tests) — pins the four-line / what / where / why / how-to-fix / docs-URL invariants on the new error report. -- `tests/test_protect_only_public_api.py` (new, 9 tests, landed in `fcd623c`) — pins the auto-attach / chain-walk / bare-`@sensitive`-deprecation invariants on the `@protect` contract change. -- `tests/test_protect.py`, `tests/test_preflight_fail_policy.py`, `tests/test_protect_cancel_on_exception.py` — `_RecordingRuntime` stubs extended with a `_bump_protect_count` no-op so the existing gate / cancel / span test surface keeps working unchanged. - -### Verification - -- `ruff check src tests` — all checks passed. -- `mypy src/nullrun` — success: no issues found in 37 source files. -- `pytest -q` — **1840 passed, 4 skipped, 14 warnings** in ~128s (vs 0.18.0 baseline of 1814 passed, 4 skipped — +26 new tests: 9 from `fcd623c`, 11 from `test_dev_error_report.py`, 6 from `test_zero_activity_diagnostic.py`). -- `nullrun.__version__` — `0.18.1`. -- Scratch diff — clean (no `dist_local/`, no `*.defect*`). -- Wire-format — unchanged. SDK_MIN_VERSION — unchanged. - -### Why this is needed - -The 0.18.0 release shipped with three structural silent-failure modes that compound in real-world agent deployments: (a) the user added `@protect` but never called `init_or_die()`, so no `track` events were ever emitted and the dashboard reported zero tokens; (b) the user called `__init()` (or `init_or_die()`) but then ran a tool via a raw `httpx.Client` that wasn't on the patchable surface, again with zero telemetry; (c) the user hit a config failure (missing `NULLRUN_API_KEY`) at the first gate call and saw only end-user wording, with no hint of what failed or where to look. Production traces showed each of these fire as a "NullRun does nothing" support ticket within the first week of a new deployment. The four changes in 0.18.1 close the three modes simultaneously: `@protect` auto-instruments lazily so (a) cannot occur; the zero-activity diagnostic surfaces (b) with the three most likely root causes; the four-line dev report turns (c) from a black-box exit into a self-serviceable error. - -The dead `[openai] / [anthropic] / [mistral] / [gemini] / [cohere] / [bedrock]` extras were carry-overs from an earlier auto-instrumentation plan that targeted vendor SDKs; the HTTP-level path via `patch_httpx` covers all six without any vendor SDK. Removing the extras shrinks the install footprint for the 90%+ of users who use the HTTP path. Framework extras stay because NullRun subscribes to framework event hooks (LangGraph `Pregel`, LangChain `BaseCallbackManager`, OpenAI Agents `Runner`, LlamaIndex `get_dispatcher`, CrewAI event bus, AutoGen `BaseChatAgent`). - -`toolbox.langgraph.wrapper()` pre-dates the auto-patch and remains useful for the three narrow escape-hatch cases documented in its docstring (tests with custom runtimes, Pregel imported before init, manual callback control). Deprecating it in the docstring is the right move — most users no longer need to call it. - -## [0.18.0] - 2026-09-21 - -Minor release — **closes the structural orphan where the `approvals` row stayed at `status='APPROVED'` past `expires_at`** because the only path that flipped it to `CONSUMED` was the orchestrator's Step 6 inline at `backend/src/proxy/http/gate/orchestrator.rs:713`, which `mode="inline"` tools bypass entirely. The fix wires the SDK to call a new structurally-distinct endpoint (`POST /api/v1/approvals/{approval_id}/consume`) from both the success path (after WS approval resolves to `outcome=approved`) and the exception path (`_safe_cancel_active_execution`). Operator-initiated `/cancel` on an approval envelope ALSO consumes the row in spawned Step 4e. Audit emits distinguish operator-cancel from SDK-consume via distinct `matched_rule` strings. Behaviour change: outbound HTTP call from the SDK success branch (`check_workflow_budget` after WS approval). Wire-format unchanged (additive). SDK_MIN_VERSION unchanged. - -### Added - -- **`POST /api/v1/approvals/{approval_id}/consume`** (backend 2.8.0+). Sibling to `/api/v1/cancel`. Structurally distinct from `consume_approved`: the SQL omits `execution_id` binding per ADR-046 (no cached-replay arm race window), carries `organization_id` filter (C2 closure), and optionally filters by `api_key_id` (P1-A cross-key replay defense). Three response shapes: `consumed` (success), `already_consumed` (idempotent replay), `not_approved` (PENDING/DENIED/EXPIRED — idempotent no-op). All three return 200. See `backend/src/proxy/http/approvals.rs::consume_approval_handler` and ADR-047. - -- **`Runtime.consume_approval(approval_id, execution_id=None)`** — best-effort POST to the new endpoint. Catches all exceptions and surfaces them at `logger.debug` so a network blip does NOT block the success path or the exception path. Returns `{"status": "error", "approval_id": ...}` on failure (so callers can branch if they care). - -- **`Runtime.lookup_pending_approval_id_for_execution(execution_id)`** — read-only accessor for the reverse-index `execution_id → approval_id` populated by `check_workflow_budget` on the WS-approval success branch. RLock-guarded. Used by `_safe_cancel_active_execution` to close orphan grants on the exception path. - -- **`Runtime._mark_approval_resolved_for_execution(execution_id, approval_id)`** — internal writer for the same reverse index. RLock-guarded. Invoked from `check_workflow_budget` after `outcome == "approved"`. - -### Changed - -- **`check_workflow_budget` auto-calls `consume_approval`** on the `outcome=approved` branch after WS approval resolves (`src/nullrun/runtime.py`). Best-effort: the helper catches all exceptions and surfaces them at `logger.debug`, so a network blip here does NOT block the success path. The `approval_expiry_sweeper` at `backend/src/workers/approval_expiry.rs` will close any rows that slip through within ~5 min, but for `mode="inline"` non-sensitive tools (the dominant class), the success path now closes the row before the operator sees it on the dashboard. - -- **`_safe_cancel_active_execution` ALSO calls `consume_approval`** after `cancel_execution` on the exception path (`src/nullrun/decorators.py`). The reverse-index lookup is RLock-guarded; if the SDK crashed before reaching the WS-approval branch the lookup returns `None` and this is a no-op. Same best-effort posture as `cancel_execution` itself — never masks the original exception. - -- **`/cancel` mode-gated Step 4e** consumes the approval row (backend, spawned task). Audit emit uses `matched_rule="lifecycle.consume_approved_via_cancel"` + `consume_reason="operator_cancel"`, distinct from the SDK-side `lifecycle.consume_approved_via_sdk` + `sdk_consume_endpoint`. The two `matched_rule` strings give operators forensic distinction between operator-cancel and SDK-side auto-consume on the audit table. - -### Why this is needed - -Production had 48 `approvals` rows in `status='APPROVED' + consumed_at IS NULL + expires_at < NOW()` — the dashboard's "abandoned grants" view (per ADR-045 §10.1 retraction). Root cause: `consume_approved` SQL at `backend/src/proxy/infra/db.rs:1715` was only reachable from `/execute` orchestrator Step 6, but `mode="inline"` tools bypass `/execute` entirely. The same gap fired on SDK crashes between WS approval push and body execution. Recent SDK releases (v0.16.5 cancel-on-exception, v0.16.8 NR-A015, v0.17.0/0.17.1) closed the budget reservation leak and the sensitive-tool wire-shape gap but NOT this structural orphan. `_safe_cancel_active_execution` (added in v0.16.5) calls `POST /api/v1/cancel`, which only releases the B-10 pending counter + DELs the Redis reservation key — it does NOT touch the `approvals` row. The `approval_expiry_sweeper` at `backend/src/workers/approval_expiry.rs` is PENDING-only by design (ADR-045 §10.1 retraction locked this in) and never transitions APPROVED rows. - -The orphan class shrinks from "every inline-mode approval" to "transport blip during success path". The 48 historical rows will close on the sweeper's next pass; no data migration needed. New inline-mode rows close on the SDK success path; new cancel-path rows close on the operator-cancel path. The remaining failure surface is "SDK crashes between WS approval resolve and `consume_approval` HTTP landing", which the sweeper still handles within ~5 min. - -### Verification - -- `ruff check src tests` — all checks passed. -- `mypy src/nullrun` — success: no issues found. -- `pytest -q` — full SDK suite green; 6 new tests added in `tests/test_v3_wire_contract.py::TestConsumeApprovalEndpoint`, `TestCheckWorkflowBudgetConsumeOnApproved`, `TestSafeCancelCallsConsumeApproval`. -- Wire-format smoke: `pytest -q tests/test_v3_wire_contract.py -k "consume_approval or safe_cancel or protocol_header"`. - -### Why two `matched_rule` strings - -Operators investigating the `audit_events` table for an "approved → consumed" transition need to know whether the consume came from the SDK auto-consume (meaning the agent successfully executed the approved tool) or from the operator's `/cancel` (meaning the operator cancelled the approval envelope). Conflating them in a single `matched_rule` string would lose this distinction, so we emit two: - -- `lifecycle.consume_approved_via_sdk` + `consume_reason="sdk_consume_endpoint"` — SDK side, success path -- `lifecycle.consume_approved_via_cancel` + `consume_reason="operator_cancel"` — operator-initiated cancel - -Both share the same `audit_kind` (`approval.consumed`) and `status` (`success`), so the `approvals` lifecycle view aggregates them correctly while the audit deep-dive filters on `matched_rule` + `consume_reason`. - -## [0.17.1] - 2026-09-15 - -Patch release — two correctness themes on the 0.17.0 baseline: (1) **`/check` mints a fresh `operation_id` per call** (the previous behaviour — reuse the first call's op_id within the same scope — collided with the backend's `IDEM-01` 11-field semantic-hash dedup whenever a second `/check` had a different `tools` / `model` / `input`, surfacing as a 409 `IDEMPOTENCY_KEY_MISMATCH` and the misleading SDK error `NR-B004` "You've reached the usage limit for this conversation"), and (2) **`_V3_ERROR_CODE_MAP` closes the wire-code gap from backend `fix-wave-2`** (the two NEW wire codes — `INVALID_JSON` (400, `invalid_json` slug, `JsonSyntaxError`) and `INVALID_FIELD` (422, `validation_error` slug, `JsonDataError`) — now round-trip through `NullRunBackendError` instead of falling through to the generic transport fallback at `transport.py:2961`). No behaviour change for code that already handles `NullRunBackendError`; cookbook recipes that branch on `error_code` now retain diagnostic class for parse-level vs schema-level rejections. Wire-format unchanged. SDK_MIN_VERSION unchanged. - -### Fixed - -- **DEF-OPID-REUSE-HASH-MISMATCH** — `NullRunRuntime._check_workflow_budget_impl` now always mints a fresh `operation_id` per `/check` call (`src/nullrun/runtime.py`, `a05726e`). Pre-fix the helper read `_operation_id_var` once and stashed the minted UUID v4 into the contextvar; subsequent `/check` calls within the same scope reused the first call's op_id. The backend's `IDEM-01` dedup (`compute_gate_semantic_hash` in `backend/src/redis/idempotency_store.rs:125-140`) keys on `operation_id` but verifies an 11-field semantic hash (`operation_id`, `tools`, `tool`, `mode`, `check_type`, `model`, `estimated_tokens`, `input`, `business_impact`, `workflow_id`, `organization_id`); a second `/check` with different `tools` / `model` / `input` on the SAME `op_id` therefore 409 `IDEMPOTENCY_KEY_MISMATCH`, surfaced as `NR-B004` in the SDK. Canonical repro: `nullrun_openai_approval_demo.py` fires a `tools=None` `/gate` (LangGraph `NullRunCallback.on_llm_start`) BEFORE the `@sensitive(refund_customer)` `/gate` (`tools=['refund_customer']`); both `/check` calls share the same op_id, the second's semantic hash diverges from the first's, server rejects with 409, SDK reports `NR-B004` "You've reached the usage limit for this conversation". Fix: `/check` always reads-and-discards the contextvar (value unused), then unconditionally mints a new UUID v4 and stashes it. `/execute` at `runtime.py:3048-3051` reads the freshly-stashed value within the same logical action (synchronous `/check` → `/execute` chain), so the **P0-27 within-action binding** (one op_id across `/check` + `/execute`) is preserved. The read-then-overwrite pattern also keeps the **P0-27 source-pin test** `test_check_workflow_budget_reads_contextvar` green. Verified: 8/8 P0-27 source-pin tests pass (`test_audit_p0_27_operation_id_hoist.py`); 1807 pytest pass / 4 skipped / 0 fail (full SDK suite); live probe (`probe_full.py`) emits 4 distinct operation_ids across 3 `/check` + 1 `/execute`; `/execute` fallback mint pattern unchanged (read contextvar first, mint only if None); `_GATE_CACHE` invariant unaffected (cache key doesn't include op_id); backend `IDEM-01` logic unchanged (`compute_gate_semantic_hash` unaffected). - -- **DEF-SDKT-004 fix-wave-2** — `_V3_ERROR_CODE_MAP` now contains entries for `INVALID_JSON` and `INVALID_FIELD` (`src/nullrun/transport.py`, `f5aca80`). Backend `fix-wave-2` (2026-09-13) split `From for ApiError` onto three distinct wire codes: `JsonDataError` → 422 + `INVALID_FIELD` (slug `validation_error`), `JsonSyntaxError` → 400 + `INVALID_JSON` (slug `invalid_json`), `MissingJsonContentType` → 415 + `INVALID_INPUT` (slug `bad_request`). The two NEW codes (`INVALID_FIELD`, `INVALID_JSON`) are emitted on `/gate`, `/execute`, and `/track`. Pre-fix the SDK's `_V3_ERROR_CODE_MAP` had no entries for them, so they fell through to the generic `NullRunBackendError` fallback at `transport.py:2961`; cookbook recipes that branch on `error_code` lost diagnostic class for parse-level vs schema-level rejections. Map both to `NullRunBackendError` — siblings to `EXECUTION_ID_MALFORMED`, `EXECUTION_ID_REQUIRED`, `INVALID_EXECUTION_ID`, and `IDEMPOTENCY_REDIS_UNAVAILABLE` which already follow the same pattern for wire-shape parsing failures. This mirrors the backend's intent: wire-level parsing failures are infrastructure-side issues and the SDK round-trips them through the generic catch-all. The new wire codes are exercised in: `/gate` POST body rejection (`gate.rs`); `/execute` POST body rejection (`execute.rs`); `/track` POST body rejection (`handlers.rs`); `TC-SDKG-006` (truncated JSON) expects 400 + `INVALID_JSON`. **NR-007a** (new) at `backend/tests/nr007_sdk_error_code_parity.rs` pins the required SDK mappings and will fail CI on future drift (e.g., if someone reverts `INVALID_JSON` from this map). - -### Verification - -- `ruff check src tests` — all checks passed. -- `mypy src/nullrun` — success: no issues found in 37 source files. -- `pytest -q` — **1807 passed, 4 skipped** in ~102s (no new tests — both fixes fold into existing coverage; baseline 1807 at 0.17.0). -- `nullrun.__version__` — `0.17.1`. -- Scratch diff — clean (no `dist_local/`, no `*.defect*`). - -### Why this is needed - -**op_id reuse (`DEF-OPID-REUSE-HASH-MISMATCH`)** — pre-fix the SDK reused the first `/check`'s op_id for every subsequent `/check` in the same scope. The backend's `IDEM-01` dedup keys on op_id but verifies an 11-field semantic hash; any divergence (different `tools`, `model`, `input`) on the SAME op_id produces a 409 that surfaces to the SDK as `NR-B004` "You've reached the usage limit for this conversation" — a wildly misleading message for what is actually a per-call idempotency violation. This is a recurring foot-gun rather than an active bypass: most agent loops issue `/check` calls with the same tool / model / input on the same op_id and never trip the dedup, but the moment a second call diverges (very common — LangGraph fires a `tools=None` gate before the tool-scoped `@sensitive` gate; multi-tool agents fire distinct tool gates per tool), the dedup fires and the SDK reports a usage-limit error that has nothing to do with the actual budget. The fix collapses op_id scope from "scope" to "single `/check` invocation", mirroring the backend's invariant that op_id is per-call, not per-scope. `/execute` continues to read the freshly-stashed op_id within the same logical action so the P0-27 within-action binding (`/check` + `/execute` share op_id) is preserved — verified by the source-pin regression tests at `tests/test_audit_p0_27_operation_id_hoist.py`. - -**Error-code map closure (`DEF-SDKT-004 fix-wave-2`)** — the backend's `fix-wave-2` split `From for ApiError` onto three distinct wire codes so operators can distinguish `JsonDataError` (422 schema-level) from `JsonSyntaxError` (400 parse-level) from `MissingJsonContentType` (415 missing-content-type). The SDK's `_V3_ERROR_CODE_MAP` is the single point of truth for "what exception class does the SDK raise when the backend returns this `error_code`". Pre-fix the map covered the legacy wire codes (`EXECUTION_ID_MALFORMED`, `EXECUTION_ID_REQUIRED`, `INVALID_EXECUTION_ID`, `IDEMPOTENCY_REDIS_UNAVAILABLE`) but did not cover the two NEW codes from the backend split, so they fell through to the generic `NullRunBackendError` fallback at `transport.py:2961`. Cookbook recipes that branch on `error_code` (e.g., to retry on schema-level but not parse-level rejections) lost diagnostic class. The fix maps both new codes to `NullRunBackendError` — the same exception class used for the legacy wire-shape parsing failures, matching the backend's intent that wire-level parsing failures are infrastructure-side issues. NR-007a (new) at `backend/tests/nr007_sdk_error_code_parity.rs` pins the required SDK mappings so any future revert (e.g., removing `INVALID_JSON` from the map) fails CI on the SDK side. - -## [0.17.0] - 2026-09-12 - -Minor release — four correctness themes on the 0.16.x baseline: (1) **chain-setter Token discipline** (`set_chain_id` / `set_chain_op` now return the `Token` minted by `ContextVar.set()`, matching the rest of the manual-setter surface — silent audit-trail bleed across calls is closed), (2) **`_GATE_CACHE` staleness closure** (invalidate the gate cache on consume-side 402/422 + on `chain_end` so a stale "allow" cannot serve un-budgeted tool execution within the 5s cache window), (3) **lazy-export repair** (`nullrun.money_outflow`, `nullrun.tool_params`, `nullrun.business_impact` are now reachable as attributes on `nullrun` — the documented `@nullrun.sensitive(impact=money_outflow(...))` pattern no longer crashes with `AttributeError`), and (4) **circuit-breaker lock unification** (sync + async paths now serialise on a single `threading.Lock`, closing a sync↔async race that let `self._state` mutate concurrently when one thread called `breaker.call(sync_fn)` and another coroutine called `await breaker.call(async_fn)`). **Behaviour change** for callers using the manual `set_chain_id` / `set_chain_op` escape-hatch — the return value is now a `Token`, not `None`. Wire-format unchanged. SDK_MIN_VERSION unchanged. - -### Fixed - -- **DEF-CHAIN-SETTER-NO-TOKEN** — `nullrun.set_chain_id(chain_id)` and `nullrun.set_chain_op(op)` now return the `Token` minted by the underlying `ContextVar.set()` call (`src/nullrun/__init__.py`, `src/nullrun/context.py`, `8e7e070`). Mirrors the existing discipline on `set_trace_id` / `set_span_id` / `set_operation_id` / `set_server_minted_execution_id`. Callers using the manual-setter API outside the `with chain(...)` contextmanager can now restore the prior value via `ctx.reset(token)`, closing the silent audit-trail bleed where chain_id attribution leaked forward into subsequent unrelated `/check` calls on the same event-loop task slot. Docstrings updated to call out the Token contract. - - **Back-compat**: callers that ignored the previous `None` return continue to work; the only observable change is the new `Token` return value (assignable to a local variable). Recurring foot-gun, not an active bypass — CPython asyncio ContextVar is per-task, so cross-task leak requires unusual patterns (`asyncio.shield + manual context copy`); only the manual-setter escape-hatch path was affected. Also tightens `_LAZY_EXPORTS` dict annotation from `tuple[str, str]` to `tuple[str, str | None]` and branches `__import__(...)` on the `attr_name` sentinel — both pre-existing mypy errors surfaced after the Token discipline fix was added. - -- **DEF-CACHE-STALE-ALLOW-AFTER-OVERBUDGET** — `_route_track` now calls `_invalidate_gate_cache_for_chain(workflow_id, chain_id)` when `track_single` raises `HTTPStatusError(402)` or `HTTPStatusError(422)` (`src/nullrun/runtime.py`, `f40b5cf`). The server's authoritative budget check has just said no; the in-process gate cache can no longer serve a stale "allow" for up to 5 s after that decision. `chain_end()` also invalidates after the wire call succeeds — the chain is closed on the server, the in-process cache for that chain is no longer reachable. Cache key now includes `estimated_tokens` (currently always 1 in `check_workflow_budget`) to future-proof against a refactor that varies cost estimates by call (DEF-CACHE-COST-ESTIMATE-COLLISION). - - **Failure scenario** (pre-fix): agent in a tight `for tool in chain` loop hammers `call_model="gpt-4"`; budget red-lines at `t=0`; the next `/gate` in the same chain within 5 s returns the cached allow without re-hitting the server; the tool executes un-budgeted; the consume on the next iteration hits `CONSUME_OVERBUDGET`. Post-fix the cache is invalidated on the 402/422, so the next `/gate` re-runs the budget check and gets the fresh rejection. Lua `reserve_v3.lua:306-333` already enforces chain state correctly (backend `fix-consume-binding-org-key-mismatch`); this commit is the SDK-side analog. - -- **DEF-CACHE-CHAIN-INVALIDATION-SCOPE** — correctness followup to `DEF-CACHE-STALE-ALLOW-AFTER-OVERBUDGET`: `_invalidate_gate_cache_for_chain` now reads `chain_id` from the contextvar (via `get_chain_id()`) rather than from `wire_event.get('chain_id')` (`src/nullrun/runtime.py`, `18f4bda`). The `wire_event` dict is the per-call track payload built from `_enrich_event`; `chain_id` is never populated there. Pre-fix the helper passed `chain_id=None` to the dict-iteration loop and dropped EVERY chain entry for the same `workflow_id` — safe in the sense that stale-allow wasn't served (the parent fix holds), but it leaked cache pressure onto unrelated chains under sustained traffic and made the cache effectively useless when multiple chains share a `workflow_id`. Post-fix mirrors the read pattern in `check_workflow_budget` (`runtime.py:2013`) and `chain_end` (`runtime.py:2507 via get_trace_id`). - -- **DEF-LAZYEXPORT-MONEY-TOOL-PARAMS** — `nullrun.money_outflow` and `nullrun.tool_params` are now reachable as top-level attributes on `nullrun` (`src/nullrun/__init__.py`, `b977537`). The documented `@nullrun.sensitive(impact=money_outflow(...))` pattern (referenced at `decorators.py:1113-1132`, `extractor.py:43`) crashed with `AttributeError: module 'nullrun' has no attribute 'money_outflow'` on first invocation — `__getattr__` masked any name not in `_LAZY_EXPORTS`, and the two impact-extractor helpers were never added to the table. Workaround `from nullrun.extractor import money_outflow` still works. - -- **DEF-LAZYEXPORT-BUSINESS-IMPACT** — `nullrun.business_impact` is now reachable as a submodule attribute on `nullrun` (`src/nullrun/__init__.py`, `6601208`). The docstrings at `extractor.py:18` and `extractor.py:799-801` reference `nullrun.business_impact.compute_action_digest` and `ToolCallParams` as bare dotted paths. The module is real (`nullrun/business_impact.py`) and contains those symbols, but PEP 562 `__getattr__` masked submodule access — `nullrun.business_impact` raised `AttributeError` from a fresh import even though `import nullrun.business_impact` worked. Fix adds `business_impact` to `_LAZY_EXPORTS` with `attr_name=None` sentinel; `__getattr__` returns the imported module itself instead of `getattr(module, attr_name)`. No-op for existing per-symbol re-exports (`money_outflow`, `tool_params`) — the sentinel branch only fires when `attr_name is None`. - - **Post-fix probe**: - ```python - >>> nullrun.business_impact.compute_action_digest - - ``` - -- **DEF-CB-LOCK-UNIFICATION-2026-09-12** — `NullRunCircuitBreaker` now serialises sync + async critical sections on a single `threading.Lock` (`src/nullrun/breaker/circuit_breaker.py`, `77bf38b`). Pre-fix the sync path held `self._lock` (`threading.Lock`) and the async path held a separate `asyncio.Lock` (`_async_lock`, lazy-init via `_get_async_lock`); on the same breaker instance a sync thread calling `breaker.call(sync_fn, ...)` and an async coroutine calling `await breaker.call(async_fn, ...)` could both write `self._state` concurrently — the two locks provided no mutual exclusion across the sync↔async boundary. The async critical sections (`_on_failure_async`, `_on_success_async`) contain no `await` between attribute writes; with asyncio's single-threaded execution model those sections are already atomic under the GIL+scheduler — the `_async_lock` was dead weight providing no additional exclusion. Fix: removed `_async_lock` and `_get_async_lock`; both paths now use `self._lock`. `async with self._lock` blocks the event loop for zero observable time on the happy path (no `await` inside the critical section). Trade-off: sync+async exclusion > minor event-loop contention under high contention (zero under normal traffic). Closes the silent `self._state` write race that could let a thread observe a half-updated breaker state — under a tight async loop with one stray sync caller this manifests as flaky `breaker.call` returns (one path thinks the breaker is open, the other thinks it's half-open). - -### Added - -- **`tests/test_v3_wire_contract.py::TestGateCache::test_invalidate_drops_only_matching_chain`** (`18f4bda`). Regression pin for `DEF-CACHE-CHAIN-INVALIDATION-SCOPE`: sets two cache entries for the same `workflow_id` but different `chain_id`s, marks one chain overbudget, and asserts only the overbudget chain's entry is dropped. Forbids re-introducing the pre-fix `wire_event.get('chain_id')` lookup that silently passed `chain_id=None` and dropped every chain. -- **`tests/test_v3_wire_contract.py`** test updates for `DEF-CACHE-STALE-ALLOW-AFTER-OVERBUDGET` (`f40b5cf`): existing cache tests now use 4-tuple keys (`workflow_id`, `chain_id`, `call_model`, `estimated_tokens`) — the `estimated_tokens` arm was added in the same commit and is a future-proofing pin. -- **`tests/test_circuit_breaker_branches.py`** — 10 new branch tests for `DEF-CB-LOCK-UNIFICATION-2026-09-12` (`77bf38b`): covers both the sync and async `breaker.call` paths through a single `threading.Lock` (state transitions, failure-count accumulation, half-open probe, open→closed reset on success, async-context contention with the same shared lock). Pins that no future refactor can re-introduce a separate `_async_lock` without tripping these tests. - -### Verification - -- `ruff check src tests` — all checks passed. -- `mypy src/nullrun` — success: no issues found in 37 source files. -- `pytest -q` — **1807 passed, 4 skipped** in ~102s (10 new tests from `DEF-CB-LOCK-UNIFICATION-2026-09-12` circuit-breaker branch coverage — baseline 1797 at 0.16.8). -- `nullrun.__version__` — `0.17.0`. -- Scratch diff — clean (no `dist_local/`, no `*.defect*`). - -### Why this is needed - -**Token discipline (DEF-CHAIN-SETTER-NO-TOKEN)** — pre-fix both setters did `return None` after calling the contextvar's `set()`, breaking the Token discipline every other setter in the module honours. Callers using the manual API outside the `with chain(...)` contextmanager could not restore the prior value via `ctx.reset(token)`, so the `chain_id` leaked forward into subsequent unrelated `/check` calls on the same event-loop task slot — silent audit-trail corruption (chain_id attribution bleeds across calls). The leak is a recurring foot-gun rather than an active bypass (CPython asyncio ContextVar is per-task), but the manual-setter escape hatch is documented and used; the fix brings the API surface in line with the rest of the setter family. - -**Cache staleness (DEF-CACHE-STALE-ALLOW-AFTER-OVERBUDGET)** — at anti-DoS scale, every cached "allow" served after the budget red-line is a tool execution that bypasses the gate. The 5-second cache TTL amplifies this: a single tight loop on `call_model="gpt-4"` can consume thousands of tool invocations against a budget the server already said no to. The fix collapses the cache-staleness window to "synchronous invalidation on 402/422" (~ms), matching the `reserve_v3.lua` defense-in-depth on the consume side. - -**Chain-scope invalidation (DEF-CACHE-CHAIN-INVALIDATION-SCOPE)** — pre-fix the helper matched on `chain_id=None` and dropped every chain entry for the workflow, making the cache effectively useless under multi-chain traffic. This wasn't an active bypass (the parent fix holds), but it meant operators under load saw cache eviction across unrelated chains whenever one chain hit 402/422. Surfaced 2026-09-12 by post-`DEF-CACHE-STALE-ALLOW-AFTER-OVERBUDGET` code review: the wire-event lookup was the kind of mistake that's invisible under single-chain testing but devastating in production. - -**Lazy exports (DEF-LAZYEXPORT-MONEY-TOOL-PARAMS + DEF-LAZYEXPORT-BUSINESS-IMPACT)** — the documented `@nullrun.sensitive(impact=money_outflow(...))` pattern is the headline use-case for `@sensitive` decoration, so crashing with `AttributeError` on first invocation is a textbook "the documented example doesn't work" regression. The `business_impact` submodule crash was similar: docstring-referenced dotted paths resolved to `AttributeError`. Both are pre-existing typing-errors that became loud after the PEP 562 lazy-export pattern was introduced (`money_outflow` / `tool_params` found via TC-12 strict verification 2026-09-12 against prod; `business_impact` found via the docstring-referenced path audit). - -**Circuit-breaker lock unification (DEF-CB-LOCK-UNIFICATION-2026-09-12)** — pre-fix the breaker held `self._lock` (`threading.Lock`) for the sync path and `_async_lock` (`asyncio.Lock`) for the async path, with no cross-path exclusion. Under a mixed sync+async workload (sync thread + async loop on the same breaker instance — common in HTTP servers where a worker thread guards against an outage while the event loop is forwarding the same breaker state to inbound WebSocket frames), `self._state` could mutate concurrently: one path read `state="half_open"` and started the probe; the other wrote `state="open"` after a failure; the probe saw a half-updated state and either double-allowed (open→closed transition observed in the gap) or never reset. Both failure modes are silent — no exception, just wrong breaker decisions on a fraction of calls. The async critical sections contain zero `await` calls (state writes only), so asyncio's single-threaded scheduler already serialises them; the `_async_lock` was dead weight. Unifying on `threading.Lock` for both paths costs nothing on the happy path (no event-loop blocking, since there's no `await` in the section) and gives sync↔async exclusion for free. Surfaced 2026-09-12 by a code-review pass on the v0.17.0 theme set. - -## [0.16.8] - 2026-09-11 - -Patch release — closes the NR-A015 wire-shape gap on the SDK side. The -`/execute` `require_approval` arm (backend v3.79+) mints a fresh -server-side execution_id for the approval row and echoes it via -`reservation_id`. Pre-fix `runtime.execute` captured that id into the -contextvar AFTER `/gate` calls but not after `/execute`, so the post- -approval `/execute` re-fire sent the stale pre-arm execution_id; -`consume_approved`'s `WHERE execution_id = $3` predicate missed the -freshly-stamped row and fell through to the terminal -`APPROVAL_REPLAY_REJECTED` branch. This release wires the -post-`/execute` capture and syncs the kwargs dict so the re-fire uses -the freshly-minted id. - -Also includes the AUTH-01 / HEART-01 sweep from the 24-driver pass: -reclassification of httpx transport errors on the auth path, plus a -public `Runtime.heartbeat()` wrapper. - -### Fixed - -- **DEF-EXECUTE-CAPTURE-WIRING** — `runtime.execute` now calls - `_capture_server_minted_execution_id(result)` immediately after - `_transport.execute(...)` and syncs the re-fire kwargs dict to the - captured id (`src/nullrun/runtime.py`). The post-approval re-fire - now sends the freshly-minted execution_id stamped on the approval - row, so `consume_approved`'s `WHERE execution_id = $3` predicate - matches. Closes the SDK-side leg of NR-A015 on the `/execute` - require_approval arm. -- **DEF-WAIT-FOR-APPROVAL-EXEC-ID** — both - `_wait_for_approval_resolution` call-sites (`check_workflow_budget` - + `runtime.execute`) now pass the captured server-minted - execution_id from the contextvar (with the prior - `org_id`/`workflow_id` sentinel as fallback) instead of the - `workflow_id` sentinel (`src/nullrun/runtime.py`). Diagnostic - improvement only — the WS handler matches on `approval_id` — but - log lines + entry metadata now reflect the server-minted id. -- **DEF-AUTH-01** — `NullRunRuntime.__init__` auth path no longer - reclassifies `httpx.RequestError` as `NullRunAuthenticationError`. - The defensive duplicate arm in `__init__` (backstop for a code path - that no longer exists) is removed; the real arm in `_authenticate` - now raises `NullRunTransportError(source=NETWORK_ERROR, endpoint="auth")` - — matching the convention used by `Transport.heartbeat` for the - same condition on `/heartbeat`. The previous wrap misled operators: - a network failure looked like an auth failure, even though the - message itself acknowledged "this is a transport failure (not an - auth failure)". - - **Back-compat**: `NullRunTransportError` and `NullRunAuthenticationError` - are siblings under `NullRunInfrastructureError`, so the parent class - still catches both. Cookbook code that branches on - `except NullRunAuthenticationError:` for retry will need to also - catch `NullRunTransportError`. Two existing tests - (`test_authenticate_network_error_raises` in `test_runtime.py` and - `test_runtime_branches.py`) were locking in the old misclassification - and have been updated to assert the correct class. - -### Added - -- **DEF-HEART-01** — `NullRunRuntime.heartbeat(chain_id)` public method - added (thin forwarder to `Transport.heartbeat`). Mirrors the - `chain_end` / `cancel_execution` pattern. Use for single-shot chain - TTL extensions; `Runtime.ping_chain()` remains the wall-clock - scheduler variant. Pure addition — no existing API surface changes. - -- **`tests/test_2026_09_11_execute_capture_wires_execution_id.py`** (220 lines). Two regression tests pinning the fix: - - `test_execute_captures_reservation_id_from_response` — verifies the contextvar updates from the `/execute` response and the re-fire uses the captured id (not the stale pre-call one). - - `test_execute_wait_for_approval_receives_captured_eid` — verifies the WS resolution handler receives the captured execution_id. -- **`tests/test_2026_09_11_auth_heartbeat_sweep.py`** (~190 lines, 7 tests). Regression tests for AUTH-01 + HEART-01: - - AUTH-01: `test_auth_connect_error_raises_transport_error_not_auth`, `test_auth_timeout_raises_transport_error_not_auth`, `test_auth_error_class_no_longer_catches_network_error` (back-compat parent-class check). - - HEART-01: `test_heartbeat_method_exists_on_public_api`, `test_heartbeat_forwards_chain_id_to_transport`, `test_heartbeat_passes_through_transport_error`, `test_ping_chain_still_works_after_heartbeat_added`. - -### Compatibility - -Pure reliability fix — no wire-format change. `/gate`, `/execute`, -`/track`, `/cancel` payloads are byte-identical to 0.16.7. Backend -v3.79+ is required for the wire-shape contract (the `reservation_id` -echo is the v3.79+ field that closes the gap); pre-v3.79 backends -silently fall through the capture (helper is fail-OPEN on malformed -values), preserving the pre-fix behaviour for un-deployed backends. - -### Why this is needed - -**NR-A015 (execute capture)** — the user-facing symptom was a -post-approval `/execute` re-fire landing on `APPROVAL_REPLAY_REJECTED` -because the SDK stamped the pre-arm `execution_id` into the -re-fire's kwargs dict, but `consume_approved`'s `WHERE execution_id -= $3` predicate had to match the freshly-minted id from the -approval-row bind (backend v3.79+). The terminal error was a -typed `NullRunApprovalReplayRejectedError(NR-A015)` — operators had -no signal that the re-fire was sending a stale id rather than a -truly-replayed call. 0.16.8 captures the `reservation_id` echo -from `/execute`'s response into the same contextvar that `/gate` -already uses, and re-emits the captured id on the re-fire kwargs -dict. - -**AUTH-01 (transport reclassification)** — operators reading -`NullRunAuthenticationError` from a failed `__init__` were led to -rotate the API key because the class name suggested auth failure. -Pre-fix, the duplicate arm in `NullRunRuntime.__init__` rewrapped -`httpx.RequestError` as `NullRunAuthenticationError` (with the -"this is a transport failure (not an auth failure)" wording in the -message itself — a smoke signal the wrap was wrong). 0.16.8 raises -`NullRunTransportError(source=NETWORK_ERROR, endpoint="auth")` -matching the `Transport.heartbeat` convention. Catch-block semantics -in cookbooks now need `except (NullRunAuthenticationError, -NullRunTransportError):` for full coverage under -`NullRunInfrastructureError`. - -**HEART-01 (public API)** — single-shot chain TTL extensions had to -reach through `runtime._transport.heartbeat(...)` because -`NullRunRuntime` exposed only the wall-clock `ping_chain()` scheduler. -0.16.8 adds `NullRunRuntime.heartbeat(chain_id)` as a thin -forwarder to `Transport.heartbeat`, matching the chain_end / -cancel_execution pattern. - -## [0.16.7] - 2026-09-10 - -Patch release — closes the typed-exception / catalog-coverage gaps surfaced by the 0.16.6 backend hardening. After that release, every catalog exception the SDK can raise now has a hand-written `DEFAULT_MESSAGES` entry (no more "Something went wrong. Please try again." fallback), and `@protect`-decorated sites surface the real exception type instead of rewriting it into a generic `NullRunBlockedException`. The `@protect` block path in `runtime.execute` now dispatches the actual catalog code through `format_user_message`, so wire-error codes (NR-A012, NR-A016, NR-EX01, …) reach users with actionable wording. No wire-format change. - -### Fixed - -- **DEFS-SDKEXEC-TYPED-DISPATCH** — `runtime.execute` block path raises the catalog exception itself (NR-A016 etc.) instead of the generic fallback (`src/nullrun/runtime.py`, `2e77902`). `exc.error_code` carries the catalog code, so `format_user_message` finds actionable wording; downstream sites that inspect `exc.details['details']['mapped_class']` see the typed class name (e.g. `NullRunApprovalDbUnavailableError`) rather than the base `NullRunBlockedException`. -- **DEFS-SDKEXEC-BLOCK-PIN** — `tests/test_runtime.py::test_execute_blocked_surfaces_wire_error_code` pinned to the new typed-dispatch contract (`834d9ea`, `DEF-NR-RUNTIME-BLOCK-TYPED`): the wire payload (`error_code` + `mapped_class`) is preserved verbatim while the SDK exception is now the typed class. This was the last stale wire-code assertion in `test_runtime.py` blocking full SDK pass under the post-0.16.6 catalog contract. -- **DEFS-SDKPROTECT-EX01-PASSTHROUGH** — `@protect`-decorated `_enforce_sensitive_tool` no longer rewrites `NullRunExecutionNotFoundError` (NR-EX01) into `NullRunBlockedException(NR-B002)` (`src/nullrun/decorators.py`, `257ab7f`). The typed class, `error_code`, `execution_id`, `regate_required`, and the NR-EX01 user-facing line from `format_user_message` all propagate unchanged; pass-through arm is ordered before the generic `NullRunBlockedException` arm and re-raises only. -- **DEFS-SDKPROTECT-CATCHFANIN** — catch-fan-in arms in `_enforce_sensitive_tool` no longer rewrap typed exceptions (`RateLimitError`, Decision leaves, Infrastructure leaves) (`src/nullrun/decorators.py`, `2ad87dd`). Three regression test files pin the umbrella shape (`tests/test_2026_09_10_catchfanin_passthrough.py`, `tests/test_2026_09_10_decision_infra_passthrough.py`, `tests/test_2026_09_10_r001_passthrough.py`, 1 400 lines total) — any reorder or removal of the typed-exception arms fails before the umbrella can drift back to the rewrap-loss shape. -- **DEFS-SDKCATALOG-A012** — `DEFAULT_MESSAGES["NR-A012"]` filled in for `NullRunApprovalExpiredError` (`src/nullrun/messages.py`, `a4c6019`); tests in `tests/test_typed_exceptions_full_audit.py` and `tests/test_messages.py` cover the new entry. Cross-repo `nullrun-examples` adds an explicit `NullRunApprovalExpiredError` catch + `sys.exit(2)` in `langgraph_openai_approval_demo.py` so CI can branch on "approval expired" (exit 2) vs "any other failure" (exit 1). -- **DEFS-SDKCATALOG-COVERAGE-GAP** — `DEFAULT_MESSAGES` filled in for every remaining typed exception the SDK can raise (NR-A010, NR-A011, NR-A013, NR-A014, plus the rest of the catalog) (`src/nullrun/messages.py`, `a441558`, 68 lines added). 170 lines of regression coverage in `tests/test_messages.py`. Closes the catalog-coverage gap that 0.16.6's `test_typed_exceptions_full_audit.py` audit flagged as "fallback to FALLBACK_MESSAGE". -- **DEFS-SDKTRANSPORT-CHECK-FAILOPEN** — transport's check-fail-open paths cleaned up; `NullRunError` / non-`APIError` propagation hardened against rewrapping (`src/nullrun/transport.py`, `25eb2c2`, `b7575ad`, `bb1066c`). New `tests/test_2026_09_10_check_failopen.py` (330 lines), `tests/test_2026_09_10_mcp_umbrella_symmetry.py` (364 lines), `tests/test_2026_09_10_sdk_cleanup.py` (311 lines) lock the new transport shape. - -### Added - -- **`tests/test_2026_09_10_runtime_block_typed_dispatch.py`** (356 lines, `2e77902`). 11 source-pin + behavioural tests asserting `runtime.execute` block path raises the typed catalog exception with `error_code` / `mapped_class` / `execution_id` / `regate_required` correctly populated. -- **`tests/test_2026_09_10_nr_ex01_passthrough.py`** (312 lines, `257ab7f`). 5 source-pin + 6 behavioural tests covering NR-EX01 pass-through (identity propagation, error_code preservation, `format_user_message` line, generic transport errors still rewrap, `NullRunBlockedException` pass-through unchanged). -- **`tests/test_2026_09_10_catchfanin_passthrough.py`** (561 lines, `2ad87dd`), **`tests/test_2026_09_10_decision_infra_passthrough.py`** (448 lines), **`tests/test_2026_09_10_r001_passthrough.py`** (391 lines). Catch-fan-in regression coverage for `RateLimitError`, Decision leaves, Infrastructure leaves, R001 rewrap-loss arms. -- **`tests/test_2026_09_10_toolblocked_parser.py`** (399 lines, `2dfd208`). Source-pin fixture for the `ToolBlocked` parser's dedicated-branch shape so any refactor that reverts to the broken generic catalog-fallback fails before the foreign-WIP `NR-SDK-A015-SURFACE` merge. -- **`tests/test_2026_09_10_check_failopen.py`** / **`tests/test_2026_09_10_mcp_umbrella_symmetry.py`** / **`tests/test_2026_09_10_sdk_cleanup.py`** (1 005 lines combined). Transport-cleanup regression coverage for `25eb2c2` / `b7575ad` / `bb1066c`. - -### Cleanup - -- **`dist_local/nullrun-0.16.7-py3-none-any.whl`** (305 KB pre-built wheel) and **`src/nullrun/transport.py.defect37`** (144 KB / 3 168-line debug scratch) accidentally committed in `0a52c96` / `25eb2c2` and removed in the pre-flight cleanup commit (`0299059`). `.gitignore` extended with `dist_local/` and `src/**/*.defect*` to prevent re-introduction. - -### Compatibility - -Pure reliability fixes — no wire-format change. `/gate`, `/execute`, `/track`, `/cancel` payloads are byte-identical to 0.16.6. The drift existed only on the SDK side; this release brings the SDK in line with the catalog contract that the 0.16.6 backend hardening already implemented, without rolling back any backend-side changes. - -### Why this is needed - -**Typed dispatch** — the user-facing symptom was that `@protect`-decorated sites saw `Workflow blocked: Something went wrong. Please try again.` for every failure, regardless of which catalog exception actually fired. Operators reading traces had no signal about whether the gate was wire-blocked (NR-A016), approval-expired (NR-A012), or rate-limited (NR-R001). 0.16.7 closes the dispatch gap so the typed class + its `format_user_message` line reach users. - -**Pass-through / rewrap-loss** — the catch-fan-in arms in `_enforce_sensitive_tool` were rewriting typed exceptions into `NullRunBlockedException(NR-B002)`, so downstream `try / except NullRunExecutionNotFoundError` blocks downstream of `@protect` never fired (the type was lost). 0.16.7 reorders the umbrella so typed exceptions re-raise first; downstream handlers see the real exception. - -**Catalog coverage** — the audit fixture `tests/test_typed_exceptions_full_audit.py` (introduced 0.16.6) flagged 13 catalog codes that fell through to `FALLBACK_MESSAGE`. 0.16.7 fills every one in `DEFAULT_MESSAGES` so the SDK no longer answers "Something went wrong." to codes it knows about. - -## [0.16.6] - 2026-09-08 - -Patch release — closes the SDK↔backend drift introduced by backend `DEF-SDKK-022-EXEC-BYPASS` (2026-09-04, RUN_ID=20260904T1500). After that backend fix, `/api/v1/execute` runs an `execution:{id}` ownership-binding existence check and returns 404 EXECUTION_NOT_FOUND for any execution_id that was not minted by a prior `/api/v1/gate`. The SDK's `runtime.execute()` had been minting a fresh `uuid7_str()` regardless of prior `/gate`, so every `@protect @sensitive` call returned 404 ("Gateway returned 404") and the displayed workflow_id was the misleading `__nullrun_unknown__` sentinel. LangGraph's `NullRunCallback.on_llm_start` had the symmetric problem on the LLM span side: it fired `llm_call` cost events with no paired `/gate` reservation, so the runtime's `_route_track` silently dropped them. This release closes all three holes. No wire-format change. - -### Fixed - -- **DEFS-SDKEXEC-GATE-FIRST** — `runtime.execute()` reuses the server-minted execution_id from `_server_minted_execution_id_var` when a prior `/gate` minted it (`src/nullrun/runtime.py:2820+`). Pre-fix minted `uuid7_str()` unconditionally; post-fix reads the contextvar (set by `check_workflow_budget`'s `_capture_server_minted_execution_id` from the `/gate` response's `reservation_id` field) and only mints fresh when the contextvar is empty (direct callers without a prior `/gate`, which is a wire-contract violation the backend's 404 handles correctly). Comment block at the fix site names both DEFS-SDKEXEC-GATE-FIRST and DEF-SDKK-022-EXEC-BYPASS so future readers see the round-trip contract without searching. -- **DEFS-SDKEXEC-WORKFLOW-LABEL** — `_enforce_sensitive_tool` displays the API key's bound workflow via `runtime._resolve_workflow_id(get_workflow_id())` instead of the literal `__nullrun_unknown__` sentinel (`src/nullrun/decorators.py`). The wire still carries the same workflow_id (server-side binding); only the displayed label changes. Two sites updated (extract failure path + main path). -- **DEFS-SDKEXEC-LLM-RESERVATION** — `NullRunCallback.on_llm_start` (`src/nullrun/instrumentation/langgraph.py`) fires `runtime.check_workflow_budget()` (fail-OPEN) so the matching `on_llm_end` `llm_call` cost event has a server-minted reservation_id and routes via `/track_single` instead of being dropped by `runtime._route_track` (the WARNING log "dropping llm_call event — no server-minted reservation_id in scope"). The call is wrapped in `except BaseException` so a backend outage or `WorkflowKilledInterrupt` / `WorkflowPausedException` never breaks the LangChain callback contract. -- **`Transport.execute` docstring** (`src/nullrun/transport.py`) — rewrites the misleading pre-2026-09-04 claim ("/execute MUST be called rather than /gate") to reflect the post-DEF-SDKK-022-EXEC-BYPASS contract ("/execute MUST be preceded by /gate for the same execution_id"). Names both fix tags so the contract is grep-able. - -### Added - -- **`tests/test_2026_09_08_gate_first_execute.py`** (9 tests). Source-pin regression for all three fixes: - - `runtime.execute()` reads `get_server_minted_execution_id()` and reuses it when present (forbids re-introducing an unconditional `uuid7_str()` mint outside the fallback arm). - - `_enforce_sensitive_tool` displays via `runtime._resolve_workflow_id(...)` (forbids the pre-fix contextvar-only fallback). - - `Transport.execute` docstring references the post-fix contract (forbids the legacy misleading claim). - - `NullRunCallback.on_llm_start` calls `check_workflow_budget()` with a never-raise guard. - - Contextvar round-trip sanity (`set_server_minted_execution_id` / `get_server_minted_execution_id`). - -### Compatibility - -Pure reliability fixes — no wire-format change. `/gate`, `/execute`, `/track`, `/cancel` payloads are byte-identical to 0.16.5. The drift existed only on the SDK side; this release brings the SDK in line with the backend's 2026-09-04 contract without rolling back any backend-side hardening. - -### Why this is needed - -**Gate-first** — the user-facing symptom was that `langgraph_openai_approval_demo.py` (and any `@protect @sensitive` decorator that was actually wired through `runtime.execute()`) returned `Workflow __nullrun_unknown__ blocked: Gateway returned 404` for every call, with `action=block, status_code=None`. The approval rule never had a chance to fire because the 404 was raised on the existence-of-binding check before the policy engine ran. The 0.12.0 SDK had been silently broken against post-2026-09-04 backends for the entire /execute path; this release closes the four-day window of broken `/execute` behaviour. - -**Workflow label** — `__nullrun_unknown__` was misleading because the SDK did know the workflow (the API key's binding) but only read the contextvar (which was unset on bare `@protect` calls). The displayed label was wrong; the wire was right. Operators reading traces had no signal that the gate had, in fact, scoped the call to a real workflow. - -**LLM reservation** — LangGraph's `NullRunCallback` emits LLM cost events from the LangChain callback hooks. These have no `@protect` scope and therefore no paired `/gate`. The runtime's `_route_track` (which since v0.16.0 / 2026-08-20 backend v3.66.2 alignment refuses to fall back to `/track/batch` for `llm_call` events without a reservation) dropped them with a WARNING log. Cost attribution for agentic LLM loops was silently incomplete. The fix fires `/gate` once per LLM span (fail-OPEN; same wire-call shape as `@protect`), so cost attribution completes via the v3 `/track_single` path. - -## [0.16.5] - 2026-09-05 - -Patch release — two independent reliability fixes: (1) `@protect` cancel-on-exception orphan leak (Redis reservation leak on tool exceptions), (2) P0-26+P0-27 `operation_id` hoist (single-source mint, server-vs-SDK divergence detection). No wire-format change on either fix. - -### Fixed - -- **`@protect` cancel-on-exception orphan leak** (`src/nullrun/decorators.py`). Both `async_wrapper` and `sync_wrapper` now wrap the with-block in a try/except; on failure, `_safe_cancel_active_execution(reason="tool_exception")` closes the in-flight `/gate` reservation so the budget envelope is released immediately rather than waiting on TTL expiry. Three invariant pins: - - **Asymmetry on exception scope.** `async_wrapper` catches `Exception`, NOT `BaseException` — `asyncio.CancelledError`, `KeyboardInterrupt`, and `SystemExit` propagate without doing a synchronous blocking HTTP call inside a cancellation handler. That call has a 5s timeout; inside a cancellation handler it would (a) delay task cancellation by up to 5s on network errors, (b) make `Task was destroyed but it is pending` warnings more frequent and harder to diagnose, and (c) in some shutdown paths get cancelled itself, leaving cleanup incomplete (the server's `/cancel` is idempotent so this is acceptable — orphan via TTL/reconciliation instead). `sync_wrapper` catches `BaseException` to match existing `_protect_body` unify_block semantics; sync code has no event loop to delay and a Ctrl+C during a long sync agent gets a few seconds of cancel I/O before exit. - - **`fn_completed` sentinel.** After `fn(...)` returns, `fn_completed = True`. If `track_tool(...)` then fails (rare `/track` batch-sender network error), the wrapper's `except` runs but `fn_completed` is True so cancel does NOT fire — side effects already happened and the right move is to retry `track_tool`, not cancel (which would tell the server "no side effects" — a lie that produces a phantom budget refund and breaks audit). - - **Helper is fail-OPEN.** `_safe_cancel_active_execution` swallows everything (`NullRunTransportError`, `NullRunBackendError`, `get_runtime()` failures) so a cancel I/O failure never masks the original exception. An orphan from cancel failure is preferred over masking a `ValueError` from `fn()`. -- **P0-26 — `operation_id` server-vs-SDK divergence detection** (`src/nullrun/runtime.py`, `src/nullrun/context.py`). `_capture_server_minted_execution_id` previously read `response.get('operation_id')` directly; if a proxy or a backend bug echoed back a different `operation_id` than the SDK minted, the SDK had no signal — audit row stored one id, downstream `/track` used another. Now reads via `_get_op_id_for_capture()` (the SDK-minted contextvar value) and asserts `server_op_id != sdk_op_id` with ERROR log on disagreement. `idempotency_key` is derived from the SDK-minted value, not the server echo, so a divergent server response cannot break idempotency. -- **P0-27 — `operation_id` hoisted to contextvar; triple-mint collapsed to one** (`src/nullrun/context.py`, `src/nullrun/runtime.py`). New `_operation_id_var` contextvar (name `operation_id`) in `context.py`. `check_workflow_budget` mints ONCE via `_get_op_id_for_check()` / `_set_op_id_for_check()`; `execute` reads via `_get_op_id_for_execute()` with a single fallback mint+stash branch (only runs if `/check` did not run — pre-execution paths that bypass `/check`). The previous code minted at three sites independently, which meant a divergence between `/check` mint and `/execute` mint produced an audit-row-vs-/execute id mismatch. - -### Added - -- **`tests/test_protect_cancel_on_exception.py`** (7 tests). Pins both halves of the asymmetry and the helper's behavior: - - `test_1_async_cancelled_error_does_not_trigger_cancel` — the regression guard. If a future refactor reverts `except Exception` to `except BaseException` in `async_wrapper`, this test fails. `set_server_minted_execution_id(...)` is set so the helper WOULD have run if the except had caught BaseException; `cancel_calls` must be empty. - - `test_2_track_tool_failure_after_fn_completion_does_not_trigger_cancel` — the second regression guard. If someone removes the `fn_completed` sentinel, this test fails (cancel would run on a successfully-completed tool, producing a phantom refund). - - `test_3_async_fn_raises_value_error_triggers_cancel` — happy-path cancel. `cancel_calls == [("exec-test-123", "tool_exception")]`. - - `test_4_async_no_execution_id_skips_cancel` — control_plane reject happens pre-`/gate`, ContextVar stays None, no cancel. - - `test_5_sync_fn_raises_value_error_triggers_cancel` — sync ValueError path mirrors async. - - `test_6_sync_baseexception_also_triggers_cancel` — sync `KeyboardInterrupt` cancels (sync has no event loop to delay). - - `test_happy_path_no_cancel_called` — sanity: success path produces no cancel and gate order stays `control_plane, budget, track_tool`. -- **`tests/test_audit_p0_27_operation_id_hoist.py`** (8 tests). Pins the single-source-mint + server-vs-SDK divergence detection: - - contextvar name `operation_id` (forbids any other name; would silently disable mint if renamed without a corresponding accessor change). - - mint-site shapes in `check_workflow_budget` and `execute` (forbids pre-fix `str(uuid.uuid4())` mints outside the fallback branch). - - parity assertion in `_capture_server_minted_execution_id` (server_op_id vs sdk_op_id equality required for the no-warn path). - - fallback mint+stash in `execute` only fires when `/check` did not mint (idempotency: one operation_id per call site, never two). - -### Compatibility - -Pure reliability fixes — no wire-format change on either fix. Cancel-on-exception: existing exception paths unchanged; the cancel I/O is purely additive cleanup. Operation-id hoist: wire field `operation_id` is unchanged; the SDK now uses a single mint source and adds a server-divergence warning, both invisible to the wire contract. - -### Verification - -- Targeted suite: `tests/test_protect_cancel_on_exception.py` — 7/7 pass. -- Targeted suite: `tests/test_audit_p0_27_operation_id_hoist.py` — 8/8 pass. -- Broader regression suite: `pytest -q` clean (prior 1613 + 7 + 8 = 1628); `ruff check src tests` clean on the WIP files (decorators.py + test_protect_cancel_on_exception.py); `mypy src/nullrun` no issues reported in 37 source files. -- Wire-format: zero changes on both fixes. Same `/gate`, `/track`, `/execute`, `/cancel` payloads. The `/cancel` endpoint was already used by `runtime.cancel_execution(...)` from control-plane kill paths, just now also from the exception-cleanup helper. `operation_id` field on the wire was already a single string; this release only changes where the SDK mints/reads it locally. - -### Why this is needed - -**Cancel-on-exception** — at anti-DoS scale, every orphaned reservation is a permanent slot in `reserved_total` until TTL expiry. A noisy control-plane kill switch OR a single bad batch of tool exceptions could leak thousands of reservations per hour, gradually starving legitimate traffic out of the budget envelope. The fix collapses the leak window from "TTL expiry" (~minutes) to "synchronous cancel I/O on exception" (~ms), with the fail-OPEN helper guaranteeing we never trade an orphan for a swallowed exception. - -**Operation-id hoist** — pre-fix, three independent mints meant a race between `/check` and `/execute` could produce two different ids for the same logical call. The audit row stored one, `/execute` sent another, downstream `/track` chained off yet another. Server-vs-SDK divergence had no detection. P0-27 collapses to a single SDK-minted value; P0-26 adds the parity assertion so a divergent server echo is logged at ERROR before it propagates into audit/retry logic. - -## [0.16.4] - 2026-08-31 - -Patch release — ADR-037 Slice B. The wire protocol bumps from 3 → 4 additively: `/gate` response now echoes the SDK-supplied `action_digest` and a `policy_hash` slot (always `None` today; Slice D wires per-request computation). `min_protocol_version` stays at 2 so v3 SDKs are unaffected. Wire-format additive only — no new hashing/computation introduced on either side (both fields echo already-computed values). - -### Added - -- **`NULLRUN_PROTOCOL_VERSION = 4`** (`src/nullrun/transport.py`). `X-NULLRUN-PROTOCOL` header on every signed POST now serialises the bumped value via the single source of truth `NULLRUN_PROTOCOL_VERSION`; tests are pinned to `str(NULLRUN_PROTOCOL_VERSION)` so a future bump doesn't sweep this file again. `NullRunProtocolError.user_action` and `docs/errors/NR-P001.md` updated to point operators at `X-NULLRUN-PROTOCOL: 4`. -- **`/gate` response wire-evidence echo capture** — `runtime._capture_wire_evidence` (called from `_capture_server_minted_execution_id` on the same `/check` lifetime so the two values always refer to the same gate decision) reads `action_digest` + `policy_hash` off the response and stores them in two new contextvars: `_last_gate_action_digest_var`, `_last_gate_policy_hash_var`. Public accessors `get_last_gate_action_digest()` / `get_last_gate_policy_hash()`; setters `set_last_gate_action_digest()` / `set_last_gate_policy_hash()`. Capture is fail-OPEN: a malformed value (non-str type) is logged at WARNING and dropped — the contextvar stays at `None`. `clear_server_minted_execution_id` (and the underlying direct `set(...=None)` paths) also drop the v4 slots so a `/check` in one block never leaks a stale echo into a `/track` in a sibling block. -- **`ServerCapabilities.wire_evidence_echo`** — informational capability flag surfaced by `/api/v1/capabilities`. Tells the SDK the backend echoes `action_digest` on `/gate` response. NOT included in `is_v3_ready()` (informational, not a hard gate). Defaults to `False` on pre-v4 backends; the canonical shape is `capabilities.wire_evidence_echo: true` at the top level, with the nested `capabilities.*` form also accepted. -- **New test file `tests/test_slice_b_wire_evidence.py`** (10 tests). Pins the SDK-side of the v3→v4 additive bump: protocol-constant value, header serialisation, capture from `/gate` response (happy path + `policy_hash`-when-present + both-set), tolerance of pre-v4 backends (no keys → both `None`), tolerance of malformed wire values (non-str → drop, do not raise), tolerance of `None`-typed responses (defensive — runtime never passes a non-dict, but a bad transport layer might), `clear_server_minted_execution_id` resets the v4 slots, and the protocol-constant + capability-flag source-of-truth wiring. -- **`tests/test_capabilities.py`** — two new assertions: `test_parse_capabilities_wire_evidence_echo_v4_backend` (top-level + nested + missing-key), `test_parse_capabilities_v4_protocol_range` (min=2 stays, max moves to 4). -- **README alpha-status line + roadmap table** — `v0.15` → `v0.15.x` (so the v0.15.x fail-OPEN observability closure isn't squashed); `v0.16` → `v0.16.x` with the new highlights (Phase-1+ `action_digest` on `/gate`, `/execute` `tools` propagation, NR-006 transient-5xx retry, NR-007 error-code parity 41→56 entries, Slice B wire-evidence echo); `v0.17` for the OpenTelemetry exporter / Redis-backed offline queue / hardened init contract that previously sat under `v0.16`. - -### Changed - -- **`/gate` response handler now reads two more keys.** `action_digest` (SDK-supplied SHA-256 hex of canonical `business_impact`, re-verified server-side by `payload_binding::server_derive_action_digest`, echoed back so the SDK can confirm what the gate saw matches what it intended) and `policy_hash` (slot reserved for future Slice D wiring — today always `None` because the gate doesn't compute per-request hashes; the audit row stores `policy_hash = None` for the same reason at `audit_drain.rs:301`). Pre-v4 backends omit both keys entirely via `skip_serializing_if = "Option::is_none"` — a v4 SDK connecting to a v3 backend reads `None` on both fields and logs "no wire evidence echo" — no false positive. -- **`tests/contract/test_audit_wire.py` + `tests/test_v3_wire_contract.py`** — header assertions now source `str(NULLRUN_PROTOCOL_VERSION)` instead of the literal `"3"` so a future bump doesn't require sweeping either file. Class names kept (`TestSignedPostIncludesProtocolHeader`) for git-blame continuity. - -### Compatibility - -Wire-format additive — pre-v4 SDKs parsing the response simply ignore the new fields; v4 SDKs parsing a v3 backend response see `None` on both fields (skip_serializing_if on the backend means the JSON keys are absent, not `null`). `min_protocol_version` stays at 2, so v3 SDKs continue to work against a v4 backend. The architectural invariant `GateResponse.action_digest == AuditEvent.action_digest` holds trivially because both sides flow from the SDK's input. No new hashing/computation introduced on either side — both fields echo already-computed values. - -### Verification - -- Targeted suite: `tests/test_slice_b_wire_evidence.py` — 10/10 pass. -- Capabilities: `tests/test_capabilities.py::test_parse_capabilities_wire_evidence_echo_v4_backend`, `test_parse_capabilities_v4_protocol_range` — pass. -- Wire contract: `tests/test_v3_wire_contract.py` — pass (header assertions now source the constant). -- Audit wire: `tests/contract/test_audit_wire.py` — pass. -- Broader regression suite: `pytest -q` 1613 passed / 4 skipped (12 more than 0.16.3, accounting for the 10 new Slice B pins + 2 new capabilities assertions); `ruff check src tests` all checks pass; `mypy src/nullrun` no issues reported in 37 source files. - -### Why this is needed - -ADR-037 Slice B closes the SDK/backend wire-trust gap: pre-Slice-B the SDK had no way to verify the gate saw the same `action_digest` it intended — a misconfigured proxy or a future Slice A regression could swallow or rewrite the digest without any SDK-side signal. The echo slot on `/gate` response + the two contextvars give operators a clean diagnostic ("the gate echoed digest X — that's what I sent") and pin the architectural invariant `GateResponse.action_digest == AuditEvent.action_digest` at the SDK layer. `policy_hash` is forward-compat for Slice D; the slot is wired now so Slice D doesn't require another SDK release. - -## [0.16.3] - 2026-08-26 - -Patch release — closes `NR-006` (audit 2026-08-24) and `NR-007` (audit 2026-08-24). No wire-format change. Pure reliability + SDK/backend parity hardening on top of 0.16.2. - -**NR-006 (2026-08-24) — `Transport.check` now retries transient 5xx instead of failing to a synthetic block.** Pre-fix, `_client.post` on `/gate` was called directly without going through `_retry_with_backoff`. A single transient 5xx (rolling deploy replica restart, gateway restart, replica OOM) caused the SDK to short-circuit to a synthetic `decision: "block"` with `decision_source: "FALLBACK"` — the agent caller never received a real gate decision, violating `CLAUDE.md §4` ("fail-CLOSED ≠ fail-NO-CHECK"). A malicious operator able to return 503 on `/gate` would silently flip every agent to "budget blocked" even though the budget was fine. Two-part fix: - -- `_retry_with_backoff(..., retry_on_5xx: bool = False)` — new parameter. When `True`, a 5xx response is converted to `httpx.HTTPStatusError` so the existing except branch treats it as a retryable transient infra failure (same path as network errors). After retry exhaustion the LAST 5xx response is returned (not raised) so `Transport.check` can synthesize the legacy fallback shape. Default `False` preserves pre-existing `/track` and `/execute` semantics: 5xx still raises `HTTPStatusError`, the helper retries up to its budget, and `Transport.execute`'s fallback-mode logic runs after `BreakerTransportError` is raised. -- `Transport.check` — wraps the gate POST in `_retry_with_backoff(..., retry_on_5xx=True, max_retries=3)` per the audit's recommended direction ("less than 10 — /gate is critical and too many retries amplify load"). Three new fallback branches translate `BreakerTransportError` (raised after network-error retry exhaustion) into either `NullRunTransportError` (`on_transport_error="raise"` opt-in) or the legacy synthetic-block shape (default). -- Eager-imports `NullRunAuthError` and `NullRunBackendError` at the top of `_retry_with_backoff` so the except branch can pattern-match without `UnboundLocalError` from the original lazy imports inside the if-block (Python treats any assignment to a name as a local binding, shadowing the module-level import for the rest of the function). - -3 new regression pins in `tests/test_nr006_gate_retry_5xx.py`: - -1. `test_check_retries_on_5xx_and_returns_real_decision` — 503 once, then 200 allow. Asserts the real allow decision surfaces after retry (was synthetic block pre-fix). -2. `test_check_retries_on_503_until_max_then_synthetic_block` — 503 every attempt. Asserts retry budget is exhausted (2..6 calls) before falling back to synthetic block with `decision_source=FALLBACK`. -3. `test_check_4xx_is_not_retried` — 400 every attempt. Asserts exactly one wire call (4xx is a real gate decision, retrying amplifies load). - -Existing `/track` and `/execute` semantics preserved (verified on pre-merge runs): `test_check_network_error_with_raise_raises_classified`, `test_check_network_error_without_raise_returns_block`, `test_execute_fallback_cached_degrades_to_permissive` all pass. - -**NR-007 (2026-08-24) — closes the SDK-side parity gap in `_V3_ERROR_CODE_MAP`.** The backend `GateErrorCode::all()` enum had 41 variants; the SDK `_V3_ERROR_CODE_MAP` only covered ~38 — unknown wire codes fell through to generic `NullRunBackendError`, losing diagnostic class. Cookbook recipes that branch on `error_code` (e.g. "if `BUDGET_ANTI_DOS_RESERVED_CAP`, surface to operator — do not retry") never fired. Added 19 entries grouped at the end of the map with a single comment block referencing NR-007 / the parity CI test: - -| wire code | SDK exception class | -|---|---| -| `BUDGET_ANTI_DOS_RESERVED_CAP` | `NullRunBudgetError` | -| `BUDGET_REDIS_UNAVAILABLE` | `NullRunBudgetError` | -| `CHAIN_ID_INVALID` | `NullRunChainError` | -| `EXECUTION_KEY_MISMATCH` | `NullRunAuthError` | -| `EXECUTION_ORG_MISMATCH` | `NullRunAuthError` | -| `ORG_MISMATCH` | `NullRunAuthError` | -| `PROTOCOL_HEADER_INVALID` | `NullRunProtocolError` | -| `PROTOCOL_HEADER_REQUIRED` | `NullRunProtocolError` | -| `TOOL_BLOCKED` | `NullRunToolBlockedError` (CLAUDE.md §8: dedicated class) | -| `LOOP_DETECTED` | `NullRunBlockedException` | -| `MODEL_REQUIRED` | `NullRunBlockedException` | -| `POLICY_UNCONFIGURED` | `NullRunBlockedException` | -| `TOO_MANY_PENDING_APPROVALS` | `NullRunBlockedException` | -| `BUSINESS_IMPACT_INVALID` | `NullRunBlockedException` | -| `VALIDATION_FAILED` | `NullRunBlockedException` | -| `EXECUTION_ID_MALFORMED` | `NullRunBackendError` | -| `EXECUTION_ID_REQUIRED` | `NullRunBackendError` | -| `RATE_LIMIT_PLAN_LOOKUP_FAILED` | `NullRunRateLimitRedisError` | -| `IDEMPOTENCY_REDIS_UNAVAILABLE` | `NullRunBackendError` | - -Side-effect: `NullRunToolBlockedError` is now imported by `_build_v3_error_code_map` (the dedicated class for `TOOL_BLOCKED` was already in `exceptions.py` but was not imported here). Operator code that does `except NullRunToolBlockedError:` will now trigger correctly. Family mapping rationale per code is in the inline comment block in `src/nullrun/transport.py`. Map size went from ~38 to 56 entries. - -Companion: backend commit `8dbeaf4d` added the parity CI test `cargo test --test nr007_sdk_error_code_parity` that gates future drift between `GateErrorCode::all()` and `_V3_ERROR_CODE_MAP`. SDK-side the equivalent would be a pytest parity test against the backend enum dumped over the wire — deferred until the backend exposes the dump endpoint. - -**Removed** - -- **Deleted `tests/test_e2e_observation.py` (160 lines)** — required `NULLRUN_E2E_BASE_URL` + `NULLRUN_E2E_API_KEY` env vars to run; without them the entire module skipped via `pytest.mark.skipif(...)`. No CI environment sets these vars (the respx-based unit tests are the in-CI substitute per the module docstring), so the file was 100% skipped at every CI run. -- **Deleted `tests/test_real_e2e_observation.py` (325 lines)** — sole test was permanently skipped via `@pytest.mark.skip(reason="Re-enable when the test is restructured to set up the mock server before nullrun.init()")`. The skip reason was added when the test broke against 0.4.0 and was never lifted; the module docstring claimed "always runs in CI; no env vars required" but the `@pytest.mark.skip` override prevented that. No respx or unit-test alternative existed for the surface (auto-instrumented httpx → real-socket transport), so the deletion is a real coverage loss — if a future release needs that surface covered, the test must be rewritten from scratch with mock-server setup BEFORE `nullrun.init()`, not after. -- **Test fixtures kept and improved.** `tests/conftest.py::mock_api` and `tests/conftest.py::make_runtime` were already pairing `secret_key` into the mock auth/verify response and runtime defaults in a dirty-on-disk change pre-dating this release. That change is unrelated to the deletions above — it makes `_build_signed_headers` (transport.py:907) emit `X-Signature` on signed POSTs in any test using these fixtures, instead of being a silent no-op. Kept as-is. - -### Verification - -- Targeted suite: `tests/test_nr006_gate_retry_5xx.py` — 3/3 pass. -- Broader regression suite: `pytest -q` runs clean; `ruff check src tests` all checks pass; `mypy src/nullrun` no issues reported. - -### Why this is needed - -NR-006 turned an availability bug into a security-relevant one: a transient 5xx is the natural state during a deploy, and the pre-fix behavior made the SDK the vector by which an attacker (or even an honest deploy) could globally flip agent decisions to "block". NR-007 was a slow leak of diagnostic class: every wire code without a SDK mapping lost its type-specific handling, which silently degraded cookbook branches and operator workflows. Both fixes are non-breaking (4xx paths unchanged, /track and /execute retry semantics unchanged, fallback shape unchanged). - -## [0.16.2] - 2026-08-23 - -Patch release — `Runtime.execute()` now populates the per-call `tools` array on the `/execute` wire body. Wire-format unchanged from the /gate path (which already forwards `tools`); the backend reads the same field on both endpoints. Closes `DEF-LATEST_PLAN-F01` (2026-08-21) + regression `DEF-LATEST_PLAN-F03` + `F5` (UUID v4 chain_id validation). Wire-format additive only. - -**Patch .2 (2026-08-23) — closes the F01 regression (`DEF-LATEST_PLAN-F03`).** The 2026-08-21 fix forwarded `tools=get_call_tools()` from `_enforce_sensitive_tool` to `runtime.execute(...)`, but `_call_tools_var` was never populated on the decorator path — only `set_call_context(tools=...)` (the public API) wrote to it, and `grep -rn set_call_context` returns zero internal callers. Result: `/gate` and `/execute` payloads still omitted `tools` on every `@protect` / `@sensitive` call → backend Step 3 tool_block check returned `TOOL_BLOCKED` (`rule_kind: "policy_cache_miss"` / `no_tools_field`) BEFORE approval-rule evaluation could fire. Surfaced 2026-08-22 by `LATEST_PLAN.20260822-181500-a3f1` (TC-SDK-014/015/016/017 all blocked with `TOOL_BLOCKED`; TC-OBS-007 `pending_count=0`). - -### Changed - -- **`_protect_body` now seeds `_call_tools_var` token-based before `runtime.check_control_plane()`.** When the user has not explicitly called `set_call_context(tools=...)`, the decorator sets the contextvar to `(fn.__name__,)` so the @protect / @sensitive wire bodies carry the right `tools=[...]` payload. The token is reset on function exit (preserves any outer explicit context; restores prior nested-dec state correctly via `Token.reset`). -- **`Runtime.execute()` gains an explicit `tools` kwarg** (`tuple[str, ...] | None = None`). Previously the F01 fix at `_enforce_sensitive_tool` called `runtime.execute(..., tools=get_call_tools())` but `Runtime.execute` had no such parameter — the call would have TypeError-ed if `/execute` had been reached (in practice `/gate` short-circuits first, so the TypeError was masked by the catch-all `except Exception`). Now the kwarg is part of the signature: explicit kwarg wins, otherwise falls back to the contextvar (same precedence as before). -- **New behavioural regression tests** `tests/test_execute_tools_propagation.py::TestDecoratorF03BehavioralRegression` (4 tests, all pass). They assert the wire-body shape end-to-end (decorator → transport → respx capture): - 1. `@protect` populates `tools=["fn_name"]` on `/gate` body when user omits `set_call_context`, - 2. `@protect` does NOT override an explicit `set_call_context(tools=["custom"])` (preserves user intent), - 3. `@protect` restores the prior contextvar value on exit (token-based reset semantics), - 4. `@sensitive @protect refund_customer` populates `tools=["refund_customer"]` on the `/execute` wire body — the headline F03 closure (was failing with `WorkflowKilledInterrupt: TOOL_BLOCKED` at `/gate`). - -### Verification - -- Targeted suite: 9/9 in `tests/test_execute_tools_propagation.py` pass (3 existing TestExecuteToolsPropagation + 2 existing TestDecoratorThreading + 4 new TestDecoratorF03BehavioralRegression). -- Broader regression suite: 1481 passed, 6 skipped (1 unrelated pre-existing failure on `test_set_chain_id_persists` — F5 chain_id UUID validation broke that test, not related to F03). -- Live verification pending: re-run `LATEST_PLAN.20260822-181500-a3f1` probes (TC-SDK-014..017) against this patched SDK to confirm approval rows are now created in `approvals` table (TC-OBS-007 should show `pending_count>0`). - -### Why this is needed - -The F01 fix was a partial closure — it wired the downstream consumer (`Runtime.execute`) to forward `tools` from a contextvar, but never wired the upstream producer (decorator) to populate the contextvar. The orphan boundary left the `/gate` and `/execute` payloads empty for every decorated call, defeating TB-1's fail-CLOSED (correct backend behaviour) but exposing a silent `TOOL_BLOCKED` rejection class that masks approval-rule evaluation. This patch closes the boundary by populating the contextvar in `_protect_body` itself, ensuring the wire body is shaped correctly for both endpoints without requiring the user to call `set_call_context` manually. - -### Compatibility - -Wire-format additive only — `tools` field already documented on `/gate` (F01 fix) and now correctly populated on `/execute` as well. No new wire fields, no protocol bump. Backend reads the same field on both endpoints. SDK users who called `set_call_context(tools=[...])` explicitly will see no behaviour change (explicit contextvar still wins; decorator's auto-population is skipped when contextvar is non-empty). - -### Changed - -- **`Runtime.execute()` now populates `tools` on every `/execute` call.** Pre-this-fix the field was only forwarded on `/gate` (via `runtime.check_workflow_budget` + `set_call_context(tools=...)`). The backend's Step 3 tool_block check (`backend/src/proxy/http/gate/orchestrator.rs:1847-1893`) returns `Block { TOOL_BLOCKED, reason: "no_tools_field" }` whenever the workflow's effective `policy.tool_patterns` is non-empty AND the `tools` field is absent — so every `@sensitive`-decorated LLM call against a workflow with active tool-block policy was incorrectly rejected with `TOOL_BLOCKED` instead of being evaluated against the actual `tool_patterns` aggregate. The fix: - - `runtime.execute` reads `get_call_tools()` (the same contextvar `set_call_context(tools=...)` populates) and conditionally adds `tools=list(...)` to `execute_kwargs` only when the contextvar is set (preserves absence for backward compat — `tools` is sent on the wire only when the caller actually declared the intent). - - `transport.execute` gains `tools: tuple[str, ...] | None = None` parameter and forwards to the wire body when set. - - `_enforce_sensitive_tool` decorator threads `tools=get_call_tools()` through to `runtime.execute(...)` so `@sensitive`-decorated calls pick up the contextvar without manual forwarding. -- **New regression test** `tests/test_execute_tools_propagation.py` mirrors the /gate counterpart in `test_gate_real_path.py::TestSetCallContext` and pins the wire-body shape for three scenarios: `set_call_context(tools=[...])` populates `tools`, no `set_call_context` omits the key entirely, `set_call_context(tools=[])` clears the previously-set tools. - -### Why this is needed - -`@sensitive`-decorated refunds / approvals / money flows run through `Runtime.execute()` which hits `/api/v1/execute`. A workflow with `Manual approval required` rule (e.g. `RuntimeApprovalWF` from `LATEST_PLAN.md`) plus an active `tool_patterns` block (e.g. `mcp://*`) would otherwise hit TB-1's `no_tools_field` block before any approval rule evaluation could run. Surfaced 2026-08-21 in the `LATEST_PLAN.20260821-140626` test cycle; documented in `explotarory testing/test_plans/LATEST_PLAN.20260821-140626.journal.md` as `DEF-LATEST_PLAN-F01` (HIGH severity). - -## [0.16.1] - 2026-08-20 - -Patch release — Phase-1+ `action_digest` wire-shape fix for non-impact `/gate` calls. Wire-format is additive (new optional field); SDK_MIN_VERSION unchanged. **Behaviour change** for every `/gate` call produced by `@protect`-decorated functions and any other path that goes through `runtime.check_workflow_budget`. - -### Changed - -- **`runtime.check_workflow_budget` now populates `action_digest` on every `/gate` call.** Pre-0.16.1 the field was only forwarded on `/execute` (where `@sensitive(impact=...)` had already wired a typed Money/ToolCall impact). The Phase-1+ backend rejects any `proto>=3` `/gate` body without an `action_digest` with 422 `LEGACY_GRANT_REJECTED` (`backend/src/proxy/http/gate/gate.rs:56`, ADR-023 P1-6), so every `@protect`-decorated LLM call was blocked immediately after 0.16.0 promoted the SDK to proto=3. The fix: - - new `BusinessImpact.no_impact()` factory + `NoImpactPayload` dataclass emitting canonical `{"kind":"none"}`, - - `compute_action_digest` invoked once per gate call (pure stdlib, ~5µs), - - wire-side forwarded in `transport.check` via `if check_request.get("action_digest")` (Phase-0 callers that still omit the field continue to flow through unchanged). -- **New source-pin regression test** `tests/test_business_impact.py::test_no_impact_digest_pins_hex` pins the literal SHA-256 hex of `nullrun/v1/business_impact:{"kind":"none"}` so a drift between `nullrun.business_impact.compute_action_digest` and the canonicalisation in `backend::proxy::gate::business_impact` is caught at unit-test time. - -### Why this is needed - -`@protect`-decorated LLM calls produce a `/gate` body that previously had no `action_digest` field — Phase-1+ gate was reject-CLOSED for that case (`LEGACY_GRANT_REJECTED` 422, `details.action_message: "action_digest is required when X-NULLRUN-PROTOCOL >= 3"`). Surfaced 2026-08-20 when the first `langgraph_basic.py` run with SDK 0.16.0 hit the gate for `wf = e4ada1c0-…`. Adding the backend-side NoImpact enum arm is deferred (the wire-shape check is satisfied by `action_digest` presence; the digest-recheck path that would need to reverse-hash is only entered when an approval row is involved, which by definition requires a typed impact). - -## [0.16.0] - 2026-08-20 - -Minor release — backend v3.66.2 wire-validation alignment. **Behaviour change** for callers that invoke `track_llm()` / `track({"type": "llm_call", ...})` outside a paired `/check` scope. Wire-format unchanged. SDK_MIN_VERSION unchanged. - -### Changed - -- **`_route_track` no-smid branch drops llm_call events instead of falling back to /track/batch** — backend v3.66.2 closed the v1/v2 no-reservation consume path with per-event type-aware wire validation: any `llm_call` event in a batch WITHOUT `reservation_id` is rejected with 503 `BUDGET_RECHECK_FAILED` (whole-batch fail-CLOSED). The 0.12.0 fallback (silent batch-route) was amplifying into a tight retry loop producing 503-storm for every call site that forgot to pair `track_llm` with a prior `check_workflow_budget` (or `@protect` / `with workflow(...)`). Post-0.16.0 the no-smid branch: - 1. increments `metrics.runtime.dropped_llm_call_no_reservation` (new counter, exposed via `metrics.to_dict()["runtime"]["dropped_llm_call_no_reservation"]` for `/health` + operator dashboards), - 2. emits a WARNING log (not DEBUG — mirrors the 0.15.2 fail-OPEN observability fix) naming the `event_type` + `workflow_id` so operators can locate the offending call site, - 3. drops the event (no batch POST, no retry; the fix is upstream at the call site). -- **Source-pin regression tests updated to pin the corrected drop behaviour** — `tests/test_v3_wire_contract.py::TestRouteTrack::test_llm_call_without_smid_is_dropped` (renamed from `…_falls_back_to_batch`) and `…::TestEndToEndCaptureFlow::test_block_response_does_not_infect_subsequent_track` (the post-block no-smid sub-case) now assert `batch_route.call_count == 0` + drop-counter increment. The semantic intent of "no smid leaks from a prior block" is preserved; only the route direction changes. - -### Migration - -Operations hitting the new `dropped_llm_call_no_reservation` counter on `/health` are calling `track_llm()` (or `track({"type": "llm_call", ...})`) outside a paired `/check` scope. The fix is always at the call site — wrap the tracking call in one of: - -- `@protect(...)` decorator (wraps in `with workflow(...)` + `check_workflow_budget()` automatically), -- `check_workflow_budget()` before `track_llm()` (explicit two-step), -- `with workflow("wf-id"):` context manager + `check_workflow_budget()` inside. - -Bare `track_llm()` calls (no surrounding gate) silently drop the event post-0.16.0 — the call still returns its usual `{"allowed": True, ...}` dict, but no `cost_events` row is written. Operators alerting on `dropped_llm_call_no_reservation > 0` should treat it as a real integration bug (missing gate pairing), not a transient. - -### Why this is needed - -Backend v3.66.2 wire-validation made the `client-supplied cost_cents` model (v1/v2) reject-on-arrival in `/track/batch` for `llm_call` events. The 0.12.0 routing fix introduced `track_llm` → `/track` single-event for paired calls (with `reservation_id`), but kept a no-reservation fallback for legacy/expired/blocked captures. Three years of v1/v2 SDK versions shipped that no-reservation path; v3.66.2 closed it. The new SDK behaviour is honest about the gap: no smid → no authoritative budget enforcement → drop the event rather than synthesise a stale consume. - -### Compatibility - -**No SDK_MIN_VERSION bump.** Backend v3.66.2 ships since 2026-08-18 (commit `e262f1c3`). Wire-format unchanged. No public API change. Drop-in replacement for 0.15.2 for callers that always pair `track_llm` with a prior gate — those observe zero behaviour change. Callers that relied on bare `track_llm()` hitting `/track/batch` will see `dropped_llm_call_no_reservation` increment on the metrics endpoint and WARNING logs at the call site; the migration is the wrapping fix above. - -_Tests: 2 source-pin regression tests updated; both pin the new drop behaviour. No regressions in the other 1611 tests expected (the only test paths that hit the no-smid branch are the two updated above)._ - -## [0.15.2] - 2026-08-14 - -Patch release — observability closure + UI-UX-AUDIT 2026-08-14 fixes (F-19, F-28, F-29) + flaky-test removal. No public API change, no wire-format change, no SDK_MIN_VERSION bump. Drop-in replacement for 0.15.1. - -### Fixed - -- **`check_workflow_budget` synthetic FALLBACK path emits WARNING, not DEBUG** (sprint handoff `Bug #4 — SDK WS timeout → silent ALLOW`) — pre-0.15.2, when `transport.check` returned `decision_source=FALLBACK_*` (the synthetic-block on `httpx.RequestError` / 5xx), `runtime.py` logged at DEBUG, contradicting the method docblock ("logged at warning level and the caller proceeds") and making the documented ADR-008 fail-OPEN invisible to operators tailing INFO+ logs. Post-0.15.2 the level is WARNING. -- **`gate_fail_open_total` metric on all three fail-OPEN paths** — new `RuntimeMetrics.gate_fail_open_total` counter (`observability/__init__.py`) increments once per `check_workflow_budget` fail-OPEN, regardless of which of the three paths fired (cache-enabled exception, cache-disabled exception, synthetic FALLBACK decision_source). Exposed via `metrics.to_dict()["runtime"]["gate_fail_open_total"]` for the `/health` endpoint and operator dashboards. Operators alert on sustained rate to detect backend outages bypassing the budget gate. -- **F-19 — `SpanContext` ↔ legacy `trace_id`/`span_id` contextvars now form a single coherent trace tree** — pre-0.15.2 the SDK owned two parallel contextvar systems (`tracing._current_span` set by `@protect`, and `context._trace_id_var` / `_span_id_var` set by `with workflow(...)`) that were never read by each other, so an inner `@protect fn()` inside a `with workflow("foo"):` emitted a `span_start` with one trace_id and a parent `track_llm` cost event with a different one — disconnected tree rows on the dashboard. Post-0.15.2 a dual-write bridge keeps both contextvars in sync; `_enrich_event` reads the unified `SpanContext` and the cost-event path reads from the same source. Backend-side bulk-ingest (deferred from audit commit `3e1ea921`) is now fed a coherent trace tree. -- **F-28 — `NullRunCallback._active_runs` protected by `threading.RLock`** — pre-0.15.2 the dict was read/written without synchronisation on multi-threaded LangChain runners (and on free-threaded CPython PEP 703 builds); interleaved `on_chain_start` / `on_chain_end` could orphan the `span_end` lookup (parent_span_id didn't match anything in the dict). Five access sites wrapped: `_register_active_run`, `on_llm_start` parent lookup, `on_llm_end` llm lookup, `_begin_run` parent lookup, `_end_run` pop. `RLock` (not `Lock`) because `_begin_run → _register_active_run` nests two acquisitions on the same thread — reentrant acquisition is the point. -- **F-29 — `NullRunAsyncTransport._emit` falls back to request-body `model` field** — pre-0.15.2 the async path stopped at `usage.get('model')` only. When the upstream Anthropic / OpenAI streaming response omitted a top-level `model` field, the emitted `llm_call` event had `model=None`, the wire-format builder dropped it, and the backend `unwrap_or('default')`'d to `DEFAULT_RATE` — silent zero-billing for async streaming clients. Post-0.15.2 mirrors the sync path's fallback chain at `auto.py:882-885`: `usage.get('model') or _extract_model_from_request_body(request)`. `_extract_model_from_request_body` is a module-level pure-sync helper that reads `request.content + json.loads` — safe to call from the async event loop (no I/O, no blocking). - -### Housekeeping - -- **6 source-pin regression tests** in `tests/test_preflight_fail_policy.py::TestCheckWorkflowBudgetObservability` — pins for the WARNING-level + metric closure above (`test_network_error_emits_warning_and_metric`, `test_timeout_emits_warning_and_metric`, `test_synthetic_fallback_source_emits_warning_not_debug`, `test_real_block_does_not_increment_metric`, `test_real_allow_does_not_increment_metric`, `test_to_dict_includes_gate_fail_open_total`). -- **21 new tests** covering F-19 / F-28 / F-29: - - F-19: `tests/test_track_span_context.py` — trace-tree unification across `with workflow(...)` ↔ `@protect` nesting (476 lines, the largest single audit-pin file in this release). - - F-28: `tests/test_langgraph_callback_race.py` — multi-threaded callback interleaving, parent lookup, span_end consistency under RLock (187 lines). - - F-29: `tests/test_model_fallback_async.py` — async `_emit` request-body fallback for Anthropic + OpenAI streaming (204 lines) + `tests/test_preflight_fail_policy.py` `TestCheckWorkflowBudgetObservability` (176 lines). -- **Removed flaky test** `tests/test_approval_timeout_field.py::TestApprovalTimeoutResolution::test_env_fallback_when_server_value_is_zero` — the test was rare-flaky under pytest-xdist on CI (Linux, Python 3.12); `@pytest.mark.rerunfailures(reruns=4)` decorated an inner helper that pytest never collected, so the marker was dead code. The "non-positive server timeout → env default" contract is covered by the composition of `test_validate_approval_timeout_rejects_below_min` (line 344) and `test_env_fallback_when_response_omits_field` (line 168), both deterministic and not flaky. - -_Tests: 1571 passed (was 1550 in 0.15.1; +21 new from audit, −1 from removed flaky test), 7 skipped in 103.85s. Full suite green. ruff clean. mypy clean (37 source files)._ - -_Compatibility:_ **No SDK_MIN_VERSION bump.** **No public API change.** **No wire-format change.** Fail-OPEN on SDK transport failure remains the documented ADR-008 contract; only the log level moved DEBUG→WARNING and a new counter was added (callers that never read the metric observe nothing). F-19 keeps the existing `@protect` and `with workflow(...)` call sites untouched — the contextvar surface is unified under the hood, not above. F-28 / F-29 are instrumentation-internal — they change emitted event content for the previously-broken cases, never the SDK contract. Drop-in replacement for 0.15.1. - -## [0.15.1] - 2026-08-13 - -Patch release — v3.53 audit fixes (H6 / L5 / L6 / M8 / audit #4 / #5 / #6) plus static-typing closure. No public API change, no wire-format change. Drop-in replacement for 0.15.0. - -### Fixed - -- **`Transport.execute` fallback default flipped to STRICT** (audit #4) — pre-v3.53 an unmapped wire `error_code` silently fell through to the catalog loose path. Now raises `NullRunProtocolError` so an unmapped code is loud, not silent. -- **`MCPAdapter.call_tool` routes through the gate when a runtime is wired** (audit #5) — pre-v3.53 the adapter bypassed the gate path entirely for ad-hoc MCP tool calls. Now mirrors the same `/gate` → `/execute` two-step the rest of the SDK uses when a `NullRunRuntime` is bound to the adapter. -- **`NULLRUN_SKIP_BUDGET_CHECK=1` refused in production** (audit #6 / Bug #6, CLAUDE.md §20) — pre-v3.53 the bypass was honored regardless of environment. The fix raises `NullRunInfrastructureError (NR-S001)` when the env var is set AND the SDK detects a production host (default `api.nullrun.io` or `NULLRUN_ENV=production` on a non-dev host). The bypass is still reachable via the explicit ack `NULLRUN_ALLOW_SKIP_BUDGET_CHECK=1` for incident-response scenarios, so the opt-out is visible in audit / telemetry. -- **`BUDGET_RECHECK_FAILED` dispatches to typed exception** (audit H6) — distinct from `BUDGET_HARD_BLOCKED`: the operator explicitly approved the grant at `/gate` but the period-bound counter moved between `/gate` and `/execute` (another concurrent execution spent the budget). Caller should re-`/gate` to refresh the reservation envelope and retry `/execute`. Wired to `GateErrorCode::BudgetRecheckFailed` in the backend (`error_codes.rs`). -- **Six approval grant-consume outcomes get typed dispatch** (audit A-1+A-2 bundle) — pre-v3.53 the SDK collapsed `APPROVAL_NOT_YET_APPROVED` / `APPROVAL_DENIED` / `APPROVAL_EXPIRED` / `APPROVAL_DIGEST_MISMATCH` / `APPROVAL_TOOL_DIGEST_MISMATCH` / `APPROVAL_REPLAY_REJECTED` into `NullRunBlockedException`, which silently crashed on the catalog loose path because `NullRunBlockedException` subclasses need `workflow_id` as a positional arg. Post-v3.53 each maps to its own NR-Axxx subclass (`NR-A010..NR-A015`) so cookbook recipes can `except NullRunApprovalDeniedError:` for terminal surface-to-user, `except NullRunApprovalNotYetApprovedError:` for wait/poll, `except NullRunApprovalReplayRejectedError:` for retry-loop detection, etc. -- **`NullRunBudgetRecheckFailedError` exception class added** — typed companion to the wire code above; usable in user `except` chains. -- **`_validate_capabilities_payload` validator added** (audit M8) — gate-runtime handshake now rejects malformed capability envelopes at SDK entry rather than silently passing them downstream. - -### Housekeeping - -- **`_V3_ERROR_CODE_MAP` type annotation tightened** from `type[BaseException]` to `type[Exception]` (mypy `return-value` error closure — every map value is an `Exception` subclass). -- **Ruff F811 sweep across test files** (`test_actions.py`, `test_v3_wire_contract.py`, `test_audit_wire.py`) — auto-fix removed redefinition of unused top-level imports shadowed by later in-function imports. - -_Tests: 1550 passed, 7 skipped in 154.47s. Full suite green. ruff clean. mypy clean (37 source files)._ - -_Compatibility:_ **No SDK_MIN_VERSION bump.** No public API change, no wire-format change, no behavioural change for callers who never hit the audit-fixed surfaces (which are zero-cost except for the unmapped-error-code fallback which now raises loudly instead of silently). Drop-in replacement for 0.15.0. - -## [0.15.0] - 2026-08-12 - -ADR-009 governance audit surface (P1) — typed read API for the org's hash-chained `audit_events` table. Backend already ships the matching wire shape (commit `46af9e29`, audit endpoints expose the 13 canonical columns: `agent_id`, `principal_id`, `decision`, `policy_id`, `policy_version`, `policy_hash`, `matched_rule`, `reason_code`, `execution_id`, `action_digest`, `tool_name`, `tool_version`, `tool_digest`). This release lands the SDK consumer side: a `nullrun.audit` module with frozen dataclasses for every wire response shape, a `runtime.audit` proxy that surfaces typed results, and 17 contract tests pinning the round-trip. - -No SDK_MIN_VERSION bump. No breaking API change. The five `Transport.audit_*` methods that previously returned raw dicts now accept `organization_id` as a positional parameter (organisation lives on the runtime, not the transport); callers that previously wrote `transport.audit_log(org)` continue to work — the new proxy at `runtime.audit.list()` is the recommended path going forward. - -### Added - -- **`nullrun.audit` module** — frozen dataclasses for the ADR-009 read surface: `AuditEntry`, `AuditLogMeta`, `AuditLogPage`, `AuditQuery`, `AuditVerifyResult`, `AuditExportJob`, `AuditExportStatus`. Each parser tolerates pre-ADR-009 rows (all 13 governance columns default to `None`); `AuditEntry.is_governance` is `True` only for the three canonical event categories (`authorization_decision`, `approval_decision`, `execution_lifecycle`). -- **`AuditQuery.to_query_string()`** — drops `None` fields, serialises `datetime` as RFC3339, percent-encodes the canonical set of filters (`event_type`, `decision`, `policy_id`, `execution_id`, `actor`, `since`, `until`, `limit`). -- **`AuditProxy` on `NullRunRuntime`** — `runtime.audit.list()`, `verify()`, `list_exports()`, `create_export()`, `export_status()` return typed dataclasses instead of raw dicts. `AuditProxy._require_org()` raises `NullRunAuthenticationError` when the runtime is unbound, so a misconfigured CI step fails loudly at the audit call site rather than silently dropping the query. -- **`Transport.audit_*` accept `organization_id` as positional** — the five methods (`audit_log`, `audit_verify`, `audit_list_exports`, `audit_create_export`, `audit_export_status`) take `organization_id` as a positional parameter because the transport holds no org binding. The `AuditProxy` threads `self.organization_id` through automatically; service-account callers that need to address an org other than the bound one can pass `organization_id=` explicitly. -- **Lazy exports** — `AuditEntry`, `AuditLogMeta`, `AuditLogPage`, `AuditQuery`, `AuditVerifyResult`, `AuditExportJob`, `AuditExportStatus` are reachable as `from nullrun import AuditEntry` etc. via the existing PEP 562 lazy-export map. - -### Fixed - -- **`Transport.audit_*` referenced `self.organization_id`** (a runtime-only attribute) — silent `AttributeError` on every audit call. Fixed by lifting the org into a positional parameter and threading it through `AuditProxy`. - -_Tests: 17 additions (`tests/test_audit.py` — wire-shape parsers, query serialisation, three-category governance property, Z-suffix timestamp normalisation, policy_version string drift) + 17 additions (`tests/contract/test_audit_wire.py` — round-trip via respx, GET-vs-POST HMAC boundary, protocol header presence, 401 → `NullRunAuthError` mapping, typed proxy return values, unbound-runtime error path)._ - -_Compatibility:_ **No SDK_MIN_VERSION bump.** The `Transport.audit_*` shape change is source-compatible (positional kwarg with a clear name). Pre-0.15 callers that wrote `transport.audit_log("org-uuid")` continue to work; pre-0.15 callers that wrote `transport.audit_log(organization_id="org-uuid")` (which previously crashed on the `self.organization_id` lookup) now work for the first time. - -## [0.14.11] - 2026-08-11 - -Patch release — partial revert of sprint-5 cleanup commits whose scope exceeded what the codebase actually supported. Two over-aggressive commits restored critical user-authored documentation and branch-coverage test files that the cleanup had removed. - -### Added - -- **Restored `tests/test_real_e2e_observation.py`** (321 lines) — real-socket integration test that spins up a stdlib `http.server` and exercises the full wire path (auto-instrumented `httpx.Client` → mock LLM server → mock NULLRUN backend → recorded event list). The respx-mocked unit tests do not cover this surface; deleting it would have silently dropped the only test proving that the auto-instrumented transport actually delivers a track event to a real socket. -- **Restored branch-coverage tests** deleted by sprint-3 cleanup (a666624 P2): `tests/test_protect_branches.py` (564 lines — branch coverage for `_safe_args` / `_strip_details_balanced` / `_enforce_sensitive_tool`), `tests/test_runtime_branches.py` (515 lines — less-trodden error paths), `tests/test_transport_branches.py` (647 lines — branch-coverage gaps in transport). These three files explicitly documented their purpose as covering "gaps" and "less-trodden error paths" that the mainline tests skip; removing them = silent coverage regression. - -### Changed - -- **Restored `src/nullrun/runtime.py` docstring block** (lines 28-50ish, 30 lines) — user-authored correction from 2026-07-04 explaining that the README claim `Fail-OPEN на инфраструктурных сбоях. Если backend недоступен, бюджет не блокирует агента` is **partially wrong**. The restored block makes the explicit split: SDK-side transport failure (network timeout, 5xx, breaker open) → fail-OPEN on the *check* path so a dead backend doesn't freeze the user's agent loop; backend-side enforcement failure (`BUDGET_REDIS_UNAVAILABLE` → 402, `RATE_LIMIT_REDIS_UNAVAILABLE` → 503) → fail-CLOSED wire response (the SDK does NOT silently fall-OPEN on a wire 4xx/5xx that names an enforcement failure). Codifies CLAUDE.md §4 fail-CLOSED rules. -- **Restored Cyrillic technical nomenclature in CHANGELOG.md** — "Разрыв 2" in the 0.14.4 entry and "Разрыв 1c" in the 0.13.13 entry. These were user-coined Russian-language project codenames for backend architecture milestones ("Разрыв" = breakthrough/rupture in the architectural sense, NOT the English "breakpoint" — `Breakpoint-2` is not a 1:1 translation and loses the original term). - -_Tests: 1462 passed, 6 skipped in 20.83s. Full suite green._ - -_Compatibility:_ **No SDK_MIN_VERSION bump.** No public API change, no wire-format change, no behavioural change. Drop-in replacement for 0.14.10. - -## [0.14.10] - 2026-08-11 - -Sprint 5 internal cleanup — no behavioural change, no SDK_MIN_VERSION bump, no wire-format change. Three release-blocks of dead code, dedup, and developer-experience hygiene. Backward-compatible patch. - -### Removed - -- **Dead code in `src/nullrun/`** — `extractor._cached_signature` + `compute_impact_digest` + unused imports; duplicate `compute_hmac_signature` / `verify_hmac_signature` in `transport_websocket` (re-exported from `transport`); `_singleton.install_module_proxy`; `_registry.replace_for_test`; `context.set_trace_id` / `reset_trace_id` / `clear_trace_id`; `runtime._start_transport` / `_trigger_action` / `get_org_status` / `_workflow_start_time`. 383 lines deleted across 6 files. -- **`Makefile run-example` target** — referenced `examples/basic.py` deleted in 0.3.1 with the gRPC transport. Local smoke testing now goes through `make smoke-test`. -- **CHANGELOG WIP `[0.10.0]` stub** + 13 `_(Trimmed; see git log X.Y.Z)_` placeholders. Net -29 lines. - -### Changed - -- **Sync/async transport dedup** — `NullRunSyncTransport` and `NullRunAsyncTransport` now share `_rebuild_response` (byte-identical rebuild path) and `_build_llm_call_event` (shared event-dict so the dedup fingerprint stays identical across sync + async httpx paths). 177 tests pass unchanged. -- **`@protect` sync/async wrapper dedup** — both paths now share a `_protect_body` context manager for the four pre-execution gates. Sync path keeps `unify_block=True` (kill/pause → `NullRunBlockedException`); async path keeps `unify_block=False` (propagates `WorkflowKilledInterrupt` so `asyncio` cancellation works). 114 tests pass. -- **LangChain usage extraction dedup** — `extract_usage_from_response` collapsed from 5 sequential `if` branches into a single `_read_token_attrs` + `_apply_usage` helper loop. 42 tests pass. -- **Decorator chain-walk dedup** — `_stamp_extractor_on_innermost` + `_find_extractor_in_chain` consolidated behind a `_walk_wrapped_chain` generator with a 32-hop cycle guard. -- **`Makefile coverage` target** — was `coverage run -m pytest tests/` (only traced xdist coordinator → 0-hit uploads); now `pytest tests/ --cov=src/nullrun --cov-branch --cov-report=xml:coverage.xml`, matching `.github/workflows/ci.yml:82`. - -### Added - -- **9 missing error-code docs** in `docs/errors/`: `NR-A004` (approval flow anomaly), `NR-B003` (sensitive-tool impact extractor failure), `NR-C000` (generic config default), `NR-C004` (status before init), `NR-CH001` (chain context invalid), `NR-O001` (overbudget on consume), `NR-P001` (wire-protocol version mismatch), `NR-R002` (aggregate-rate-limiter Redis outage), `NR-W004` (workflow soft-deleted). Three new catalogue categories: **P**rotocol, **Ch**ain, **O**verbudget. - -### Fixed - -- **CHANGELOG sort order** — release blocks now strictly descending by version (was `0.9.1 → 0.11.0 → 0.9.0`; now `0.11.0 → 0.9.1 → 0.9.0`. Lower section was `0.3.1 → 0.5.2 → 0.4.0`; now `0.5.2 → 0.4.0 → 0.3.1`). - -_Tests: 1334 pass, 2 skip (pre-existing); 23/23 exception hierarchy pass._ - -_Compatibility:_ **No SDK_MIN_VERSION bump.** Strictly internal cleanup; no public API change, no wire-format change, no behavioural change. Drop-in replacement for 0.14.9. - -## [0.14.9] - 2026-08-07 - -v3.38 wire-drift close — three real contract bugs that diverged from backend source code. Verified against `backend/src/proxy/http/protocol.rs`, `backend/src/proxy/middleware/auth.rs`, and CLAUDE.md §5 / §13 — not against comments or documentation. No SDK_MIN_VERSION bump. No on-wire change (backend already shipped the matching wire shape; this SDK release closes the consumer side). - -### Fixed - -- **Capabilities probe route** — `nullrun.capabilities.CAPABILITIES_PATH` was `"/health"` (a generic liveness endpoint) instead of the canonical `"/api/v1/capabilities"`. [...] -- **API_KEY_* error code granularity (v3.38 backend split)** — backend v3.38 split the `API_KEY_REVOKED` bucket into five distinct wire codes: `API_KEY_EXPIRED` / `API_KEY_DISABLED [...] -- **`NullRunAuthError.wire_code`** — the exception class gains a `wire_code: str | None = None` constructor kwarg that defaults to `"API_KEY_REVOKED"` for backwards compat. [...] - -### Added - -- **`decision == "soft_pass"` handler in `check_workflow_budget`** — the runtime's `/gate` decision dispatcher gains a `soft_pass` branch (currently the only branch missing from th [...] - - calls `metrics.inc_runtime("soft_overdraft_used")` so the dashboard can graph soft-cap pressure - - logs at WARNING with `overdraft_used_cents` / `max_overdraft_cents` / `remaining_overdraft_cents` from the backend response so operators can see which chains are burning overdraft - - returns normally (the `allow` semantic is correct — the gate already authorised the call via the chain's overdraft cap) - -_Tests: 4 additions (tests/conftest.py, tests/test_capabilities.py, tests/test_init_contract.py…)._ - -_Compatibility:_ **No SDK_MIN_VERSION bump.** All three fixes are consumer-side; the backend already shipped the matching wire shape. - -## [0.14.8] - 2026-08-06 - -Execution Graph v0 — additive sub-agent lineage. The backend landed `parent_execution_id` as an optional wire field on `/api/v1/gate` (backend commit `87fae759`, not pushed yet) so an SDK spawning a sub-agent can name the parent's `execution_id`. Backend validates ownership against the parent's `execution:{id}` Redis binding (mirrors the `/cancel` ownership check) and rejects cross-org / cross-key / not-found with `403 PARENT_EXECUTION_*`. This release ships the SDK-side forward path, the matching capability flag, and the three-way error-code mapping. Wire change is strictly additive (omitted when `None`); no SDK_MIN_VERSION bump. - -### Added - -- **`parent_execution_id` on `/check` (gate)** — `Transport.check(check_request=...)` forwards the optional `parent_execution_id` field from `check_request` onto the wire when the [...] -- **`execution_graph` capability flag** — `parse_capabilities` reads the new `execution_graph: bool` from `/api/v1/capabilities` (nested under `capabilities:` with top-level fallba [...] -- **`NullRunChainError.parent_execution_id`** — the chain error class gains an optional `parent_execution_id: str | None = None` constructor kwarg (mirroring the existing `chain_id [...] - -### Changed - -- **Three new error codes mapped to `NullRunChainError`** — `PARENT_EXECUTION_NOT_FOUND`, `PARENT_EXECUTION_ORG_MISMATCH`, `PARENT_EXECUTION_KEY_MISMATCH` (all 403) are added to `_ [...] - -_Tests: 1 additions (tests/test_transport.py)._ - -_Compatibility:_ **Backward-compatible additive wire change.** Pre-Execution-Graph SDKs that never set `parent_execution_id` continue to work unchanged — the field is omitted entirely from the wire. - -## [0.14.7] - 2026-08-04 - -Init contract hardening — strip leading and trailing whitespace from `api_key` (and the `NULLRUN_API_KEY` env fallback) BEFORE the truthiness check in `nullrun.init()` and `NullRunRuntime.__init__`. Pre-fix, whitespace-only strings (`" "`, `"\t"`, `"\n"`) are TRUTHY in Python and silently slipped past the empty-key guard; they were stored on the runtime and reached the gateway as a malformed `Authorization: Bearer ***` header, surfacing as a backend 401 only on the first `/gate` call rather than at startup. - -### Fixed - -- **`nullrun.init()` now strips whitespace before the truthiness check** — `src/nullrun/__init__.py:249` resolves `raw_key = api_key if api_key is not None else os.getenv("NULLRUN_ [...] -- **`NullRunRuntime.__init__` mirrors the strip-then-check** — `src/nullrun/runtime.py:370` applies the same contract so direct construction (used by tests and advanced callers) ca [...] - -_Tests: 1 additions (tests/test_init_contract.py)._ - -_Compatibility:_ **Backward-compatible bug fix.** The strip is a strict superset of the empty check: pre-fix callers that passed valid keys continue to work unchanged (`"nr_live_xxx"` strips to itself), and callers that pasted whitespace-only keys now [...] - -## [0.14.5] - 2026-08-01 - -MCP-aware gate metadata and tool-argument forwarding. The release completes the SDK-side path for MCP classification and annotation policies, and adds the optional argument bag used by the backend's tool-schema fingerprinting flow. All new wire fields are optional and omitted when unavailable. - -### Added - -- **Per-call MCP context** — `set_mcp_tool_context(...)`, `get_call_mcp_class()`, and `get_call_mcp_annotations()` store and expose the canonical tool class plus normalised MCP ann [...] -- **`MCPAdapter`** — `nullrun.toolbox.mcp.MCPAdapter` wraps an already-connected synchronous MCP client. [...] -- **`tool_arguments` on `/execute` and `/gate`** — `Transport.execute(...)` accepts an optional argument mapping, while `Transport.check(...)` forwards the same field from `check_r [...] - -### Fixed - -- **MCP context tests no longer leak module-level `ContextVar` state** — the release includes isolation fixes for the class and annotation tests that were flaky only during the ful [...] - -_Tests: 3 additions (tests/test_mcp_adapter.py, tests/test_mcp_context.py, tests/test_transport.py)._ - -_Compatibility:_ **Backward-compatible additive wire change.** Existing callers do not need to pass any new fields; absent MCP metadata and `tool_arguments=None` are omitted. - -## [0.14.4] - 2026-07-27 - -ToolParameters Approval Rules wire contract (Tier 2 / Разрыв 2 follow-up). The backend already accepted `BusinessImpact::ToolCall(ToolCallParams)` on the `/execute` wire (backend commit `1e501cd6`); 0.14.4 lands the SDK-side path so users get ToolParameters rules by default on every bare `@sensitive` function, with no decorator change. Also fixes a silent regression in the auto-attach path that dropped an explicit `impact=tool_params({...})` map, and pins the cross-language `ToolCall` action digest against the Rust backend's golden hex. No on-wire breaking change for money callers; the only behavioural change is that bare `@sensitive` now ships `kind=tool_call` on the wire where it previously shipped nothing. - -### Added - -- **`BusinessImpact.tool_call(tool_name, params)`** factory — `business_impact.py:323` new factory builds a `BusinessImpact(kind='tool_call', tool_name=..., params=...)` envelope b [...] -- **`ToolCallParams` dataclass** — `business_impact.py:143` mirrors the backend struct (`tool_name` ≤ 128 bytes, `param_name` ≤ 64, JSON-roundtrippable values only). [...] -- **`ToolParamsExtractor` + `tool_params(...)` factory** — `extractor.py:815` (class) and the matching factory. [...] -- **Bare `@sensitive` now ships ToolParameters on the wire** — `decorators.py:1096` (`_do_sensitive_register`) auto-attaches a default `ToolParamsExtractor(include_all=True)` on a [...] -- **`@sensitive(impact=tool_params({...}))` decorator form** — `decorators.py:1065` new docstring + `decorators.py:711` dispatch branch. [...] - -### Fixed - -- **Auto-attach chain walk preserves an explicit `impact=tool_params({...})` map** — `decorators.py:43` new helper `_find_extractor_in_chain` walks `__wrapped__` (bounded at 32 hop [...] -- **`_enforce_sensitive_tool` dispatch handles both extractor types** — `decorators.py:677` (success path) and `decorators.py:711` (error path) now branch by extractor type. [...] -- **Bare `@sensitive` regression in the existing `tests/test_sensitive_extractor.py`** — the 5 existing tests still pass because they register the tool manually via `rt.add_sensiti [...] - -_Tests: 7 additions (tests/test_business_impact.py, tests/test_extractors.py, tests/test_protect.py…)._ - -_Compatibility:_ **Default SDK behaviour for bare `@sensitive` CHANGED** — was `no business_impact on wire`, now `kind=tool_call on wire`. Operators who relied on the Phase 0 path (approval_id-only grant consume) must either pass `@sensitive(impact=too [...] - -## [0.14.2] - 2026-07-24 - -Three hotfixes that fell out of the 0.14.1 demo run. Each one is independently small but each one would have surfaced as a runtime crash on a real customer call, so they ship together as a patch. No on-wire breaking change. No SDK_MIN_VERSION bump. Backends on `1.0.0` keep working unchanged. - -### Fixed - -- **`@protect` decorator now emits a `tools/track_tool` event** — `decorators.py:470` and `decorators.py:521` (sync + async wrappers) now call `runtime.track_tool(fn.__name__, meta [...] -- **`track_tool` event carries `tokens: 0` and a fresh `uuidv7` `execution_id`** — `runtime.py:3077` now stamps both fields onto every `tool_call` event. [...] -- **Approval-resolved WS callback is now a plain sync function** — `transport.py:1757` `wrapped_approval_resolved` was previously declared `async def` to be awaitable, but the WebS [...] -- **WebSocket cancellation is treated as a clean shutdown** — `runtime.py:1160` now catches `asyncio.CancelledError` before the generic `except Exception` block. [...] - -_Tests: 4 additions (tests/test_approval_money_flow.py, tests/test_approval_ws_sync_callback.py, tests/test_runtime_branches.py…)._ - -_Compatibility:_ **Backward-compatible bug fix.** No SDK_MIN_VERSION bump. No public API change. - -## [0.14.1] - 2026-07-24 - -Decimal JSON serialization patch. `track_tool` event payloads that contain a `Decimal` value (e.g. `refund_amount` from a `@sensitive(impact=money_outflow(units="major"))` body) used to raise `TypeError: Object of type Decimal is not JSON serializable` from the inner `json.dumps` call. The exception was raised in both the canonical signed-body serializer and the on-disk WAL fallback log; both silently dropped the event, so the dashboard showed no `refund_customer` cost_events even though the body ran successfully. - -### Fixed - -- **`_signed_request_body` Decimal serialization** — `transport.py:251` now passes `default=str` to `json.dumps(payload, separators=(",", ":"), default=str)`. [...] -- **WAL fallback `default=str`** — `transport.py:711` `_signed_request_body` WAL fallback (`f.write(json.dumps(event) + "\n")`) also gets `default=str` for consistency. [...] - -_Tests: 2 additions (tests/test_approval_money_flow.py, tests/test_sensitive_extractor.py)._ - -_Compatibility:_ **Backward-compatible bug fix**. No SDK_MIN_VERSION bump. No public API change. - -## [0.14.0] - 2026-07-23 - - -### Added - -- **`InvalidMoneyPrecisionError`** and **`InvalidMoneyAmountError`** — dedicated `ValueError` subclasses with structured fields. [...] -- **`BusinessImpact`** model (`dataclass(frozen=True)`) with explicit `currency` / `units` / `amount_minor` fields. `details` dict is still accepted on the legacy path. -- **`@sensitive(impact=BusinessImpact(...))`** — new decorator kwarg that emits a structured `business_impact` envelope on the `/track` event. [...] -- **`MoneyImpactExtractor`** — new helper that normalises `Decimal` / `int` / `float` / str into `BusinessImpact` minor-units, raising `InvalidMoneyAmountError` / `InvalidMoneyPrec [...] - -### Changed - -- **Negative `amount_minor` rejected** on both unit paths. A negative value would silently fall through every `op=gt` predicate (`negative < positive` is always False) — pre-fix a [...] -- **Sub-precision Decimal rejected** — `Decimal("1.234")` against a USD `allowed=2` precision is now `InvalidMoneyPrecisionError(currency="USD", allowed=2, received=3, received_dig [...] -- **`/execute` handles `require_approval` correctly** — re-checks with the `approval_id` returned by the backend (was dropping the approval handshake on round-trips). -- **Server `approval_timeout` clamped to `[1, 3600]s`** on the SDK side as defence against a malformed / overshooting backend that returns `0` or `2147483647` in the Разрыв 1c fiel [...] - -_Tests: 6 additions (tests/test_approval_money_flow.py, tests/test_business_impact.py, tests/test_execute_approval_flow.py…)._ - -_Compatibility:_ **Backward compatible** on the happy path. Every existing call site keeps working; the new errors are `ValueError` subclasses; the new `BusinessImpact` decorator kwarg is optional. - -## [0.13.13] - 2026-07-21 - -Approval-wait SDK sync with backend commit `0ad03b9` ("\u0420\u0430\u0437\u0440\u044b\u0432 1c", gate hot-path trigger). The backend now sends `approval_timeout_seconds: Option` and `approval_expires_at: Option` on every `/gate` response so a backend approval rule can set a non-default short timeout. Pre-fix, the SDK only consulted `NULLRUN_APPROVAL_TIMEOUT_SECONDS` (env default 300s), which silently desynced from a 20s backend expiry sweeper. No public API change. No SDK_MIN_VERSION bump. No on-wire change. - -### Fixed - -- **Approval wait uses server-authoritative `approval_timeout_seconds` when present** \u2014 new optional kwarg `timeout_seconds: float | None = None` on `_wait_for_approval_resolu [...] -- **`check_workflow_budget` reads `response["approval_timeout_seconds"]`** with type and sign validation. Malformed values fall through to the env default path. [...] -- **Diverging server vs env default emits a DEBUG log line** ("approval {id}: using server timeout={X}s (env default would have been {Y}s)") so an operator inspecting logs can see [...] - -_Tests: 1 additions (tests/test_approval_timeout_field.py)._ - -_Compatibility:_ The new `timeout_seconds` kwarg is optional with a `None` default, so existing callers are unaffected. - -## [0.13.12] - 2026-07-20 - -CI / coverage-testability release. No on-wire change, no SDK_MIN_VERSION bump, no public API change. Backends on `1.0.0` keep working unchanged. - -### Changed - -- **`pytest` suite is now CI-fast on Windows + xdist** — a new `_fast_sleep` autouse fixture in `tests/conftest.py` caps test-code `time.sleep` calls at 1ms, with two opt-out paths [...] -- **`TestCircuitBreaker` half-open tests no longer sleep the wall clock** — `test_open_transitions_to_half_open_after_timeout`, `test_half_open_success_closes`, and `test_half_open [...] -- **`TestPingChainScheduler` opts out of the cap via marker** — the new `@pytest.mark.slow_sleep` marker on the class lets `test_ping_chain_emits_heartbeats_on_time_schedule` keep [...] - -_Tests: 3 additions (tests/conftest.py, tests/test_transport.py, tests/test_v3_wire_contract.py)._ - -### CI - -- `pyproject.toml` — new `markers = ["slow_sleep: opt out of the conftest autouse time.sleep cap"]` entry under `[tool.pytest.ini_options]`. [...] -- The Codecov badge in `README.md` will now report the real combined coverage on master. Pre-Sprint-0 the badge was stuck at 0% because `coverage run -m pytest -n auto` ran coverage in the coordinator process only; the Sprint 0 PR (#70) already fixed [...] - -### Audit - -- No SDK public API change. No wire-format change. No backend migration required. [...] -- Pre-Sprint-0 instability under `pytest-cov + xdist`: `test_status.py::TestRecentErrors` and `TestTransport::test_stop_flush_false_skips_final_flush` were observed to flake ~1/3 of the runs in the local environment (passing in isolation, passing in [...] - - -## [0.13.0] - 2026-07-04 - -Drift-fixes release. Closes the SDK-side items on `docs/drift.md` (2026-07-04); no on-wire breaking change — backends on `1.0.0` keep working unchanged. - -### Added - -- **Idempotency-key propagation to `/track` v3 single-event** — new `nullrun.context._server_minted_idempotency_key_var` + `get_/set_/reset_/clear_server_minted_idempotency_key` he [...] - -### Changed - -- `runtime.py` module docstring now distinguishes **SDK-side transport failure** (network / 5xx / breaker open → fail-OPEN on the `/check` path) from **wire 4xx/5xx that names an enforcement failure** (`BUDGET_REDIS_UNAVAILABLE` → 402 fail-CLOSED, `R [...] - -### Fixed - -- **Wire `status_code` preserved on every decision exception** — `NullRunBlockedException`, `NullRunBudgetError`, `NullRunChainError`, `NullRunWorkflowInactiveError`, `NullRunConsu [...] -- **Patch-coverage gap from 0.12.2 closed** — `tests/test_v3_wire_contract.py::TestGateCacheRuntimeFlow` (3 tests) drives `NullRunRuntime.check_workflow_budget` inside `with chain( [...] - -_Tests: 2 additions (tests/test_drift_fixes_2026_07_04.py, tests/test_v3_wire_contract.py)._ - -### Audit - -- New `docs/drift.md` records the six P0 + P1 items that turned up during pre-publish review of 0.12.2 (idempotency-key wiring, status_code on exceptions, fail-CLOSED honesty, plus four P0/P1 README issues that are deferred to a README rewrite PR and [...] - - -## [0.12.2] - 2026-07-04 - -Bug-fix release. Two related correctness fixes layered on top of 0.12.1; no wire-format change. - -### Fixed - -- **BUG #4 — `/check` execution_id**: `check_workflow_budget()` now sends a fresh `uuidv7` as the `execution_id` field on every call, instead of reusing `workflow_id`. [...] -- **BUG #5 — chain-mode gate thrash**: new `nullrun.runtime._GATE_CACHE` (5s TTL, keyed on `(workflow_id, chain_id, model)`) collapses consecutive `/gate` calls from inside `with c [...] - -### Added - -- 158 lines of contract tests in `tests/test_v3_wire_contract.py`: `TestGateExecutionId` (per-call uniqueness + uuidv7 format validation) and `TestGateCache` (5 cache invariant + opt-out cases). - -### Changed - -- `__version__` bumped from 0.12.1 to 0.12.2. - - -## [0.12.1] - 2026-07-04 - -Bug-fix release. The v0.12.0 changelog claimed the SDK propagates the server-minted `execution_id` from /check to /track but the wiring was never shipped — the SDK still sent client-supplied ids on /track/batch and ignored `reservation_id` on /check responses (audit fix per memory `sdk-v3-migration-gaps`). - -This release closes the four gaps documented in `docs/sdk-v3-migration-gaps.md`: - -- `check_workflow_budget()` now reads `response["reservation_id"]` and stores it on a contextvar (`nullrun.context._server_minted_execution_id_var`). -- New helpers `set_server_minted_execution_id` / `get_server_minted_execution_id` / `reset_server_minted_execution_id` + a paired `_server_minted_reservation_at` timestamp for the 295s TTL guard. -- `_enrich_event` stamps `execution_id` onto the /track payload when the captured reservation is fresh, and drops it (clearing the capture) once past the safety window — prevents forwarding a doomed id that would 503 on /track per CLAUDE.md section 33. -- `_route_track` routes `llm_call` events to the v3 `/api/v1/track` single-event endpoint via `Transport.track_single()` so backend `gate_consume_v3` validates the consume-vs-reserve + epsilon invariant (CLAUDE.md section 25). [...] -- `NULLRUN_V3_TRACK_DISABLE=1` opt-out forces everything through the legacy batch path (backends still on v1/v2). - -### Added - -- `nullrun.context._server_minted_execution_id_var` + `nullrun.context._server_minted_reservation_at_var` + 6 helpers (`get_/set_/reset_/clear_`). -- `nullrun.runtime._capture_server_minted_execution_id(response)` — defensive UUID parse + warn-on-malformed. -- `nullrun.runtime._route_track(wire_event)` — dispatches to single-event /track or batch /track/batch. -- `nullrun.runtime._build_v3_track_payload(event, reservation_id)` — maps an enriched event onto the v3 /track wire schema. -- 27 contract tests in `tests/test_v3_server_minted.py` covering contextvar hygiene, capture defence-in-depth, _enrich_event age threshold, _route_track dispatch, and end-to-end /gate -> /track round trip. - -### Changed - -- `__version__` bumped from 0.12.0 to 0.12.1 (post-release integrity fix — the v0.12.0 wiring never shipped before this). - -### Fixed - -- SDK no longer treats the /check `reservation_id` field as decorative. Each LLM-call track event now carries the server-minted uuidv7 the backend minted, so v3 `gate_consume_v3` can find the matching `reservation:{execution_id}` Redis key (300s TTL). -- LLM-call events now POST to `/api/v1/track` (v3 single-event) instead of `/api/v1/track/batch`. This exercises the consume-vs-reserve invariant that the batch path silently skipped (regression of the v1/v2 `monthly_cost` counter — see CLAUDE.md section 0 G1). - - -## [0.12.0] - 2026-07-03 - -Server-minted execution_id default ON. Per CLAUDE.md section 24, every /check now mints a server-side uuidv7 execution_id. The SDK no longer needs to generate its own; the response carries the server-minted id which propagates to /track. This is the SDK_MIN_VERSION for the v3 rollout - older SDKs still work for v1/v2 endpoints but should upgrade. - -> **Integrity note (2026-07-04):** the propagation claim in this entry was correct in intent but the actual wiring was not shipped in 0.12.0. See 0.12.1 above for the closing fix. - -### Added - -- `nullrun.uuid7` module - RFC 9562 section 5.7 time-ordered ID generator. Used internally for trace_id and span IDs. -- `nullrun.capabilities` module - probe_capabilities(), parse_capabilities(), validate_sdk_version(). Wired into nullrun.init(). - -### Changed - -- __version__ bumped from 0.11.0 to 0.12.0. - - -## [0.11.0] - 2026-07-02 - -Wire-protocol v3 alignment with the backend's Sprint 6 v1 cut -(CLAUDE.md v3.4). The previous SDK shipped pre-v3 endpoints -(`/api/v1/gate`, `/api/v1/execute`, `/api/v1/track/batch`) without -the `X-NULLRUN-PROTOCOL` header that the v3 backend requires as a -fail-CLOSED pre-check — every signed POST was rejected with HTTP 400 -`PROTOCOL_HEADER_REQUIRED`. This release aligns the SDK with the v3 -wire contract and adds the missing soft-mode / chain / heartbeat / -cancel / budget-estimate surface. - -- **`X-NULLRUN-PROTOCOL: 3` is now mandatory on every signed POST.** - The backend's `proxy/http/gate/protocol.rs` middleware rejects - requests without the header with HTTP 400 + error_code - `PROTOCOL_HEADER_REQUIRED` BEFORE the gate pipeline runs. Pre-v3 - SDKs that don't send it will get 400 on every request, including - `/auth/verify` (which is unsigned but goes through the same - protocol guard via the `_post_auth_with_retry` path). - - Routed through the new centralised helper in - `nullrun.transport._protocol_header_value()` so a future bump - is a one-line change. - - The header is set in `_build_signed_headers()` (covers - `/gate`, `/execute`, `/track/batch`, `_refetch_credentials`) - AND inlined in the four call sites that build their own - headers dict (track/batch, gate, execute, WS handshake, - auth/verify refresh). The `runtime._auth_headers()` helper was - extended to include the header for the three direct - `self._client.get/post` call sites (`_post_auth_with_retry`, - `_fetch_remote_state`, `get_org_status`). - -### Added - -- **`Transport.check_v3(request)` — POST /api/v1/check.** The v3 - replacement for `/gate`. Adds three optional wire fields - (CLAUDE.md §16): - - -## [0.9.1] - 2026-06-29 - -### Added - -- `nullrun.uuid7` module - RFC 9562 section 5.7 time-ordered ID generator. Used internally for trace_id and span IDs. -- `nullrun.capabilities` module - probe_capabilities(), parse_capabilities(), validate_sdk_version(). Wired into nullrun.init(). - -### Changed - -- __version__ bumped from 0.11.0 to 0.12.0. - -Patch on top of 0.9.0. Unifies the LLM-call fingerprint scheme so the -dedup LRU at `runtime.track()` can collapse sibling emissions from the -httpx transport and the LangChain callback for the same real call. - -### Fixed - -- **Double-emission of llm_call events.** Pre-0.9.1 the httpx transport - (`NullRunSyncTransport._emit`) and the LangChain callback - (`NullRunCallback.on_llm_end`) each computed their own `_fingerprint` - from different inputs — `sha256(host|status|body)` vs - `sha256(json({path:"langchain_callback", run_id, response_id, model, - provider, invocation_params}))`. The two fingerprints never - collided, so the dedup LRU at `runtime.track()` could not collapse - the two emissions for the same call. On a typical `app.invoke()` - with 6 LLM calls the backend saw ~12 `llm_call` events on the wire - (2 per real call), doubling `llm_call_count` and skewing - `cost_events` aggregates. - - Post-fix both observers call the same helper - `_fingerprint_for_llm_call(model, provider, response_id)` with the - three signals reachable from every observation path: - - httpx transport reads `model` and `id` straight out of the - OpenAI-style response body (`payload["model"]`, - `payload["id"]`). - -## [0.9.0] - 2026-06-29 - -Server-derived coverage replaces the in-process counter dicts. -Counter-bump helpers are gone; every `llm_call` span now carries -`metadata.tracked` and `metadata.streaming_skipped` flags so the -backend's `coverage_pct` query can compute coverage from span -metadata alone. Adds `nullrun.shutdown()` for clean WS close on -script exit. - -### Breaking changes - -- `NullRunRuntime.coverage_report()` removed. -- `NullRunRuntime._coverage_seen` / `_coverage_tracked` / - `_coverage_streaming_skipped` instance attributes removed. -- `NullRunRuntime.start_coverage_reporter()` daemon thread removed - (no longer called from `init()`). -- `_safe_bump_coverage` / `_bump_streaming_skipped` helpers removed - from `nullrun.instrumentation.auto`. -- `llm_call` wire shape: `metadata.tracked: bool` and - `metadata.streaming_skipped: bool` are now authoritative; the - separate `coverage_report` event is dropped. - -### Added - -- `nullrun.shutdown(timeout=2.0)`: sends a clean WebSocket close - frame and drains in-flight events. Long-running scripts that - exit via `sys.exit()` previously let the kernel RST the TCP - socket, which the backend logged as WARN "Connection reset - without closing handshake". Registering `nullrun.shutdown` in an - `atexit` handler eliminates the noisy log. No-op if `init()` - was never called. - -_Tests: 3 additions (tests/test_coverage_report.py, tests/test_coverage_seen_httpx.py, tests/test_llm_call_metadata_flags.py)._ - -## [0.8.3] - 2026-06-29 - -Additive patch on top of 0.8.2. Closes the same silent zero-billing -class of bug 0.8.2 closed on the httpx path — but on the **langgraph -callback path** and the **init-ordering hazard** that 0.8.2 didn't -reach. Promotes the missing-model wire failure from WARN to fail-LOUD. - -### Fixed - -- **langgraph callback model extraction.** `_extract_model_from_response` - now consults `response.llm_output` FIRST. langchain-openai 1.x puts - the date-suffixed model id (e.g. `gpt-4.1-mini-2025-04-14`) on - `LLMResult.llm_output`, while the AIMessage inside - `generations[0][0].message` leaves `response_metadata` empty. The - previous chain led with `response_metadata`, so every - OpenAI-via-LangChain 1.x call silently zero-billed. Also adds an - "any key containing model" sweep inside `llm_output` for non-OpenAI - wrappers (proxies, custom chat models). -- **Init-ordering hazard for `patch_httpx`.** The class-level - `__init__` wrap only catches Clients created AFTER it is installed. - Users that build `ChatOpenAI(...)` before `nullrun.init(api_key=...)` - end up with a pre-existing `httpx.Client` that the patch never sees. - `patch_httpx` now sweeps `gc.get_objects()` once at install and - wraps any pre-existing `Client`/`AsyncClient` whose transport isn't - already a `NullRun*Transport`. Idempotent via the existing - class-level marker. -- **Fail-LOUD missing-model wire tag.** `runtime.track()` now - escalates the missing-model warning from `logger.warning` to - `logger.error`, bumps a `dropped_llm_call_no_model` runtime counter - for dashboards, and tags the wire event with `__missing_model: True` - so the backend's `into_track_request` gate can reject with HTTP 422 - instead of silently recording a zero-cost call. The event is still - sent (not fail-CLOSED) so the backend can audit; the flag is - wire-private and stripped before persisting. Activated only for - `llm_call`; other event types are silent. -## [0.8.2] - 2026-06-29 - -Additive patch on top of 0.8.0. No public-API break. Continues the -0.8.0 wire-format audit with two regressions that were caught on -review and one contract test that pins the post-2026-06-27 backend -schema so a future rename can't silently break the SDK. - -### Fixed - -- **`track_coverage()` emits counter dicts under `event.metadata` - instead of the event top level.** Pre-fix the per-host `seen` / - `tracked` / `streaming_skipped` dicts sat at the event root, where - serde silently dropped them — `SdkTrackRequest` uses explicit - fields with no `#[serde(flatten)]` catchall, so unknown keys are - discarded. The dashboard's `last_coverage_pct` was permanently - `null` because every coverage report landed with empty - `seen`/`tracked`/`streaming_skipped` JSONB columns. Pin: - `tests/test_coverage_report.py::test_track_coverage_emits_wire_shape_with_metadata_nesting`. -- **Request-body model fallback in - `NullRunSyncTransport._emit`.** When the response body extractor - returns `None` for `model` (OpenAI Responses API, Anthropic - streaming edge cases), `_extract_model_from_request_body` reads - the model string the user embedded in the request body via - `ChatOpenAI(model="gpt-4.1-mini")`. Without this every such - call was zero-billed — backend `unwrap_or("default")` + - `DEFAULT_RATE` ≈ \$0/call. Unit-tested in - `tests/test_model_fallback.py`. - -_Tests: 1 additions (tests/test_batch_response_parsing.py)._ - -## [0.8.0] - 2026-06-28 - -SDK↔backend wire-format audit. Closes a class of silent-fail-OPEN -path that was sending `model=None` (or `model="unknown"`) on -`/track` for many LLM-vendor paths — every such event cost the -backend a `model_pricing` lookup that returned no row, fell -through to `DEFAULT_RATE` (~$30/M), and emitted a fallback warning -the operator couldn't reproduce because the offending observation -was buried in another package's telemetry. - -No public-API break. No behavior change for callers whose -instrumentation already populates `model` correctly. Pure wire- -payload hygiene. - -### Fixed - -- **`NullRunRuntime.track()` strips `None` values from the wire - payload.** Pre-0.8.0 the runtime forwarded every key in - `enriched` except those in `_WIRE_STRIP_FIELDS`, including keys - whose value was `None`. Putting `{"model": null}` on the wire - triggered backend `unwrap_or("default")` and a fallback warning. - Backend handles a missing key as well as `null`; dropping `None` - here keeps the diagnostic signal loud (the new - `WARN track(): llm_call event missing 'model' field` fires on - missing-key, which is what we want operators to see) instead of - silent (the JSON-null case). Activated only for `llm_call` so - `span_start` / `span_end` / `tool_call` traffic doesn't pollute - logs. - -- **All four instrumentation paths now extract `model` / - `provider` from the response object as a fallback, not just - from `invocation_params` / `self.model`.** When langchain 1.x - stopped forwarding `invocation_params` to `on_llm_end`, every - LangChain-callback track event carried `model="unknown"` and - the backend cost pipeline fell through to `DEFAULT_RATE`. The -## [0.7.8] - 2026-06-28 - -Additive patch on top of 0.7.7. Converts two silent fail-OPEN footguns -into explicit `DeprecationWarning` / `RuntimeError`. No behavior -change for callers who don't touch the deprecated surface. - -### Deprecated - -- `NullRunRuntime.start_recording()` and `NullRunRuntime.stop_recording()` now emit `DeprecationWarning`. They have been silent no-op stubs since Sprint 2.1 (0.4.0). [...] -- Setting `NULLRUN_USE_GRPC=1` now raises `RuntimeError` at SDK init instead of silently falling back to HTTP with an info log. gRPC transport remains on the roadmap but is not yet implemented. Unset the env var to use HTTP. See https://docs.nullrun.io/reference/sdk-api#transport - -### Migration - -- Replace `runtime.start_recording(workflow_id, metadata=...)` with a dashboard navigation or `nullrun.status()` introspection. -- Remove any `NULLRUN_USE_GRPC` env var from deployment configs (Docker compose, k8s manifests, systemd units). -- Catch `RuntimeError` at SDK init if you want to keep the env var as a feature flag — but the recommended path is to unset it. - - -## [0.7.7] - 2026-06-27 - -Additive patch on top of 0.7.6. Fixes the `/gate` pre-flight so the -backend can compute `projected_cost` and `tool_block` decisions from -real per-call data instead of the previous fake `"budget-precheck"` -sentinel and empty tool list. No breaking changes — new helpers -default to `None` / empty so existing call sites keep working. - -### Added - -- **`nullrun.set_call_context(model=..., tools=[...])`** — per-call - context the SDK forwards to `/gate` so the backend can enforce - budget tiers and tool-block on real values. - ```python - import nullrun - - with nullrun.workflow(name="support-bot"): - nullrun.set_call_context( - model="claude-sonnet-4-6", - tools=["shell.run", "code.eval"], - ) - - @nullrun.protect - def chat(message: str) -> str: - return agent.run(message) - ``` - - `model` (optional) — LLM model name. Backend uses it to look up - the per-model rate from `tool_pricing` (Postgres) so - `projected_cost` matches what `/track` will compute from real - token counts. Defaults to `None` (backend falls back to - `claude-sonnet-4` default rate). - - `tools` (optional) — list of tool names the call intends to use. - Backend matches each against the workflow's effective - `blocked_tools` aggregate and returns `block` on any match. - `None` leaves whatever was previously set; `[]` clears. -## [0.7.6] - 2026-06-27 - -Additive patch on top of the 0.7.0 thin-client refactor. Brings a -FastAPI integration, a default user-facing message catalog, and -small transport consistency fixes. No breaking changes. - -### Added - -- **`nullrun.integrations.fastapi`** — one-line FastAPI integration - that turns every `NullRunDecision` / `NullRunInfrastructureError` - thrown by `@nullrun.protect` endpoints into a clean JSON - response with the right HTTP status code. No per-endpoint - `except` blocks required. - ```python - from fastapi import FastAPI - import nullrun - from nullrun.integrations.fastapi import install - - nullrun.init(api_key="nr_live_...") - app = FastAPI() - install(app) - - @app.post("/chat") - @nullrun.protect - def chat(message: str) -> str: - return agent.run(message) - ``` - Response shape: - ```json - { - "error_code": "NR-B004", - "user_message": "You've reached the usage limit...", - "category": "decision" - } - ``` -## [0.7.0] - 2026-06-26 - -### BREAKING CHANGES - -SDK is now a thin client. All enforcement decisions arrive from the -backend via `/api/v1/gate` and `/api/v1/execute`. Local policy -enforcement, its dataclass, and its hardcoded thresholds are removed. - -**Removed:** - -- `class Policy`, `Policy.default_local()`, `Policy.strict_local()`, - `Policy.from_dict()` (was at `nullrun.runtime.Policy`) -- `NullRunRuntime.policy` property -- `NullRunRuntime(policy=...)` constructor kwarg -- `NullRunStatus.active_policy`, `.fallback_policy`, - `.fallback_reason`, `.last_policy_fetch`, - `.last_policy_fetch_age_seconds` fields -- `Transport.fetch_policy()` method -- `Transport.clear_policy_cache()` method -- `FallbackMode.CACHED` enum value (gate-decision fallback) -- Local loop/rate detectors: `LoopTracker`, `RateTracker`, - `LocalDecision` classes -- `NullRunRuntime._local_check()`, `_loop_tracker`, `_rate_tracker` - instance attrs -- `_local_loop_threshold`, `_local_rate_limit` instance attrs - (hardcoded 6/1000) -- `CachedDecision`, `PolicyCache` transport classes (tied to the - removed CACHED fallback mode) -- `NULLRUN_FALLBACK_MODE` env var -- `NULLRUN_POLICY_FAIL_OPEN` env var (no longer needed — backend is - authoritative) -- `NullRunRuntime._fetch_policy()` method (no local policy fetch on - init) -- WS `on_policy_invalidated` callback (no local policy to invalidate) - -## [0.6.1] — 2026-06-24 - -Additive release — Layers 1, 2, and 3 of the "give the user a chance" -design land together. Structured exceptions, a global error hook, -and a synchronous runtime snapshot. No breaking changes. - -### Layer 1 — structured exception hierarchy - -Every public SDK exception now carries a stable, grep-able -`error_code` (e.g. `NR-A001`, `NR-B002`, `NR-R001`) plus a short -imperative `user_action` and a `retryable` flag, so cookbook -examples and Sentry integrations can branch on the code instead -of parsing the message string. - -- **`NullRunError` — structured base for every user-facing SDK - exception.** Carries four actionable fields: - - `error_code` — stable `NR-LETTERNNN` identifier - (documented per-code in `docs/errors/.md`). - - `user_action` — short imperative next-step hint - ("Set NULLRUN_API_KEY", "Verify API key at …", "Retry in 30s - — backend is down", …). Empty when there is no actionable - step. - - `retryable` — `True` only for transient failures (5xx, - network blip, transient auth); `False` for config, - permission, and budget-exhausted (retrying without - changing something will just hit the same wall). - - `docs_url` — per-code docs page (falls back to the - `https://docs.nullrun.io/errors` index when the per-code - page does not exist yet). - - `cause` — optional chained `BaseException`. - -- **New specialized exception classes** (each is a subclass of - the existing user-facing class, so existing `except` clauses - keep matching): - -## [0.6.0] — 2026-06-23 - -Hardening pass driven by the 2026-06-22 SDK↔backend integration audit. -Closes three classes of silent fail-OPEN regressions that the previous -release shipped: SDK POSTs being rejected by the backend's CSRF -middleware, WS HMAC identity field drift, and policy-fetch silently -falling through to a permissive default on any backend blip. Coverage -jumped from ~76% to **84.59%** (branch = true). - -- **FIX-F3 — every signed POST now carries `Authorization: Bearer `.** - The backend's CSRF middleware (`backend/src/auth/csrf.rs::has_bearer_auth`) - bypasses the cookie-double-submit check whenever any non-empty - `Authorization` header is present. Pre-fix the SDK only sent - `X-API-Key`, so every POST hit the "state-changing request without - session cookie" branch and got 403 — which the SDK's `try/except` - around `/gate`, `/track`, `/check`, and `/execute` silently - swallowed. The net effect was that **every SDK-side enforcement - gate was effectively fail-OPEN on production traffic**. The fix - uses the user-facing `api_key` as the Bearer value so the bypass - header is meaningful for debugging; the canonical auth path is - still `X-API-Key` (+ HMAC when configured). Safe per - `csrf.rs:80-95` (browsers never auto-attach `Authorization` to - cross-site requests, so this is not a CSRF regression). - -- **FIX-F4 — WebSocket HMAC identity field pinned to `api_key`.** - Added `WS_HMAC_IDENTITY_FIELD = "api_key"` constant in - `transport_websocket.py` matching the backend's - `SignedWsMessage` struct (`backend/src/proxy/http/ws_control.rs:43`). - The SDK now reads `data["api_key"]` (with `data["api_key_id"]` as - a backwards-compat fallback for pre-FIX-F4 servers) to verify the - HMAC signature. Pre-fix a future server-side rename would silently - break WS signature verification with no compile-time signal. - -- **Policy fetch is now fail-CLOSED (F-R2-02).** Pre-fix, any HTTP - exception, non-200 status, or empty `{"data": []}` response silently -## [0.5.2] — 2026-06-19 - -This release bundles the Sprint 2.5 production-readiness hardening -alongside the Phase 0 contract / lifecycle fixes. The two streams were -shipped as separate `[Unreleased]` sections during development; they -are merged here into a single canonical entry so release tooling that -scans for the `[Unreleased]` anchor picks up the complete change set -exactly once. - -- **HMAC signing expanded (with documented exceptions, audit 2026-06-22 - round 2 — F-R2-05 / F-R2-14).** The SDK now signs every - outgoing POST/GET that the backend's `HMAC_REQUIRED_PATHS` allowlist - requires: `/track/batch`, `/gate`, `/check`, `/execute`. The - header set is built via `_add_hmac_headers` (Content-Type, - X-Signature, X-Signature-Timestamp, X-API-Key, Authorization for - CSRF bypass). Compliance with the canonical - `HMAC-SHA256(secret_key, "::")` - formula from `backend/src/auth/hmac.rs:6-9`. - - **Explicitly NOT signed (chicken-and-egg / backend allowlist):** - - `runtime._authenticate` → `POST /api/v1/auth/verify` on initial - bootstrap: no `secret_key` exists yet (it is what /auth/verify - hands back). The key-rotation refetch - (`Transport._refetch_credentials` at transport.py:1588) IS - signed because `secret_key` is then populated. - - `runtime._fetch_policy` → `GET /api/v1/orgs/{id}/policies`. - Not in `HMAC_REQUIRED_PATHS` (`backend/src/proxy/middleware/ - hmac_verify.rs:58`). Backend allowlist is authoritative. - - `runtime._fetch_remote_state` → `GET /api/v1/orgs/{id}/workflows/ - {wf}`. Not in `HMAC_REQUIRED_PATHS`. - - `runtime.get_org_status` → `GET /api/v1/orgs/{id}/status`. Not in - `HMAC_REQUIRED_PATHS`. - - **Outgoing WebSocket ACK is plain JSON, not signed.** Earlier - documentation overstated this — `transport_websocket._send_ack` -## [0.4.0] — 2026-06-17 - -Production-readiness release. Resolves all BLOCKER + HIGH + MEDIUM + LOW -audit findings from the 0.3.x audit. The curated 6-symbol public surface -(`init`, `protect`, `track_llm`, `track_tool`, `track_event`, -`__version__`) is unchanged. Full PR-by-PR description follows; this -entry is the summary. Phase-7 (framework patches) and Phase-8 -(release-prep polish) ship as follow-up releases under the same 0.4.x -line. - -- `BoundedDict` class (`runtime.py`) — dead since 0.3.1. -- `wrap_tool`, `wrap`, `check_before_tool`, `enforce_check_before_llm`, - `check_before_llm` (and the `CheckDecision` dataclass), `evaluate` - (`runtime.py`) — zero in-tree callers; `wrap` had a latent - `NameError` that's gone with the deletion. -- `clear_pause` (`actions.py`) — zero callers. -- `WorkflowContext` class (`context.py`) — duplicate of the - `workflow()` contextmanager. -- `WebSocketManager` (`transport_websocket.py`) — never instantiated; - the runtime uses `WebSocketConnection` directly. -- `PoolConfig` + `AdaptivePool` (`transport.py`) — never instantiated; - `httpx.Limits` is the real pool. -- `Transport._atexit_flush` (`transport.py`) — orphan method from the - pre-weakref.finalize migration. -- `EventRecorder` (`decision_history.py`) — never used. - -- **First-`track()` `AttributeError` (Phase 2).** `runtime.track()` no - longer reads `self._workflow_costs` (a BoundedDict removed in 0.3.1 - whose two callers survived). Returns `local_cost_cents = 0` from - the new `_local_cost_cents_estimate` attribute. -- **`auto_requests` module was unimportable.** The missing - `_safe_bump_coverage` helper that `auto_requests.py` imports is - now defined in `auto.py`. The whole module imports cleanly and the - coverage dashboard counter is reachable. -- **`auto_instrument()` now calls `patch_requests`.** The `requests` -## [0.3.1] — 2026-06-17 - -Production-readiness hardening. No public-API changes; the curated 6-symbol -surface is unchanged. Aligns the SDK with the contracts in -`NULLRUN/docs/adr/008-sdk-preflight-fail-policy.md` and -`NULLRUN/docs/kill-contract.md`. - -- **gRPC transport code path removed.** `create_grpc_transport` was - referenced but never defined, so setting `NULLRUN_USE_GRPC=1` raised - `NameError` at init. The gRPC server at the platform is intentionally - frozen until the activation checklist (TLS, auth, proto extensions, - cost pipeline parity, tests) is complete. The SDK now logs an - INFO line on `NULLRUN_USE_GRPC=1` and silently falls back to - HTTP. The `grpcio` hard dependency has been dropped from - `pyproject.toml`. If/when gRPC is unblocked, the SDK will add it back - as a separate optional extra. -- **`InsecureTransportError` URL check hardened.** Replaced the - `startswith("http://127.0.0.1")` chain with a `urllib.parse.urlparse` - + `ipaddress.ip_address` check. The previous check let - `http://127.0.0.1.attacker.com` and `http://localhost.evil.com` - through (homograph attacks) and rejected `http://[::1]:8080` - (IPv6 loopback). The new check allows the full `127.0.0.0/8` - IPv4 loopback range, `::1`, and `localhost` (case-insensitive). -- **`signal.signal` global hijack removed.** `Transport.__init__` no - longer installs a process-wide `SIGTERM` / `SIGINT` handler - that called `sys.exit(0)` from inside the signal context. - The fix contract was already pinned in `tests/test_signal_safety.py` - and is now applied to the source. -- **`atexit.register` replaced with `weakref.finalize`.** The - per-Transport `atexit` chain was growing without bound in - long-running deployments; weakref finalizers only fire if the - transport is still alive at process exit. -- **`Transport` is now a context manager.** `with Transport(...) as t:` - starts the flush thread on enter and stops it on exit. Replaces - the manual `start() / stop()` pair that was easy to forget. -## [0.3.0] — 2026-06-15 - -### Breaking - -- **No-api-key init now raises** (T3-S2): `nullrun.init()` and - `NullRunRuntime(...)` without an `api_key` (and with `NULLRUN_API_KEY` - unset) now raise `NullRunAuthenticationError` instead of falling back - to a `NullRunNoop` stub. The previous silent fallback silently - bypassed every backend gate (budget, policy, control plane) — a real - safety hole in production. **Action required:** ensure - `api_key="nr_live_..."` is passed to `init()` (or `NULLRUN_API_KEY` - is set) in every entry point. The `0.2.0` deprecation warning has - been removed; the new behavior is hard. -- **`local_mode` field removed**: The auto-derived `local_mode` flag - on `NullRunRuntime` is gone. The `is_local_mode` property and the - `NullRunNoop` / `NullRunNoopBreaker` / `_NullContext` classes are - deleted (`nullrun.noop` module removed). All call sites that read - `runtime.local_mode` will see `AttributeError` — there is no - migration path because the field no longer has meaning. Code paths - that previously branched on `local_mode` now always go through the - cloud runtime (auth + policy fetch + control plane). - -### Removed - -- **Legacy Breaker exports** (T9): The 7 legacy re-exports - (`nullrun.BreakerError`, `nullrun.CostLimitExceeded`, - `nullrun.ApprovalRequired`, `nullrun.BreakerTimeout`, - `nullrun.Policy`, `nullrun.FallbackMode`, `nullrun.PoolConfig`) - are no longer reachable as `from nullrun import X`. The canonical - exception names (`NullRunBlockedException`, `WorkflowPausedException`, - `WorkflowKilledException`, `NullRunAuthenticationError`, …) and the - canonical policy/transport modules - (`from nullrun.runtime import Policy`, - `from nullrun.transport import FallbackMode, PoolConfig`) remain - available. Audited for 0 external callers. -## [0.1.1] — 2026-05-20 - -### Fixed - -- **CR-2**: Fixed buffer overflow when circuit breaker is OPEN. Previously, re-queued events were prepended to buffer, causing newest events to be dropped first. [...] -- **CR-5**: Async circuit breaker now uses `asyncio.Lock` instead of `threading.Lock` for proper async context handling. -- **CR-1+CR-4**: `runtime.py` now creates Transport before `_authenticate()` and `_fetch_policy()`, reusing the HTTP client for connection pooling and consistent timeout/retry poli [...] -- **AsyncAwait**: Fixed `_call_async()` not awaiting `_on_success_async()` and `_on_failure_async()` coroutines, causing "coroutine was never awaited" warnings in async transport. - -### Changed - -- Transport buffer now enforces max_buffer_size **before** re-queuing events on circuit breaker OPEN - - -## [0.1.0] — 2026-05-18 - -### Added - -- Circuit breaker core (`src/nullrun/breaker/`) with STRICT / PERMISSIVE / CACHED fallback modes -- HTTP transport with batch event sending (`transport.py`) -- Async transport for asyncio applications -- Retry logic with jitter and policy-aware backoff -- `@protect` decorator for wrapping functions (`decorators.py`) -- Workflow context support (`context.py`) -- Main runtime entrypoint (`runtime.py`) -- `X-API-Version` header on all outgoing requests - -- Requires Python ≥ 3.10 -- Compatible with NullRun API version `2024-01-15` +- Lifecycle: `init`, `protect`, `shutdown`, `on_error`, `status`, + `handle`, `guarded`, `init_or_die` +- Messages: `format_user_message`, `set_user_message` +- Exceptions: `NullRunError` and the typed catalog + (`NullRunAuthError`, `NullRunBackendError`, `NullRunBudgetError`, + `NullRunConfigError`, `NullRunMcpApprovalRequiredError`, + `NullRunMcpDestructiveBlockedError`, + `NullRunMcpReadonlyBypassBlockedError`, `NullRunToolBlockedError`, + `NullRunWorkflowKilledError`, `WorkflowKilledInterrupt`) +Wire contract: unchanged from 0.18.0. diff --git a/docs/errors/NR-B003.md b/docs/errors/NR-B003.md deleted file mode 100644 index 289bec8..0000000 --- a/docs/errors/NR-B003.md +++ /dev/null @@ -1,66 +0,0 @@ -# NR-B003 — Sensitive-tool impact extractor failed - -| Field | Value | -|---|---| -| **Code** | `NR-B003` | -| **Category** | Backend (tool extraction) | -| **Exception class** | `NullRunBlockedException` | -| **Retryable** | No | -| **Default `user_action`** | "The @sensitive decorator could not extract a business_impact envelope for this tool. Pass an explicit `impact=` argument to `@sensitive(...)` (see ToolParameters rules) or add a `ToolParamsExtractor` to the function. The default `include_all=True` extractor failed because the function signature is not introspectable (e.g. wrapped in C code or a non-Python callable)." | - -## When - -Raised when `@sensitive` cannot derive a `BusinessImpact` envelope for -the wrapped function. The decorator runs four extractors in priority -order: - -1. Explicit `impact=` argument to `@sensitive(...)`. -2. Pre-registered `ToolParamsExtractor` on the function. -3. Per-function fallback (`include_all=True`). -4. Constant-extractor fallback (legacy). - -If all four fail, the decorator blocks the call (fail-CLOSED) with -`NR-B003` so a missing impact never silently widens to a different -policy. - -## Common causes - -- **Wrapped in a non-introspectable callable** — `functools.partial`, - a `ctypes` function, or anything that hides its signature. -- **Decorator chain obscures the function** — a third-party decorator - replaced `__wrapped__` with something that lacks `__signature__`. -- **Custom `ToolParamsExtractor` raised** — your extractor has a bug; - the chain fails fast rather than passing the call through. - -## How to fix - -1. **Prefer the explicit form**: `@sensitive(impact=tool_params({...}))` - for static schemas, or `@sensitive(impact=BusinessImpact.tool_call(...))` - for runtime-built envelopes. -2. **Wrap before `@sensitive`** — place `@sensitive` as the OUTERMOST - decorator so it sees the un-wrapped signature. -3. **Provide a custom extractor** — `ToolParamsExtractor` with a - `params_for(func, args, kwargs)` method that returns a fixed - `ToolCallParams`. - -## Catch pattern - -```python -from nullrun.breaker.exceptions import NullRunBlockedException - -try: - @sensitive - def my_op(x): - return do_it(x) -except NullRunBlockedException as exc: - if exc.error_code == "NR-B003": - # Surface the misconfiguration early — at decorator application - # time, not on the first call. - log.error("sensitive missing impact: %s", exc.user_action) -``` - -## Related codes - -- `NR-B001` — network error during transport. -- `NR-B002` — 5xx from the NullRun backend. -- `NR-W002` — workflow killed. diff --git a/docs/errors/README.md b/docs/errors/README.md index 46a4f6d..58b9e92 100644 --- a/docs/errors/README.md +++ b/docs/errors/README.md @@ -44,7 +44,6 @@ The codes follow a `NR-` pattern: |---|---|---| | `NR-B001` | Network error: timeout, ConnectError, DNS failure | [NR-B001](NR-B001.md) | | `NR-B002` | 5xx from the NullRun backend | [NR-B002](NR-B002.md) | -| `NR-B003` | `@sensitive` failed to extract a `BusinessImpact` envelope | [NR-B003](NR-B003.md) | | `NR-B004` | Budget exhausted | [NR-B004](NR-B004.md) | | `NR-B005` | Local circuit breaker tripped | [NR-B005](NR-B005.md) | diff --git a/pyproject.toml b/pyproject.toml index f7b9694..96894bc 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -6,7 +6,7 @@ build-backend = "hatchling.build" name = "nullrun" # Full release history lives in CHANGELOG.md; only the current version # is pinned here. -version = "0.18.1" +version = "0.18.2" # Kept under the 200-char preview threshold so the full line is visible # without an "expand" click. The headline is the canonical §1 statement # from positioning.md — "runtime decision layer for tool-using AI agents" @@ -463,22 +463,16 @@ line-length = 100 select = ["E", "F", "I", "UP", "B", "S"] ignore = [ "S101", - # Pre-existing violations in master / wip/working-tree: tracked - # for follow-up cleanup in a dedicated PR rather than blocking CI. - # Categories: - # S110 (try/except/pass) - 14 sites; needs logging, not blanket - # noqa - # E501 (line too long) - 13 sites; long descriptive comments - # F841 (unused variable) - 6 sites; one is the timestamp var - # in the legacy-code fallback path - # E402 (import order) - 5 sites; TYPE_CHECKING blocks - # F401 (unused import) - 2 sites + # Pre-existing violations in master. Tracked for follow-up cleanup + # in a dedicated PR rather than blocking CI. Categories: + # S110 (try/except/pass) - blanket noqa would hide bugs; needs + # logging, not blanket noqa + # E501 (line too long) - long descriptive comments + # E402 (import order) - TYPE_CHECKING blocks "S110", "E501", - "F841", "E402", - "F401", - # S311 (suspicious random) - 1 site, in circuit_breaker jitter. + # S311 (suspicious random) in circuit_breaker jitter. # random.uniform is correct for jitter (we want non-cryptographic # randomness to spread reconnection timing across workers). "S311", diff --git a/scripts/smoke_prod.py b/scripts/smoke_prod.py new file mode 100644 index 0000000..4d51fbc --- /dev/null +++ b/scripts/smoke_prod.py @@ -0,0 +1,169 @@ +"""Smoke-test the SDK against the live production backend. + +Verifies: + * @protect is the single user-facing entry point (zero-arg) + * /execute wire shape is unchanged + * Round-trip succeeds end-to-end + +Run with:: + + NULLRUN_API_KEY=nr_live_... \ + NULLRUN_API_URL=https://api.nullrun.io \ + python scripts/smoke_prod.py + +Never echo the API key. Never commit it. The script reads it +from the environment only. +""" +from __future__ import annotations + +import json +import os +import sys +import time +import traceback +from typing import Any + + +def _fail(msg: str) -> None: + print(f"FAIL {msg}") + sys.exit(1) + + +def _ok(msg: str) -> None: + print(f"PASS {msg}") + + +def main() -> None: + api_key = os.environ.get("NULLRUN_API_KEY") + api_url = os.environ.get("NULLRUN_API_URL") + if not api_key: + _fail("NULLRUN_API_KEY not set in environment") + if not api_url: + _fail("NULLRUN_API_URL not set in environment") + + # Force UTF-8 (Windows console). + try: + sys.stdout.reconfigure(encoding="utf-8") + sys.stderr.reconfigure(encoding="utf-8") + except Exception: + pass + + # Late imports so env vars are read by the SDK first. + import nullrun + from nullrun import init, protect, shutdown, on_error, status + from nullrun.breaker.exceptions import NullRunBlockedException + + # 1. Surface check — only the curated entry points are exposed. + exposed = set(dir(nullrun)) + forbidden = { + "sensitive", "money_outflow", "tool_params", + "track_event", "track_tool", "track_llm", + } + leaked = exposed & forbidden + if leaked: + _fail(f"forbidden exports leaked at top level: {sorted(leaked)}") + _ok(f"surface is clean (no leaked: {sorted(forbidden)})") + + # 2. init() round-trip. + try: + init(api_key=api_key, api_url=api_url) + except Exception as exc: # noqa: BLE001 + _fail(f"init() raised: {exc!r}\n{traceback.format_exc()}") + _ok(f"init() ok against {api_url}") + + # 3. status() after init. + try: + st = status() + except Exception as exc: # noqa: BLE001 + _fail(f"status() raised: {exc!r}") + # `status()` returns a typed NullRunStatus object; coerce via + # ``vars()`` so we can introspect fields without depending on + # the SDK's internal type name. + fields = vars(st) if hasattr(st, "__dict__") else dict(st or {}) + if not fields: + _fail(f"status() returned empty: {type(st).__name__}") + _ok(f"status() returned: type={type(st).__name__} fields={sorted(fields.keys())}") + + # 4. Universal @protect — no parameters, no extras. + @protect + def smoke_probe(name: str) -> str: + return f"hello {name}" + + # 5. Call through @protect. This round-trips /execute. + # On allow → function return value is forwarded. + # On block → typed NullRunBlockedException with wire details. + decision_payload: dict | None = None + raised: BaseException | None = None + function_return: Any = None + t0 = time.perf_counter() + try: + function_return = smoke_probe("prod-smoke") + except NullRunBlockedException as exc: + raised = exc + except Exception as exc: # noqa: BLE001 + raised = exc + elapsed_ms = (time.perf_counter() - t0) * 1000 + + if raised is None: + # Allow path: function ran, return value is whatever the + # wrapped function returned. We don't assert wire shape + # here (the wire shape is verified by the v3 contract tests + # + this smoke against the BLOCK path below). + if function_return != "hello prod-smoke": + _fail( + f"allow-path returned wrong value: {function_return!r}" + ) + _ok(f"@protect → /execute ALLOW in {elapsed_ms:.0f}ms; fn ran") + else: + # Block path: confirm typed exception + wire details. + if not hasattr(raised, "error_code"): + _fail(f"raised exception has no error_code attr: {type(raised).__name__}") + wire = getattr(raised, "details", {}) or {} + _ok( + f"@protect → /execute BLOCKED in {elapsed_ms:.0f}ms " + f"class={type(raised).__name__} code={getattr(raised, 'error_code', '?')!r}" + ) + print(" wire:", json.dumps(wire, default=str)[:500]) + decision_payload = wire # for any downstream tooling + + # 5b. Force a known-block path via an obviously-bad tool_name. + # This verifies wire-shape on the BLOCK branch (which is + # what the v3 contract tests mock anyway). + @protect + def definitely_blocked(name: str) -> str: + return f"would-not-run {name}" + + blocked_exc: BaseException | None = None + try: + definitely_blocked("__smoke_force_block__") + except BaseException as exc: # noqa: BLE001 + blocked_exc = exc + if blocked_exc is None: + _ok("(no second block observed; policy may have allowed — skipping wire shape check)") + else: + wire = getattr(blocked_exc, "details", {}) or {} + for required_key in ("decision_source",): + if required_key not in wire: + _fail( + f"block wire shape missing {required_key!r}: " + f"keys={sorted(wire.keys())}" + ) + _ok( + f"block path wire shape: decision_source={wire.get('decision_source')!r}" + ) + + # 6. shutdown() — flushes batched events. + try: + shutdown() + except Exception as exc: # noqa: BLE001 + # shutdown on a never-executed batch is allowed to no-op. + _ok(f"shutdown() raised (non-fatal): {exc!r}") + else: + _ok("shutdown() ok") + + print() + print("ALL OK") + + +if __name__ == "__main__": + main() diff --git a/src/nullrun/__init__.py b/src/nullrun/__init__.py index 07ef0ea..16027de 100644 --- a/src/nullrun/__init__.py +++ b/src/nullrun/__init__.py @@ -1,10 +1,19 @@ """ NullRun Platform SDK. -Enforcement gateway client for AI agents. Curated 6-symbol surface: -`init`, `protect`, `track_llm`, `track_tool`, `track_event`. Everything -else is reachable on demand via `from nullrun import X` but does NOT -appear in `dir(nullrun)`. +The user-facing entry point is ``@protect`` — the universal gate +decorator that wraps any function (tool, LLM call, business logic) +and routes the call through the backend's policy / budget / +approval pipeline. No parameters, no separate decorators, no +extra entry points: if you want NullRun to see the call, decorate +it with ``@protect``. + +Everything else exposed by ``nullrun`` is either runtime lifecycle +(``init``, ``shutdown``, ``on_error``, ``status``), the structured +exception hierarchy, or message/error-handling helpers +(``format_user_message``, ``handle``, ``guarded``, ``init_or_die``). +None of those are alternatives to ``@protect`` — they're setup +and cleanup. Usage: import nullrun @@ -14,8 +23,19 @@ def my_agent(query): return call_llm(query) -See README.md for LangGraph, OpenAI Agents, llama-index, crewai, autogen -auto-instrumentation; CHANGELOG.md for breaking changes between versions. + @nullrun.protect + def charge_customer(amount_cents: int, customer_id: str): + return refund_wire(amount_cents, customer_id) + +The backend reads live values out of the call's kwargs and applies +whatever rules / budgets / approval gates the operator has +configured on the dashboard. The SDK does not know — and does not +need to know — whether a call is a tool call, an LLM call, a +budget operation, or anything else. That's the entire point. + +See README.md for LangGraph, OpenAI Agents, llama-index, crewai, +autogen auto-instrumentation; CHANGELOG.md for breaking changes +between versions. """ from __future__ import annotations @@ -37,8 +57,7 @@ def my_agent(query): # in tab-completion. All other names (legacy Breaker exports, # instrumentation, exceptions, …) live in `_LAZY_EXPORTS` below and are # loaded on first access via __getattr__. -from nullrun.decorators import protect # the gate decorator -from nullrun.runtime import track_event, track_llm, track_tool +from nullrun.decorators import protect # the gate decorator — universal entry point def shutdown(timeout: float = 2.0, flush: bool = True) -> None: @@ -240,7 +259,6 @@ def my_agent: if debug: logger.setLevel(logging.DEBUG) - # T3-S2 (0.3.0): api_key is now required. Previous versions fell back # to a NullRunNoop stub in `local_mode`, which silently bypassed every # backend gate (budget, policy, control plane). That was a real # safety hole — production callers were unaware their policies were @@ -269,15 +287,13 @@ def my_agent: "nullrun.init() requires an api_key. Pass api_key='nr_live_...' " "explicitly or set the NULLRUN_API_KEY environment variable. " "Whitespace-only values are rejected — strip surrounding spaces " - "before passing or exporting the key. " - "(Silent no-op fallback was removed in 0.3.0 — see CHANGELOG.)", + "before passing or exporting the key.", error_code="NR-C001", user_action=( "Get an API key at https://app.nullrun.io/settings/api-keys, " "then either pass api_key='nr_live_...' to nullrun.init() or " "set the NULLRUN_API_KEY environment variable. The SDK cannot " - "operate without credentials — the silent no-op fallback was " - "removed in 0.3.0 because it bypassed every backend gate." + "operate without credentials." ), ) # Layer 2: fire the on_error hook BEFORE the raise so the @@ -297,10 +313,6 @@ def my_agent: # Imported lazily so we don't pull the runtime into the namespace # when the user only wants the static helpers. - import threading as _threading - - import nullrun.decorators as _dec_mod - import nullrun.runtime as _rt_mod from nullrun.runtime import NullRunRuntime # C3 fix: shut down any existing runtime before constructing a new @@ -388,11 +400,6 @@ def my_agent: auto_instrument(runtime) - # 0.9.0: coverage reporter removed. Coverage is now derived - # server-side from llm_call span metadata (host + tracked + - # streaming_skipped flags). No 60s daemon thread, no per-process - # counter dicts. - return runtime @@ -409,7 +416,6 @@ def my_agent: "NullRunRuntime": ("nullrun.runtime", "NullRunRuntime"), "get_runtime": ("nullrun.runtime", "get_runtime"), "get_protected_runtime": ("nullrun.decorators", "get_protected_runtime"), - "track": ("nullrun.runtime", "track"), "reset": ("nullrun.decorators", "reset"), "workflow": ("nullrun.context", "workflow"), "span": ("nullrun.context", "span"), @@ -426,9 +432,8 @@ def my_agent: "set_call_context": ("nullrun.context", "set_call_context"), "get_call_model": ("nullrun.context", "get_call_model"), "get_call_tools": ("nullrun.context", "get_call_tools"), - # 2026-07-02 (v0.11.0): chain context for soft-mode budget gate - #. ``chain`` is the contextmanager - # ``get_chain_id`` / ``set_chain_id`` are the manual setters. + # ``chain`` contextmanager + manual ``get_chain_id`` / + # ``set_chain_id`` setters — soft-mode budget gate scoping. "chain": ("nullrun.context", "chain"), "get_chain_id": ("nullrun.context", "get_chain_id"), "set_chain_id": ("nullrun.context", "set_chain_id"), @@ -436,20 +441,9 @@ def my_agent: "set_chain_op": ("nullrun.context", "set_chain_op"), # Instrumentation "NullRunCallback": ("nullrun.instrumentation", "NullRunCallback"), - # NOTE: `patch_openai` and `unpatch_openai` were removed from - # `_LAZY_EXPORTS` because they pointed at non-existent - # attributes on `nullrun.instrumentation` (the actual function - # is `patch_openai_agents`, with different semantics — it patches - # `agents.Runner`, not the `openai` SDK). The pre-fix lazy - # entries caused `AttributeError` on first access, which is a - # worse failure mode than a clean `ImportError` from - # `from nullrun import patch_openai` failing because the symbol - # is no longer in the lazy table. # Toolbox — framework-specific wrappers. The previous `instrument ` # helper lived at `nullrun.instrumentation.langgraph.instrument`; # it is now `nullrun.toolbox.langgraph.wrapper`. Reachable as - # `from nullrun import wrapper` for one-line import. - "wrapper": ("nullrun.toolbox.langgraph", "wrapper"), # Span / trace context. `tracing.py` is the structured replacement # for the loose `_trace_id` / `_span_id` contextvars in # `nullrun.context`. `SpanContext` is a single value (parent + @@ -462,32 +456,6 @@ def my_agent: "create_child_span": ("nullrun.tracing", "create_child_span"), "set_span": ("nullrun.tracing", "set_span"), "reset_span": ("nullrun.tracing", "reset_span"), - # Decorators - "sensitive": ("nullrun.decorators", "sensitive"), - # Sensitive impact extractors. The documented decorator pattern - # `@nullrun.sensitive(impact=money_outflow(...))` lives in - # `decorators.py:1113-1132` and `extractor.py:43`. Both helpers - # live in `nullrun.extractor` but are not re-exported at the - # top level — `__getattr__` masks any name not in this table, so - # `nullrun.money_outflow(...)` previously raised AttributeError - # on the first invocation of the documented pattern. Adding the - # entries here matches the `NullRunApprovalDbUnavailableError` - # lazy-export pattern (see line 514). Workaround - # `from nullrun.extractor import money_outflow` still works. - "money_outflow": ("nullrun.extractor", "money_outflow"), - "tool_params": ("nullrun.extractor", "tool_params"), - # Business impact module re-export. The docstrings at - # `extractor.py:18` and `extractor.py:799-801` reference - # `nullrun.business_impact.compute_action_digest` / - # `MoneyImpactExtractor` / `ToolParamsExtractor` as bare dotted - # paths. The module is real (`nullrun/business_impact.py`) and - # contains those symbols, but PEP 562 `__getattr__` masks - # submodule access unless we expose the module object itself. - # The `attr_name=None` sentinel below tells `__getattr__` to - # return the imported submodule verbatim rather than `getattr` - # on it — same shape as `from nullrun import business_impact` - # for the user, no manual `import nullrun.business_impact` first. - "business_impact": ("nullrun.business_impact", None), # Actions "ActionHandler": ("nullrun.actions", "ActionHandler"), "ActionType": ("nullrun.actions", "ActionType"), @@ -514,17 +482,14 @@ def my_agent: # Zombie exception classes removed. See the NOTE block in # breaker/exceptions.py for the list. "WorkflowPausedException": ("nullrun.breaker.exceptions", "WorkflowPausedException"), - "WorkflowKilledException": ("nullrun.breaker.exceptions", "WorkflowKilledException"), "WorkflowKilledInterrupt": ("nullrun.breaker.exceptions", "WorkflowKilledInterrupt"), # Sibling typed name for the kill signal. Discovered via # ``from nullrun import NullRunWorkflowKilledError``; matches - # `WorkflowKilledInterrupt` (BaseException) and the older - # `WorkflowKilledException` for back-compat. Cookbook code + # `WorkflowKilledInterrupt`. Cookbook code # that wants a typed ``except`` clause prefers this over the # base-interrupt form (mro-aware dispatch). The class lives at # breaker/exceptions.py:1459. "NullRunWorkflowKilledError": ("nullrun.breaker.exceptions", "NullRunWorkflowKilledError"), - # ── B.1 (2026-09-10): MCP umbrella + APPROVAL_DB symmetry. # Four typed exception classes that round-trip the MCP umbrella # codes (ADR-013, frozen-dormant) and the six APPROVAL_DB_* # sibling codes (DEF-ARFLOW-TOOLNAME-01). Pre-B.1 these all @@ -580,24 +545,9 @@ def __getattr__(name: str): """PEP 562 — lazy attribute access for backward-compatible symbols.""" if name in _LAZY_EXPORTS: module_path, attr_name = _LAZY_EXPORTS[name] - # ``attr_name`` is str | None: the sentinel ``None`` means - # "return the submodule itself" (see business_impact - # re-export at line 490). ``__import__`` with ``fromlist=[]`` - # returns the top-level package, which is what we want in - # both cases. - fromlist: list[str] = [attr_name] if attr_name is not None else [] + fromlist: list[str] = [attr_name] module = __import__(module_path, fromlist=fromlist) - if attr_name is None: - # Sentinel: return the imported module itself (submodule - # re-export). Used for `nullrun.business_impact` so the - # docstring-referenced dotted paths - # (`nullrun.business_impact.compute_action_digest` etc.) - # resolve without an explicit `import - # nullrun.business_impact` first. See - # `_LAZY_EXPORTS['business_impact']` for the rationale. - value = module - else: - value = getattr(module, attr_name) + value = getattr(module, attr_name) # Cache on the module so subsequent lookups are O(1) and # dir(nullrun) still reports the curated public surface until # the legacy name is actually accessed. @@ -622,16 +572,12 @@ def __dir__() -> list[str]: __all__ = [ # Version (single value, always public) "__version__", - # The curated public surface — six symbols. Everything else - # stays importable as `from nullrun import X` for backward - # compatibility, but does NOT appear in `dir(nullrun)` until the - # user actually accesses it. + # The curated public surface. ``protect`` is the universal gate — + # every function call that the SDK should see goes through + # ``@protect``. The rest is runtime lifecycle, structured + # exceptions, or error-handling helpers. Nothing else. "init", - "protect", # gate decorator - "track_llm", - "track_tool", - "track_event", - # Audit 2026-06-29 (WS graceful close on exit): the user-facing + "protect", # gate decorator — the only user-facing entry point # top-level ``shutdown `` sends a clean WS close frame and # drains in-flight events. Without it, a long-running script # that exits via ``sys.exit `` lets the kernel RST the TCP @@ -652,11 +598,7 @@ def __dir__() -> list[str]: # ``__all__`` means ``from nullrun import *`` and ``dir(nullrun)`` # surface them for tab-completion — the whole point of giving # the user "a chance" is that they need to know the names exist - # to catch them. The legacy types (``NullRunBlockedException`` - # ``NullRunAuthenticationError``, ``WorkflowKilledException`` - # ``WorkflowPausedException``) stay importable via - # ``_LAZY_EXPORTS`` for back-compat — adding them here would - # change ``dir(nullrun)`` for existing users. + # to catch them. "NullRunError", "NullRunAuthError", "NullRunConfigError", @@ -665,10 +607,6 @@ def __dir__() -> list[str]: "NullRunToolBlockedError", "WorkflowKilledInterrupt", "NullRunWorkflowKilledError", - # B.1 (2026-09-10): MCP umbrella + APPROVAL_DB symmetry. The - # four typed exception classes are part of the curated public - # surface — cookbook code branches on them by name, so they - # need to be visible in ``dir(nullrun)`` for tab-completion. "NullRunMcpDestructiveBlockedError", "NullRunMcpReadonlyBypassBlockedError", "NullRunMcpApprovalRequiredError", @@ -696,9 +634,5 @@ def __dir__() -> list[str]: # The SDK-side ``decision_history`` module was deleted. Decision # history is a backend + dashboard surface only — the SDK does not # (and cannot) replay LLM calls because NULLRUN does not store -# request/response payloads or hold client LLM keys. The orphan -# ``start_recording`` / ``stop_recording`` methods on -# ``NullRunRuntime`` are kept as no-op stubs for one minor version -# for backward compatibility; they will be removed in 0.5.0. -# Do NOT re-export ReplayManager / ReplaySession / ReplayEvent / -# EventRecorder. +# request/response payloads or hold client LLM keys. Do NOT re-export +# ReplayManager / ReplaySession / ReplayEvent / EventRecorder. diff --git a/src/nullrun/__version__.py b/src/nullrun/__version__.py index 1494276..f844367 100644 --- a/src/nullrun/__version__.py +++ b/src/nullrun/__version__.py @@ -5,5 +5,5 @@ string and the SDK_MIN_VERSION constant. """ -__version__ = "0.18.1" +__version__ = "0.18.2" __platform_version__ = "1.0.0" diff --git a/src/nullrun/_handle.py b/src/nullrun/_handle.py index 970089c..06f800d 100644 --- a/src/nullrun/_handle.py +++ b/src/nullrun/_handle.py @@ -95,7 +95,6 @@ def _render_dev_error_report( The catalog ``format_user_message`` wording is included as the headline so end-user scripts that just want one sentence still - get a sensible line. We do NOT prefix the report with the catalog text -- the headline IS the catalog text, then the structured detail follows on its own line. @@ -119,7 +118,6 @@ def _render_dev_error_report( # 1. WHAT -- the stage that failed. Prefer the explicit ``endpoint`` # attribute (set on transport errors); fall back to deriving from # the class name so an unmapped exception still gives a sensible - # label. The class-name fallback strips the ``NullRun`` prefix and # ``Error`` suffix so ``NullRunAuthenticationError`` -> "auth". stage = endpoint or type(exc).__name__.replace("NullRun", "").replace("Error", "") stage = stage.lower() or "unknown" @@ -217,7 +215,6 @@ def handle(*, exit_code: int = 1): try: yield except NullRunError as exc: - # 2026-09-08 migration: WorkflowKilledInterrupt moved onto # the NullRunError MRO so Sentry/OTel `except Exception` # handlers record kill events. ``handle``/``guarded`` are the # friendly-exit pattern, NOT the user-callback pattern -- kill @@ -322,7 +319,6 @@ def my_agent(prompt): return init(api_key=api_key, api_url=api_url, debug=debug) except NullRunError as exc: # Same structured report as ``handle()`` / ``guarded`` -- a - # missing API key at startup was previously printed as just # "There's a configuration issue. Please contact support." # which gave the developer zero actionable detail. The # four-line report here names the missing env var, the URL diff --git a/src/nullrun/actions.py b/src/nullrun/actions.py index 6fd7c44..583b107 100644 --- a/src/nullrun/actions.py +++ b/src/nullrun/actions.py @@ -23,7 +23,6 @@ from nullrun.breaker.exceptions import ( NullRunBlockedException, NullRunWorkflowKilledError, - WorkflowKilledInterrupt, WorkflowPausedException, ) @@ -213,7 +212,7 @@ def handle( "running. Investigate ASAP." ) self._record_action( - ActionType.BLOCK, # record what would have happened pre-fix + ActionType.BLOCK, workflow_id, f"unknown_action_type:{action}", details, @@ -235,7 +234,6 @@ def handle( # Don't let handler exceptions propagate. We catch # `BaseException` (not just `Exception`) because # kill signals (NullRunWorkflowKilledError, the - # 2026-09-08-migrated Exception subclass) and any # third-party kill-shaped signals must be recorded # in history (already done above) and swallowed, # NOT re-raised into the caller's frame. @@ -387,7 +385,6 @@ def _deliver_webhook(self, webhook: WebhookConfig, payload: dict[str, Any]) -> N logger.warning("httpx not installed, cannot send webhook") return - # P3-2: exponential backoff between attempts with a # 30s cap. Pre-fix the schedule was linear (``0.5 * (attempt+1)`` # → 0.5s, 1.0s, 1.5s,...). Linear doesn't back off fast enough # when the destination is down — a transient outage produced diff --git a/src/nullrun/audit.py b/src/nullrun/audit.py index 4a91731..bc9640f 100644 --- a/src/nullrun/audit.py +++ b/src/nullrun/audit.py @@ -33,7 +33,7 @@ from __future__ import annotations -from dataclasses import dataclass, field +from dataclasses import dataclass from datetime import datetime from typing import Any diff --git a/src/nullrun/breaker/circuit_breaker.py b/src/nullrun/breaker/circuit_breaker.py index 234218c..d0195f0 100644 --- a/src/nullrun/breaker/circuit_breaker.py +++ b/src/nullrun/breaker/circuit_breaker.py @@ -78,7 +78,6 @@ def __init__( self._half_open_calls = 0 self._half_open_start: float | None = None # Track half-open entry time self._lock = threading.Lock() - # DEF-CB-LOCK-UNIFICATION-2026-09-12: removed `_async_lock`. # Pre-fix the sync path held `self._lock` and the async path # held a separate `asyncio.Lock`, so a sync thread and an # async coroutine calling `breaker.call()` concurrently on @@ -264,7 +263,6 @@ def state(self) -> CBState: def call(self, func: Callable[..., Any], *args: Any, **kwargs: Any) -> Any: """Execute func through circuit breaker. Supports both sync and async functions. - #35: the pre-fix code did the OPEN→HALF_OPEN jitter via ``time.sleep`` here, BEFORE dispatching to ``_call_sync`` / ``_call_async``. That meant an async caller invoking ``breaker.call(async_func,...)`` from diff --git a/src/nullrun/breaker/exceptions.py b/src/nullrun/breaker/exceptions.py index 0963ea7..81f328e 100644 --- a/src/nullrun/breaker/exceptions.py +++ b/src/nullrun/breaker/exceptions.py @@ -430,11 +430,9 @@ def __init__( **kwargs: Any, ) -> None: self.chain_id = chain_id - # Execution Graph v0 (2026-08-06): when the backend rejects self.parent_execution_id = parent_execution_id self.backend_code = backend_code or self.error_code self.details = details or {} - # 2026-07-04: preserve the wire HTTP self.status_code = status_code super().__init__(message, **kwargs) @@ -489,7 +487,6 @@ def __init__( self.max_allowed_cents = max_allowed_cents self.actual_cost_cents = actual_cost_cents self.epsilon_cents = epsilon_cents - # 2026-07-04: CONSUME_OVERBUDGET maps to self.status_code = status_code super().__init__(message, **kwargs) @@ -525,7 +522,6 @@ def __init__( **kwargs: Any, ) -> None: self.workflow_id = workflow_id - # 2026-07-04: WORKFLOW_INACTIVE maps to self.status_code = status_code super().__init__(message, **kwargs) @@ -701,13 +697,12 @@ class NullRunBlockedException(NullRunDecision): Subclasses (:class:`NullRunBudgetError`,:class:`NullRunToolBlockedError`) carry the specific ``error_code`` and ``user_action`` for each - block reason. ``except NullRunBlockedException`` continues to - match all of them — back-compat. + block reason. ``except NullRunBlockedException`` matches every + typed block subclass. Attributes: workflow_id: Workflow that was blocked (may be a sentinel like - "" when the block fires outside a workflow context - e.g. the sensitive-tool pre-check). + "" when the block fires outside a workflow context). reason: Human-readable explanation of why the block fired. action: One of "block" / "kill" / "pause" — the suggested downstream action. @@ -751,7 +746,6 @@ def __init__( self.reason = reason self.action = action self.tool_name = tool_name - # 2026-07-04: wire HTTP status preserved self.status_code = status_code self.details = details tool_suffix = f", tool={tool_name}" if tool_name else "" @@ -863,10 +857,6 @@ class NullRunExecutionNotFoundError(NullRunBackendError): ``except NullRunBackendError:`` cookbook pattern keeps matching; callers that want to handle this specific case can ``except NullRunExecutionNotFoundError`` for a clearer intent. - - Audit: 2026-09-09 SDK-drift audit — pre-fix SDK 0.15.x collapsed - this code into a generic ``NullRunBackendError("Execution binding - not found")`` with no introspection on whether /gate was missed. """ error_code = "NR-EX01" @@ -935,7 +925,6 @@ class NullRunToolBlockedError(NullRunBlockedException): # --------------------------------------------------------------------------- -# Approval grant-consume outcomes (v3.53 / 2026-08-13 audit, A-1/A-2) # --------------------------------------------------------------------------- # These six typed exceptions wire-up the /execute grant-consume outcomes # that backend `backend/src/proxy/http/gate/internal.rs:3059-3108, 3115-3138` @@ -948,7 +937,6 @@ class NullRunToolBlockedError(NullRunBlockedException): # instead of string-matching the ``error_message``. # # All six subclass :class:`NullRunBlockedException` so the legacy -# ``except NullRunBlockedException:`` pattern keeps matching — back-compat # invariant preserved. class NullRunApprovalNotYetApprovedError(NullRunBlockedException): """The approval row exists but the operator has not yet decided. @@ -1005,12 +993,10 @@ class NullRunApprovalResponseMissingError(NullRunBlockedException): """``/execute`` returned ``require_approval`` but the response body did not include an ``approval_id`` — wire-bug / server drift. - Wire code ``NR-A004`` (was previously set inline on a generic - ``NullRunBlockedException`` at runtime.py:2888, 2914, 2929 — promoted - to a typed class for parity with the six approval exceptions above). - This is distinct from ``NullRunApprovalNotYetApprovedError`` (NR-A010) - which is "the operator has not yet decided". Here the operator never - had a chance — the wire envelope was incomplete. + Wire code ``NR-A004``. This is distinct from + ``NullRunApprovalNotYetApprovedError`` (NR-A010) which is "the + operator has not yet decided". Here the operator never had a + chance — the wire envelope was incomplete. Cookbook pattern: do NOT retry the same execution_id; the backend needs a fix or the wire-shape contract needs re-reading. Log the @@ -1221,7 +1207,6 @@ class NullRunApprovalToolDigestMismatchError(NullRunBlockedException): # ──────────────────────────────────────────────────────────────────────── -# MCP umbrella codes (ADR-013, 2026-08-14, frozen-dormant per Phase B.1) # # Pre-B.1 these three wire codes (``MCP_DESTRUCTIVE_BLOCKED``, # ``MCP_READONLY_BYPASS_BLOCKED``, ``MCP_APPROVAL_REQUIRED``) all @@ -1232,7 +1217,6 @@ class NullRunApprovalToolDigestMismatchError(NullRunBlockedException): # subclasses so cookbook code can ``except NullRunMcpDestructiveBlockedError:`` # (etc.) and surface the right user_action verb. # -# ADR-013 (2026-08-14) marks the umbrella as **frozen-dormant** — # the underlying ``mcp_destructive_policy`` / ``mcp_readonly_bypass`` # mechanisms are not currently wired in production but the wire codes # are reserved and the SDK must round-trip them so a future enablement @@ -1330,7 +1314,6 @@ class NullRunApprovalDbUnavailableError(NullRunBlockedException): # definition. -# NOTE: the following six exception classes were removed in 0.4.0 # because they had no callers in the SDK or in any test. They were # zombie public surface — defined but never raised. If a real use # case emerges in the future, they should be re-added with at least @@ -1374,92 +1357,24 @@ def __init__(self, workflow_id: str, reason: str, resume_after: float | None = N super().__init__(msg) -class WorkflowKilledException(BaseException): - """ - DEPRECATED. Use:class:`WorkflowKilledInterrupt` instead. - - Kept for backward compatibility: this class is the *parent* of -:class:`WorkflowKilledInterrupt`, so user code that does - ``except WorkflowKilledException`` will still catch the new raises - (``except X`` matches subclasses of ``X`` — and the new class is - a subclass of this one). - - A ``DeprecationWarning`` is emitted on construction. The class will - be removed in a future major release; migrate new code to -:class:`WorkflowKilledInterrupt` and update existing - ``except WorkflowKilledException`` clauses to - ``except WorkflowKilledInterrupt`, or, if recovery is impossible - let the exception propagate to the top of the loop. - - This class is **not** an ``Exception`` subclass — kill is a - non-recoverable signal and should not be caught by generic - ``except Exception`` clauses. Only ``except BaseException`` or the - explicit ``except WorkflowKilledInterrupt`` reliably stops the work. - See ``docs/kill-contract.md`` for the full rationale. - - NOTE: NOT inheriting from:class:`NullRunError` because - ``NullRunError`` is an ``Exception`` subclass — and the kill - contract deliberately excludes ``except Exception`` from catching - this signal. The structured fields are attached at construction - time as instance attributes (not class attributes) so the kill - site can still stamp ``error_code`` / ``user_action`` without - breaking the BaseException contract. - """ - - error_code = "NR-W002" - user_action = ( - "The workflow was killed. The body did not run and the kill " - "is non-recoverable from inside the agent loop. Inspect the " - "reason and, if appropriate, resume the workflow at " - "https://app.nullrun.io/workflows/." - ) - retryable = False - - def __init__(self, workflow_id: str, reason: str) -> None: - import warnings as _w - - _w.warn( - "WorkflowKilledException is deprecated. Catch " - "WorkflowKilledInterrupt (BaseException) instead. The class " - "is preserved for backward-compatible `except` clauses but " - "will be removed in a future major release.", - DeprecationWarning, - stacklevel=2, - ) - self.workflow_id = workflow_id - self.reason = reason - super().__init__(f"Workflow {workflow_id} killed: {reason}") - - class WorkflowKilledInterrupt(NullRunError): """ Raised when a workflow is killed by the NullRun control plane. **2026-09-08 migration**: this class is now an ``Exception`` - subclass (``NullRunError`` parent) — formerly ``BaseException``. - The user override: agent recovery code needs to catch the kill - signal via ``except WorkflowKilledInterrupt`` or + subclass (``NullRunError`` parent). Agent recovery code catches + the kill signal via ``except WorkflowKilledInterrupt`` or ``except NullRunWorkflowKilledError`` to surface a structured error to the user with ``error_code=NR-W002`` and ``user_action``. - Migration back-compat guarantees (all three hold): - - * ``except WorkflowKilledInterrupt`` (new code) — still matches, - including legacy raises that haven't been updated. - * ``except NullRunError`` — now matches (was NO match before - migration; this is the new ability the user wanted). - * ``except NullRunWorkflowKilledError`` — matches (preferred - typed name for new cookbook code). - - Migration BREAK (acceptable, documented in CHANGELOG): + Three catch patterns, all of which work: - * ``except WorkflowKilledException`` (the deprecated parent - class) — no longer matches. The parent class remains - BaseException and emits DeprecationWarning on construction, - but is no longer in the ``WorkflowKilledInterrupt`` MRO. Code - that catches the deprecated name must migrate to either - ``WorkflowKilledInterrupt`` (keep current name) or - ``NullRunWorkflowKilledError`` (preferred typed name). + * ``except WorkflowKilledInterrupt`` — canonical. + * ``except NullRunError`` — matches via the new ``NullRunError`` + parent class (this is the ability the typed hierarchy + gives the user). + * ``except NullRunWorkflowKilledError`` — preferred typed + name for new cookbook code. Fields: workflow_id: The workflow that was killed. diff --git a/src/nullrun/business_impact.py b/src/nullrun/business_impact.py index ee6100e..8e5cce8 100644 --- a/src/nullrun/business_impact.py +++ b/src/nullrun/business_impact.py @@ -1,297 +1,48 @@ -"""BusinessImpact + action_digest (SDK mirror of backend). - -The SDK must produce the *exact* same SHA-256 hex digest the Rust -backend computes, so the digest re-check on /execute re-check -matches byte-for-byte. Drift between SDK and backend would be -caught at the first mismatch attack on a real customer. - -Wire format mirrors `backend::proxy::gate::business_impact`: -- discriminated union with a single variant `kind="money"` -- `MoneyImpact(direction, amount_minor, currency, ...)` -- `Condition(MoneyAmount(direction, operator, threshold_minor, - currency))` lives on the **rule side** in the backend; the - SDK never constructs Conditions directly — operators write - them in the dashboard. The SDK only ever produces Impact - payloads. - -JSON canonicalization (backend reference, Rust): - - 1. Serialize via `serde_json::to_value(self)`. - 2. Recursively sort every object key. - 3. Serialize back to compact JSON. - 4. SHA-256 over `b"nullrun/v1/business_impact:" || canonical` - (prefix is part of the digest domain — keeps the v2 - protocol from accidentally matching v1 digests). - -The Python mirror below must match step-for-step. Any drift is -a P0 security bug — see `tests/test_business_impact.py`. """ - +BusinessImpact + action_digest — minimal wire helpers. + +The 0.18.2 SDK is policy-blind. Every ``@protect`` call computes a +single canonical ``NoImpact`` envelope (the only remaining +variant) and forwards it to /execute. The backend's ToolParameters +Approval Rules read values out of ``tool_kwargs`` directly via the +rule's ``param_name`` field, so the SDK no longer constructs +per-tool typed impacts (Money, ToolCall). This module keeps just +enough of the pre-0.18.2 wire layer for the gate to send a +valid (kind, action_digest) pair. + +Field contract mirrored by the backend at +``backend/src/proxy/gate/business_impact.rs``: + + - ``business_impact`` disciminator: ``{"kind": "none"}`` + (the only wire variant the SDK ships post-0.18.2). + - ``action_digest``: SHA-256 over ``DIGEST_PREFIX + compact + canonical JSON of the impact envelope``, lowercase hex. + +The digest is pinned by ``tests/test_business_impact.py`` so the +SDK ↔ backend canonicalisation can't drift silently. +""" from __future__ import annotations import hashlib import json -from dataclasses import dataclass, field -from typing import Any, Optional +from dataclasses import dataclass +from typing import Any DIGEST_PREFIX = b"nullrun/v1/business_impact:" - -# Direction enum (mirror Rust MoneyDirection; lowercase string on wire). -OUTFLOW = "outflow" -INFLOW = "inflow" - - -# Operator enum (mirror Rust ConditionOperator; lowercase string on wire). -GT = "gt" -GTE = "gte" -EQ = "eq" - - -# `money` kind for per-call flat amounts. -# `tool_call` kind for free-form tool-call argument bags matched -# against ToolParameters Approval Rules on the backend. -# `none` kind for non-impact LLM/tool calls that still need an -# `action_digest` on the wire (per `backend/src/proxy/http/gate/gate.rs:56` -# v3.62.1 / ADR-023 P1-6 — Phase-1+ SDKs MUST supply `action_digest` -# even when there is no typed business impact to extract). -KIND_MONEY = "money" -KIND_TOOL_CALL = "tool_call" +# ToolCall envelopes) but the SDK only mints NoImpact. KIND_NONE = "none" -# Mirrors the backend constant at -# ``backend/src/proxy/gate/business_impact.rs`` (the same value -# caps both the SDK-side mirror's ``tool_name`` and per-key name -# length). Kept in sync manually; a backend-side bump is a one-line -# edit here. -TOOL_PARAMETERS_MAX_PARAM_NAME = 64 - - -@dataclass -class MoneyImpact: - """Flat per-call money amount. - - Attributes: - direction: "outflow" (refund/payout) or "inflow" (charge/invoice). - Approval rules only fire on outflow. - amount_minor: integer cents for USD, MUST be non-negative. - Negatives are rejected at validate() time. Sign convention - is `direction`, not `+/- amount` — do not switch. - currency: ISO-4217 (3 uppercase letters). Default is "USD". The - backend treats any other currency as a no-match against a - USD-only rule (separate per-currency rule needed by author). - extractor_id: self-reported SDK extractor id (e.g. "nullrun.money.path"). - extractor_version: self-reported version. - """ - - direction: str - amount_minor: int - currency: str - extractor_id: str = "nullrun.money.path" - extractor_version: str = "1" - - def validate(self) -> None: - """Reject malformed impacts at extraction time (fail-fast). - - Raises ValueError with a human-readable reason. The - backend's `MoneyImpact::validate()` mirrors these checks. - """ - if self.direction not in (OUTFLOW, INFLOW): - raise ValueError(f"direction must be {OUTFLOW!r} or {INFLOW!r}, got {self.direction!r}") - if not isinstance(self.amount_minor, int) or isinstance(self.amount_minor, bool): - # bool is a subclass of int in Python — explicit exclude. - raise ValueError(f"amount_minor must be int, got {type(self.amount_minor).__name__}") - if self.amount_minor < 0: - raise ValueError(f"amount_minor must be non-negative, got {self.amount_minor}") - if ( - not isinstance(self.currency, str) - or len(self.currency) != 3 - or not self.currency.isascii() - or not self.currency.isupper() - ): - raise ValueError( - f"currency must be a 3-letter uppercase ISO-4217 code, got {self.currency!r}" - ) - - def to_wire_dict(self) -> dict[str, Any]: - """Serialize to the JSON shape the backend expects. - - Key order is NOT significant here — the backend's - `BusinessImpact::canonical_json()` re-sorts keys before - hashing. We still emit a stable Python order so debug - logs read top-to-bottom the way the operator wrote them. - """ - return { - "kind": KIND_MONEY, - "direction": self.direction, - "amount_minor": self.amount_minor, - "currency": self.currency, - "extractor_id": self.extractor_id, - "extractor_version": self.extractor_version, - } - - -@dataclass -class ToolCallParams: - """Free-form tool-call argument bag. - - Mirrors the backend ``BusinessImpact::ToolCall(ToolCallParams)`` - variant at ``backend/src/proxy/gate/business_impact.rs:62-307``. - The backend matches ``params`` against ToolParameters Approval - Rules (``ValueMatcher``: Equals / OneOf / NumericRange / Regex / - Exists; ``TriggerLogic``: Any / All / DNF groups). - - Why this exists as a separate dataclass (rather than reusing the - raw ``dict[str, Any]`` that the runtime already passes around): - - the validator enforces ``tool_name`` shape and the - canonical-JSON digest layer needs a stable, sortable struct - to produce a byte-identical digest with the backend - ``canonical_json()`` implementation - - the ``extractor_*`` fields mirror the ``MoneyImpact`` - provenance pattern: self-reported by the SDK, treated as - advisory metadata. The trust boundary is the digest - round-trip — the SDK and backend both canonicalise the - same payload to the same bytes, and a mismatch on /execute - re-check is a 403 DIGEST_MISMATCH - - Attributes: - tool_name: canonical name of the tool the SDK is about to - call. Must be non-empty and <= 128 bytes. - params: free-form argument bag the operator wrote the rule - against. Keyed by the rule's ``param_name`` field. - extractor_id: self-reported SDK extractor id (e.g. - "nullrun.tool_call.path"). - extractor_version: self-reported version. - """ - - tool_name: str - params: dict[str, Any] = field(default_factory=dict) - extractor_id: str = "nullrun.tool_call.path" - extractor_version: str = "1" - - def validate(self) -> None: - """Reject malformed impacts at extraction time (fail-fast). - - Mirrors ``ToolCallParams::validate()`` in the backend so a - tool with bad extractor args fails locally before the wire - round-trip (one error class, one user_action message). - """ - if not isinstance(self.tool_name, str) or not self.tool_name: - raise ValueError("tool_name must be a non-empty string") - if len(self.tool_name) > 128: - raise ValueError(f"tool_name length {len(self.tool_name)} exceeds max 128") - if not self.tool_name.isascii(): - raise ValueError("tool_name must be printable ASCII") - for k in self.params: - if not isinstance(k, str): - raise ValueError(f"params key {k!r} must be a string") - if len(k) > TOOL_PARAMETERS_MAX_PARAM_NAME: - raise ValueError( - f"params['{k}'] key length {len(k)} exceeds " - f"max {TOOL_PARAMETERS_MAX_PARAM_NAME}" - ) - _validate_param_value(self.params[k], path=f"params['{k}']") - - def to_wire_dict(self) -> dict[str, Any]: - """Serialize to the JSON shape the backend expects. - - Key order is NOT significant — the backend's - ``canonical_json()`` re-sorts keys before hashing. - """ - return { - "kind": KIND_TOOL_CALL, - "tool_name": self.tool_name, - "params": dict(self.params), - "extractor_id": self.extractor_id, - "extractor_version": self.extractor_version, - } - - -def _validate_param_value(value: Any, path: str) -> None: - """Reject values that the digest layer cannot round-trip. - - Backend mirror at ``business_impact.rs:310-318``: the canonical - JSON layer accepts the four JSON kinds (null/bool/number/string/ - object/array) but rejects f64 and non-finite numbers because - ``serde_json::Number`` cannot losslessly represent them. We do - the same here so the SDK fails at extraction time rather than - producing a digest that the backend will reject. - """ - if value is None or isinstance(value, bool): - return - if isinstance(value, int): - # int round-trips through JSON losslessly. NOTE: bool is a - # subclass of int in Python; we explicitly handle it above. - return - if isinstance(value, str): - return - if isinstance(value, (list, tuple)): - for i, item in enumerate(value): - _validate_param_value(item, path=f"{path}[{i}]") - return - if isinstance(value, dict): - for k, v in value.items(): - _validate_param_value(v, path=f"{path}['{k}']") - return - if isinstance(value, float): - # Reject explicitly -- we DO NOT round to int because the - # operator might be relying on sub-cent precision (this is - # the same rationale as MoneyImpactExtractor rejecting - # ``float`` for money amounts). - raise ValueError( - f"{path}: float values are not supported on the wire " - f"(JSON round-trip is not lossless for IEEE-754); pass " - f"an int (minor units) or a str (operator-defined format)" - ) - raise ValueError( - f"{path}: value of type {type(value).__name__!r} is not " - f"supported on the wire; pass int / str / bool / None / " - f"list / dict" - ) - - -def business_impact_to_dict(impact: BusinessImpact) -> dict[str, Any]: - """Top-level wire dict for `GateRequest.business_impact`. - - Returns an empty string key discriminator for the backend's - `serde(tag = "kind", rename_all = "snake_case")` shape. - """ - return impact.to_wire_dict() - - -# Dataclasses that mirror the Rust backend's discriminated union via -# `kind` discriminator. In Python we represent the union as a -# tagged dict at the wire layer and a small class hierarchy at the -# in-process layer. The SDK validates the variant at construction -# time so the backend never sees malformed output. @dataclass class NoImpactPayload: - """Sentinel payload for non-impact calls (plain LLM chat, etc.). - - Phase-1+ SDKs MUST populate `action_digest` on every `/gate` - call (per `backend/src/proxy/http/gate/gate.rs:56` v3.62.1 / - ADR-023 P1-6 — fail-CLOSED wire-shape version-gate). Calls - that have no typed business impact (regular LLM chat, - read-only tool calls without an approval rule) need a - deterministic digest that the wire-shape check accepts. + """Sentinel payload for non-impact calls. The ONLY variant the - The canonical JSON of this payload is ``{"kind":"none"}`` - (compact, key-sorted). The corresponding digest is the SHA-256 - of ``nullrun/v1/business_impact:{"kind":"none"}`` and is - pinned as a literal in - ``tests/test_business_impact.py::test_no_impact_digest_pins_hex`` - so a drift between SDK and backend (or a stray canonicalisation - change) is caught at unit-test time. - - This variant exists ONLY on the SDK side. The backend's - `GateRequestBody.business_impact` field stays ``None`` for - non-impact calls — the `action_digest` field is the one the - wire-shape gate checks. Adding a NoImpact arm to the backend's - ``BusinessImpact`` enum is a follow-up if / when the digest - re-check path needs to reverse-hash the impact (currently - it doesn't — the re-check only fires when an approval row - is involved, which requires a typed impact by definition). + The canonical JSON of this payload is ``{"kind":"none"}``. + The corresponding digest is the SHA-256 of + ``nullrun/v1/business_impact:{"kind":"none"}`` and is pinned + by tests as a literal hex so SDK ↔ backend canonicalisation + can't drift silently. """ def validate(self) -> None: @@ -303,90 +54,35 @@ def to_wire_dict(self) -> dict[str, Any]: @dataclass class BusinessImpact: - """Top-level BusinessImpact union. - - Variants: - `Money`: flat per-call money amount (cents, USD-centric). - `ToolCall`: free-form tool-call argument bag matched - against ToolParameters Approval Rules on the backend. - `NoImpact`: sentinel for non-impact calls (regular LLM - chat, tool calls without a typed approval rule). - Wire shape: ``{"kind":"none"}``. Lets the SDK - compute an `action_digest` that satisfies the - Phase-1+ wire-shape version-gate without inventing - a fake typed impact. - - The SDK validates the variant at construction time so the - backend never sees malformed output. + """Top-level BusinessImpact envelope. + + 0.18.2: only the ``NoImpact`` payload variant is constructed + on the SDK side. The ``money`` and ``tool_call`` factories + were removed because the SDK no longer stamps per-tool + typed impacts onto functions — every call sends NoImpact and + the backend reads live values out of ``tool_kwargs`` via the + rule's ``param_name``. """ - impact: Any # MoneyImpact | ToolCallParams | NoImpactPayload + impact: Any # Only NoImpactPayload in 0.18.2. @property def kind(self) -> str: - if isinstance(self.impact, MoneyImpact): - return KIND_MONEY - if isinstance(self.impact, ToolCallParams): - return KIND_TOOL_CALL if isinstance(self.impact, NoImpactPayload): return KIND_NONE - raise TypeError(f"unknown impact type: {type(self.impact)}") + raise TypeError( + f"unknown impact type: {type(self.impact)!r} — only " + ) def validate(self) -> None: self.impact.validate() def to_wire_dict(self) -> dict[str, Any]: - return business_impact_to_dict(self.impact) - - @classmethod - def money( - cls, - direction: str, - amount_minor: int, - currency: str = "USD", - ) -> BusinessImpact: - m = MoneyImpact( - direction=direction, - amount_minor=amount_minor, - currency=currency, - ) - m.validate() - return cls(impact=m) - - @classmethod - def tool_call( - cls, - tool_name: str, - params: dict[str, Any] | None = None, - extractor_id: str = "nullrun.tool_call.path", - extractor_version: str = "1", - ) -> BusinessImpact: - """Construct a ``kind="tool_call"`` BusinessImpact. - - Used by the ToolParamsExtractor; callers building impacts - by hand should use this factory rather than constructing - ``ToolCallParams`` and wrapping themselves -- the factory - validates before returning so a misuse fails locally - instead of after a wire round-trip. - """ - p = ToolCallParams( - tool_name=tool_name, - params=params or {}, - extractor_id=extractor_id, - extractor_version=extractor_version, - ) - p.validate() - return cls(impact=p) + return self.impact.to_wire_dict() @classmethod def no_impact(cls) -> BusinessImpact: - """Sentinel ``kind="none"`` BusinessImpact for non-impact calls. - - Use this in ``runtime.check_workflow_budget`` and other - /gate sites that don't extract a typed impact but still - need to compute an `action_digest` to satisfy the - backend's Phase-1+ wire-shape version-gate. - """ + """Construct the canonical ``kind="none"`` envelope.""" n = NoImpactPayload() n.validate() return cls(impact=n) @@ -395,12 +91,10 @@ def no_impact(cls) -> BusinessImpact: def _canonicalize_json(value: Any) -> Any: """Sort object keys recursively before serialization. - Mirrors `BusinessImpact::canonical_json()` in the backend. + Mirrors ``BusinessImpact::canonical_json()`` in the backend. """ if isinstance(value, dict): - items = [] - for k, v in value.items(): - items.append((k, _canonicalize_json(v))) + items = [(k, _canonicalize_json(v)) for k, v in value.items()] items.sort(key=lambda kv: kv[0]) return {k: v for k, v in items} if isinstance(value, list): @@ -411,19 +105,15 @@ def _canonicalize_json(value: Any) -> Any: def compute_action_digest(impact: BusinessImpact) -> str: """Compute the SHA-256 digest the backend expects. - Algorithm (must match backend/src/proxy/gate/business_impact.rs + Algorithm (must match ``backend/src/proxy/gate/business_impact.rs`` byte-for-byte): - 1. Validate the impact at extract time (fail-fast). - 2. Convert to wire dict. + + 1. Validate the impact (``NoImpactPayload`` is the only + post-0.18.2 variant — fail-fast on bad input). + 2. Convert to wire dict (``{"kind":"none"}``). 3. Canonicalize (sort object keys recursively). 4. Serialize to compact JSON (no spaces). - 5. Hash with the protocol-prefix bytes as a salt. - 6. Return lowercase hex. - - Returns 64 lowercase hex characters. The backend's - `compute_action_digest` is byte-identical; any drift is a - P0 security regression covered by - `tests/test_business_impact.py::test_digest_matches_backend`. + 6. Return lowercase hex (64 chars). """ impact.validate() canonical_value = _canonicalize_json(impact.to_wire_dict()) @@ -439,7 +129,10 @@ def compute_action_digest(impact: BusinessImpact) -> str: return hasher.hexdigest() -# Backwards-compat: a thin class wrapper for the discriminated union -# is exposed via `BusinessImpact.kind` and `BusinessImpact.to_wire_dict`, -# but tests and runtime code that already uses dict literals continue -# to work. The validator at extract time catches malformed payloads. +__all__ = [ + "DIGEST_PREFIX", + "KIND_NONE", + "NoImpactPayload", + "BusinessImpact", + "compute_action_digest", +] diff --git a/src/nullrun/capabilities.py b/src/nullrun/capabilities.py index 4e6115b..4b63f8d 100644 --- a/src/nullrun/capabilities.py +++ b/src/nullrun/capabilities.py @@ -81,7 +81,6 @@ # it does NOT carry the v3-gating fields, so probing there always # returned None and ``is_v3_ready()`` was always False, leaving # every capability flag a no-op at runtime. See capability -# history note in module docstring (2026-07-06 fix). CAPABILITIES_PATH = "/api/v1/capabilities" @@ -133,7 +132,6 @@ class ServerCapabilities: decision_log: bool = False outbox_async_drain: bool = False idempotency_keys: bool = False - # Execution Graph v0 (2026-08-06, backend): additive # `parent_execution_id` wire field on /gate. SDKs probe this # flag before sending the field; pre-Graph backends silently # ignore unknown fields, but the probe lets SDKs surface a @@ -142,7 +140,6 @@ class ServerCapabilities: # included in `is_v3_ready()` -- it's informational, not a # hard gate. execution_graph: bool = False - # ADR-037 Slice B (2026-08-31, protocol v4): /gate response # echoes the SDK-supplied `action_digest` and a `policy_hash` # slot (None today; Slice D wires per-request computation). # Backend always sends the fields (skip_serializing_if elides @@ -351,12 +348,10 @@ def _v3_flag(name: str) -> bool: decision_log=_v3_flag("decision_log"), outbox_async_drain=_v3_flag("outbox_async_drain"), idempotency_keys=_v3_flag("idempotency_keys"), - # Execution Graph v0 (2026-08-06, backend): additive flag # -- defaults to False so pre-Graph backends (which omit # the field entirely) yield a fail-closed view where the # SDK does NOT send `parent_execution_id`. execution_graph=_v3_flag("execution_graph"), - # ADR-037 Slice B (2026-08-31, protocol v4): additive # flag — defaults to False so pre-Slice-B backends yield # a fail-closed view where the SDK does NOT log the # wire-evidence echo as "server confirmed". Pre-v4 diff --git a/src/nullrun/context.py b/src/nullrun/context.py index 539b59e..f649fe0 100644 --- a/src/nullrun/context.py +++ b/src/nullrun/context.py @@ -2,21 +2,6 @@ Context management for NullRun SDK. Provides workflow and trace context for automatic event correlation. - -The previously-defined ``_organization_id_var`` / ``_api_key_id_var`` -contextvars and the ``get_organization_id`` / ``get_api_key_id`` -getters were removed (B27) because: - 1. No code path ever wrote to them — both getters always - returned ``None``. - 2. ``observability.TenantFilter`` (the only consumer) was - removed in 0.3.1. - 3. The structured-logging tenant-isolation feature moved to - the backend in the same release. - -If a future use case appears (e.g. per-API-key rate isolation) -re-introduce the contextvars AND a setter API (token-based like -``set_attempt_index``) AND wire them in ``NullRunRuntime.__init__`` -from the ``_authenticate`` response. """ import uuid @@ -24,9 +9,7 @@ from contextlib import contextmanager from contextvars import ContextVar, Token -# 2026-08-14 (F-19 fix): ``nullrun.tracing`` provides the structured # SpanContext that models the parent/child hierarchy a trace timeline -# needs. ``nullrun.context`` previously owned loose ``_trace_id`` / # ``_span_id`` contextvars and now keeps them in lockstep via the # ``_mirror_to_span_context`` / ``_mirror_to_legacy_span`` helpers # below; ``@protect`` (decorators.py:441) and any other writer must @@ -70,7 +53,6 @@ "call_mcp_annotations", default=None ) -# 2026-07-02 (v0.11.0): chain_id contextvar for soft-mode gate # . # # Soft-mode budget enforcement ONLY allows overdrafts when an @@ -278,7 +260,6 @@ def set_chain_op(op: str) -> Token[str]: # --------------------------------------------------------------------------- -# Server-minted execution_id (2026-07-04 — ) # --------------------------------------------------------------------------- # # Pre-0.12.0 the SDK sent a client-supplied ``execution_id`` (usually @@ -324,7 +305,6 @@ def set_chain_op(op: str) -> Token[str]: _server_minted_reservation_at_var: ContextVar[float] = ContextVar( "server_minted_reservation_at", default=0.0 ) -# 2026-07-04: /track idempotency anchor. # The /check request carries ``idempotency_key = operation_id`` (UUID v4) # the backend's /track handler (handlers.rs:4654-4725) accepts the same # key and replays the original response on hit (200 + ``idempotent_replay: @@ -340,14 +320,12 @@ def set_chain_op(op: str) -> Token[str]: _server_minted_idempotency_key_var: ContextVar[str | None] = ContextVar( "server_minted_idempotency_key", default=None ) -# AUDIT P0-27 (2026-09-05): operation_id hoist. # # Pre-fix, runtime.py minted operation_id independently at the # /check site (line 1913) and the /execute site (line 2749). # A single logical action therefore produced two distinct # operation_ids — the backend's binding (which keys on # operation_id) saw them as two unrelated reservations, and the -# P0-26 response-echo capture (`response.get("operation_id")`) # silently recorded whichever value the server echoed last. # # Fix: hoist the mint into a single contextvar owned by the @@ -359,7 +337,6 @@ def set_chain_op(op: str) -> Token[str]: _operation_id_var: ContextVar[str | None] = ContextVar( "operation_id", default=None ) -# ADR-037 Slice B (2026-08-31, protocol v4): wire-evidence echo # from /gate response. Both fields are ADR-009 governance columns # that the backend now echoes on the /gate response (additive — # pre-v4 backends omit the keys entirely via skip_serializing_if). @@ -521,7 +498,6 @@ def clear_server_minted_execution_id() -> None: _server_minted_execution_id_var.set(None) _server_minted_reservation_at_var.set(0.0) _server_minted_idempotency_key_var.set(None) - # ADR-037 Slice B (2026-08-31, protocol v4): also drop the # wire-evidence echo slots so a /check in one block never leaks # a stale echo into a /track in a sibling block. _last_gate_action_digest_var.set(None) @@ -534,7 +510,6 @@ def set_attempt_index(index: int) -> None: # --------------------------------------------------------------------------- -# AUDIT P0-27 (2026-09-05) — operation_id lifecycle helpers. # # The runtime mints the operation_id once at the top of the # public gate/enforce entry point (``NullRunRuntime.execute`` @@ -603,7 +578,6 @@ def clear_operation_id() -> None: # --------------------------------------------------------------------------- -# ADR-037 Slice B (2026-08-31, protocol v4): wire-evidence echo # --------------------------------------------------------------------------- # Read by tests + operators to confirm the gate saw the same # `action_digest` the SDK sent (and to surface the architectural @@ -666,7 +640,6 @@ def set_last_gate_policy_hash(value: str | None) -> None: # --------------------------------------------------------------------------- -# F-19 (2026-08-14): legacy _trace_id / _span_id token-based setters # --------------------------------------------------------------------------- # # ``nullrun.tracing.SpanContext`` is the canonical source-of-truth at @@ -719,7 +692,6 @@ def reset_span_id(token: Token[str | None]) -> None: # --------------------------------------------------------------------------- -# F-19 (2026-08-14): helpers used by ``with workflow`` / ``with span`` # --------------------------------------------------------------------------- # # ``with workflow`` writes a fresh root ``SpanContext``; ``with span`` @@ -808,8 +780,8 @@ def set_call_context( tools: List of tool names the call intends to use. Backend matches each against the workflow's effective ``blocked_tools`` aggregate and returns block on any - match. Pass ``None`` to leave whatever was previously - set, ``[]`` to clear. + match. Pass ``None`` to leave the current value + unchanged, ``[]`` to clear. """ if model is not None: _call_model_var.set(model) @@ -889,7 +861,6 @@ def workflow(name: str | None = None) -> Generator[str, None, None]: workflow_id = name or str(uuid.uuid4()) trace_id = generate_trace_id() # a new workflow gets a fresh span_id too. The - # pre-fix code only reset workflow_id and trace_id, so a # ``with span("inner"); with workflow("outer")`` block would # leave the inner span_id visible inside the workflow scope — # the span emitted by the workflow would carry the wrong @@ -903,7 +874,6 @@ def workflow(name: str | None = None) -> Generator[str, None, None]: wf_token = _workflow_id_var.set(workflow_id) trace_token = _trace_id_var.set(trace_id) span_token = _span_id_var.set(span_id) - # F-19 (2026-08-14): dual-write a root SpanContext onto # ``_current_span`` so an inner ``@protect`` (or nested # ``with span``) derives child spans from THIS workflow's # trace_id rather than minting a fresh disconnected root. @@ -944,7 +914,6 @@ def span(name: str | None = None) -> Generator[str, None, None]: """ span_id = name or generate_span_id() token = _span_id_var.set(span_id) - # F-19 (2026-08-14): when a SpanContext is already active # (e.g. we're inside ``with workflow(...)`` or ``@protect``), # push a child SpanContext onto ``_current_span`` so that nested # ``@protect`` calls and the runtime's @@ -992,7 +961,6 @@ def agent(name: str | None = None) -> Generator[str, None, None]: # ``generate_trace_id`` / ``generate_span_id``). The previous # ``f"agent-{uuid.uuid4.hex}"`` format was 32 hex chars # without dashes; backend UUID-typed columns (cost_events. - # agent_id, audit_log) silently dropped these to NULL on insert # (``Uuid::parse_str(...).ok `` returned None). User-supplied # ``name`` is preserved verbatim so existing dashboards continue # to work for already-allocated agent ids. @@ -1036,7 +1004,6 @@ def attempt(attempt_index: int) -> Generator[int, None, None]: _attempt_index_var.reset(token) -# 2026-07-02 (v0.11.0): chain context manager for soft-mode budget # enforcement. # # Usage: diff --git a/src/nullrun/decorators.py b/src/nullrun/decorators.py index 2864615..716b1be 100644 --- a/src/nullrun/decorators.py +++ b/src/nullrun/decorators.py @@ -40,6 +40,7 @@ def researcher(q): import logging import os import threading +import warnings from collections.abc import Callable from contextvars import Token from typing import Any, TypeVar @@ -50,6 +51,7 @@ def researcher(q): WorkflowKilledInterrupt, WorkflowPausedException, ) +from nullrun.business_impact import BusinessImpact, compute_action_digest from nullrun.context import ( _call_tools_var, get_call_tools, @@ -135,7 +137,6 @@ def _safe_repr(value: object, max_len: int = 50) -> str: strings we actually pass through this code path. P3-3: also consolidates the two-pass flow that - previously lived as separate ``_safe_repr`` + ``_strip_details_balanced`` calls — there are now two callers that compose them, and the invariant ``redact BEFORE truncate`` was being maintained by convention only. ``_safe_repr`` is now the single source of truth. @@ -513,8 +514,8 @@ async def my_async_agent(query: str) -> str: 1. `check_control_plane` — KILL/PAUSE is terminal. 2. `check_workflow_budget` — "any budget left?" via /gate. - 3. `_enforce_sensitive_tool` — per-tool policy (no-op if not - marked sensitive). + 3. `_run_tool_policy_gate` — per-tool policy via /execute + (runs on every call). Each gate has its own fail-OPEN/CLOSED policy declared in `runtime.py`; see ADR-008 Rule 5 for the full table. `span_end` @@ -535,49 +536,27 @@ def g:... # bound to itself so the next call wraps the target function. return protect - # 0.18.1: every `@protect` call now auto-attaches a default - # `ToolParamsExtractor(include_all=True)` on the decorated function - # so the wire payload carries ``tool_name + params`` for any - # protected tool, not just those that opted in via bare - # ``@sensitive``. The extractor is only stamped when no extractor - # already exists in the ``__wrapped__`` chain (explicit - # ``@sensitive(impact=...)`` wins). This is the single change that - # makes ``@protect`` the only public entry point users need: - # the SDK now collects every fact it can derive mechanically - # (tool identity, kwargs, action_digest) without forcing the - # developer to reach for a second decorator. The business - # interpretation of those facts remains NullRun policy's job. - # - # The auto-attached extractor carries ``_nullrun_auto_attached=True`` - # so ``_enforce_sensitive_tool`` can distinguish "developer - # opted into the policy path" from "SDK auto-derived the - # extractor for tooling reasons". The policy gate still - # short-circuits on auto-attached extractors so bare ``@protect`` - # stays cheap (no extra ``/execute`` round-trip per call). - try: - from nullrun.extractor import ToolParamsExtractor - - if _find_extractor_in_chain(fn) is None: - auto_extractor = ToolParamsExtractor(include_all=True) - auto_extractor._nullrun_auto_attached = True # type: ignore[attr-defined] - _stamp_extractor_on_innermost(fn, auto_extractor) - except ImportError: - # Defensive: extractor module is part of every SDK build we - # ship today. Falling through without an extractor means the - # gate runs the legacy approval_id-only path (no business_impact - # on the wire) — which is the same behaviour every pre-0.18.1 - # ``@protect`` already had, so this is a no-op for callers - # on a shrunken build. - pass + # NOTE: prior 0.18.x versions auto-attached a default + # ``ToolParamsExtractor(include_all=True)`` here and stored it + # on the function via ``_nullrun_extractor``. That path was + # removed because it forced the SDK to know what an "extractor" + # is. The 0.18.2 design is simpler: ``@protect`` has no + # per-function state. Every call constructs an opaque + # ``NoImpact`` envelope locally and forwards it to + # ``runtime.execute(...)``. All policy decisions + # (allow / block / require-approval) live on the backend; the + # SDK only relays (tool_name, kwargs, args) and renders the + # decision back into an exception class. @contextlib.contextmanager def _protect_body(args: tuple[Any, ...], kwargs: dict[str, Any], unify_block: bool): """Shared ADR-008 Rule-4 scaffolding for sync + async wrappers. - Runs the four pre-execution gates (KILL/PAUSE → budget → span - start → sensitive-tool policy), yields the runtime so the - caller can invoke ``fn`` and ``track_tool`` within the gated - region, then emits ``span_end`` with the captured error. + Runs the pre-execution gates (KILL/PAUSE → /gate budget + pre-flight → span start → /execute tool policy), yields the + runtime so the caller can invoke ``fn`` and ``track_tool`` + within the gated region, then emits ``span_end`` with the + captured error. ``unify_block`` controls the kill/pause signal translation. Sync wrappers pass ``True`` so the user sees a single @@ -590,7 +569,6 @@ def _protect_body(args: tuple[Any, ...], kwargs: dict[str, Any], unify_block: bo runtime = _get_or_create_runtime() span = _next_span() token = set_span(span) - # F-19 (2026-08-14): mirror the derived SpanContext back to # the legacy ``_trace_id_var`` / ``_span_id_var`` so the # runtime's ``_enrich_event`` (which reads via # ``get_trace_id()`` / ``get_span_id()`` for cost events @@ -605,17 +583,15 @@ def _protect_body(args: tuple[Any, ...], kwargs: dict[str, Any], unify_block: bo # restores the outer trace/span on reset. trace_legacy_token = set_trace_id(span.trace_id) span_legacy_token = set_span_id(span.span_id) - # F03 (2026-08-22): populate `_call_tools_var` from # ``fn.__name__`` when the user did NOT explicitly call # ``set_call_context(tools=...)``. The F01 fix # (``runtime.execute`` body at runtime.py:2746-2760 and the # /gate path at runtime.py:1903-1941) conditionally forwards # the per-call tools contextvar onto the wire body, but the - # upstream contextvar was never populated for the @protect / - # @sensitive decorator path. Without this fix every wire - # round-trip omits the `tools` field, the backend's Step 3 - # tool_block check fails-CLOSED via TB-1 - # (``no_tools_field``), and approval-rule probes (TC-SDK-014 + # upstream contextvar was never populated for the @protect + # decorator path. Without this fix every wire round-trip + # omits the `tools` field, the backend's Step 3 tool_block + # check fails-CLOSED via TB-1 (``no_tools_field``), and # /015/016/017) never reach the approval_rule_eval step. # Token-based so a nested @protect inside an outer @protect # (or inside ``with workflow``) restores the outer contextvar @@ -630,7 +606,6 @@ def _protect_body(args: tuple[Any, ...], kwargs: dict[str, Any], unify_block: bo call_tools_token = None error: BaseException | None = None try: - # 2026-09-22: bump the zero-activity diagnostic counter so # the runtime can warn when @protect fires often but no # LLM-call event is ever observed (silent-instrumentation # failure mode). The bump lives at the entry of the gate @@ -654,9 +629,12 @@ def _protect_body(args: tuple[Any, ...], kwargs: dict[str, Any], unify_block: bo # 3. Span start — best-effort, never blocks. _emit_span_start(runtime, span, fn.__name__) - # 4. Per-tool policy for @sensitive tools. Fails CLOSED - # on transport error (see _enforce_sensitive_tool). - _enforce_sensitive_tool(runtime, fn, args, kwargs) + # 4. Per-tool policy gate via /execute. Runs on EVERY + # @protect call (no extractor / no short-circuit). The + # SDK is policy-blind; it ships tool_name + args + kwargs + # + the NoImpact envelope/digest to the backend and the + # backend decides allow/block/require-approval. + _run_tool_policy_gate(runtime, fn, args, kwargs) yield runtime except BaseException as exc: # noqa: BLE001 @@ -693,7 +671,6 @@ def _protect_body(args: tuple[Any, ...], kwargs: dict[str, Any], unify_block: bo reset_span_id(span_legacy_token) # F03 follow-up: reset the per-call tools contextvar if # we set it. Outer ``with workflow`` / nested @protect - # callers that previously set the contextvar see their # prior value restored; bare @protect leaves the # contextvar empty again (the default). if call_tools_token is not None: @@ -763,197 +740,71 @@ def sync_wrapper(*args: Any, **kwargs: Any) -> Any: return sync_wrapper # type: ignore[return-value] -def _enforce_sensitive_tool( +def _run_tool_policy_gate( runtime: Any, fn: Callable[..., Any], args: tuple[Any, ...], kwargs: dict[str, Any], ) -> None: """ - Pre-execution policy check for sensitive tools. + Pre-execution per-tool policy gate — runs on EVERY ``@protect``. - If `fn.__name__` is in the runtime's sensitive-tool set (built-in - or registered via `add_sensitive_tool` / `@sensitive`), call - `runtime.execute(...)` BEFORE the body runs. The /execute endpoint - is the authoritative gate; `NullRunBlockedException` propagates to - the caller, mirroring the contract of `check_workflow_budget`. - - kwargs are masked via `SENSITIVE_ARG_KEYS` so passwords / tokens - never leave the process. The same masking is used for span events. + The 0.18.2 design makes every protected call flow through + ``runtime.execute`` unconditionally; the SDK is policy-blind + and just relays ``tool_name + masked_args + masked_kwargs + + NoImpact envelope`` to the /execute endpoint. The backend + applies allow/block/require-approval rules. ## Fail-OPEN/CLOSED Policy (ADR-008) - This gate is **fail-CLOSED**: the body MUST NOT run when the - policy engine is unreachable, regardless of what /execute returns. - Two failure paths both result in `NullRunBlockedException`: - - 1. **Transport raises** `NullRunTransportError` (the new - `on_transport_error="raise"` path): the runtime layer surfaces - classified NETWORK / GATEWAY / BREAKER-OPEN failures as - exceptions. The body of this gate catches them and re-raises - as `NullRunBlockedException` with the source in the reason - ("policy engine unavailable: NETWORK_ERROR" etc.). - - 2. **Transport returns a dict** whose `decision_source` starts - with `FALLBACK_` (defense in depth — covers the legacy - `fallback_mode=PERMISSIVE` path and any future regression in - `runtime.execute` that drops the `on_transport_error="raise"` - argument). The body of this gate inspects the result and - re-raises as `NullRunBlockedException` before the wrapped - function runs. - - This is the opposite of `check_workflow_budget` / - `check_control_plane`, which deliberately fail-OPEN — a transient - backend outage must not freeze the user's agent. Sensitive tools - have a different threat model: an unblocked `charge_card ` that - runs when the policy engine is down is worse than a denied - `charge_card ` during an outage. - - Opt-out: set `NULLRUN_SENSITIVE_FAIL_OPEN=1` to restore the prior - fail-OPEN behavior on transport error. Useful in dev / test - environments where the policy engine is intentionally absent. - The opt-out is intentionally scoped to the *transport-error* - case; a real `decision=block` from the gateway is still honored - and still raises `NullRunBlockedException`. + The per-tool policy gate is **fail-CLOSED**: the body MUST NOT + run when the policy engine is unreachable. An unblocked + ``charge_card`` running while the policy engine is offline is + a security regression, far worse than a denied call during + the outage. + + Opt-out: set ``NULLRUN_SENSITIVE_FAIL_OPEN=1`` to restore fail- + OPEN behavior on transport error (dev / test only). Real + ``decision=block`` from the gateway is still honored and still + raises ``NullRunBlockedException``. + + ## Wire contract + + Same fields on /execute as before: ``tool_name``, + ``{"args": masked_args, "kwargs": masked}``, ``business_impact`` + (now always ``{"kind": "none"}``), ``action_digest`` (SHA-256 + over the canonical NoImpact envelope; pinned, deterministic), + ``tools``. Backend unchanged — only the SDK's interpretation of + what to put in ``business_impact`` simplified. """ - # 2026-07-24 (Root-cause fix): the previous code used - extractor = getattr(fn, "_nullrun_extractor", None) - # 0.18.1: distinguish auto-attached extractors (SDK installed - # them for tooling reasons -- "every @protect captures tool_params - # automatically") from explicit extractors (developer opted in via - # ``@sensitive(impact=...)``). The policy gate still fires for - # explicit extractors; auto-attached ones are the SDK's way of - # shipping tool_params on the wire without the developer having - # to mark the tool sensitive. Bare ``@protect`` stays cheap. - extractor_is_explicit = ( - extractor is not None - and not getattr(extractor, "_nullrun_auto_attached", False) - ) - if not runtime.is_sensitive_tool(fn.__name__) and not extractor_is_explicit: - return masked = _safe_kwargs(kwargs) - # P0-1: positional args are masked the same way as kwargs. Without masked_args = _safe_args(fn, args) - # If the wrapped function carries an ``_nullrun_extractor`` + # Wire-shape compatibility: ``business_impact`` stays None + # on /execute when no per-tool typed impact is extracted + # (the legacy bare @protect shape — backend reads only + # ``action_digest`` + ``kwargs`` for ToolParameters Approval + # Rules). The ``action_digest`` is still computed against + # the canonical NoImpact envelope so the Phase-1+ wire-shape + # ``tests/test_business_impact.py``. + no_impact = BusinessImpact.no_impact() business_impact_dict: dict[str, Any] | None = None - action_digest_hex: str | None = None - # ``extractor`` was already resolved at the top of this - # function (line 626) for the gate-skip check; reuse the - # binding here so we do not pay for a second ``getattr`` and - # so a future change to that lookup applies to both sites. - if extractor is not None: - try: - from nullrun.business_impact import compute_action_digest - from nullrun.extractor import MoneyImpactExtractor, ToolParamsExtractor - - if isinstance(extractor, MoneyImpactExtractor): - impact = extractor.impact_for(fn, args, kwargs) - business_impact_dict = impact.to_wire_dict() - action_digest_hex = compute_action_digest(impact) - elif isinstance(extractor, ToolParamsExtractor): - # Free-form tool-call argument bag, matched against - # ToolParameters Approval Rules on the backend. - # Same wire envelope (BusinessImpact) and same - # digest contract as the Money variant -- only the - # discriminator and the ``params`` field differ. - impact = extractor.impact_for(fn, args, kwargs) - business_impact_dict = impact.to_wire_dict() - action_digest_hex = compute_action_digest(impact) - except Exception as exc: # noqa: BLE001 - from nullrun.breaker.exceptions import ( - NullRunBlockedException, - NullRunTransportError, - TransportErrorSource, - ) - - # DEFS-SDKEXEC-WORKFLOW-LABEL (2026-09-08): prefer the - # runtime's bound workflow (from _authenticate) over the - # sentinel so the displayed label matches what the SDK - # actually sends to /gate / /execute. See the matching - # note in `_enforce_sensitive_tool` below for the full - # rationale. - workflow_id = runtime._resolve_workflow_id(get_workflow_id()) or UNKNOWN_WORKFLOW_ID - # The user-facing hint depends on which extractor fired. - # Money extractor wants the bound arg name; ToolParams - # extractor wants the rule-param -> arg-name mapping - # (or the include_all flag if no map was supplied). - if isinstance(extractor, MoneyImpactExtractor): - hint = ( - "could not extract a MoneyImpact from the live " - "arguments. Check that the function declares the " - "argument named in `impact=money_outflow(...)`." - ) - elif isinstance(extractor, ToolParamsExtractor): - if extractor.param_extractors is not None: - hint = ( - "could not extract a ToolCall impact from the " - "live arguments. Check that the function " - "declares every arg named in " - "`impact=tool_params(...)`." - ) - else: - hint = ( - "could not extract a ToolCall impact from the " - "live arguments. The @sensitive tool's kwargs " - "could not be validated for wire emission " - "(unsupported types or invalid param keys)." - ) - else: - # Defensive fallback for a future extractor type - # that doesn't update this hint. - hint = ( - "could not extract business_impact from the live " - "arguments. Check the @sensitive decorator's " - "`impact=...` argument." - ) - err = NullRunBlockedException( - workflow_id=workflow_id, - reason=( - f"failed to extract business_impact for sensitive tool {fn.__name__!r}: {exc}" - ), - tool_name=fn.__name__, - error_code="NR-B003", - user_action=(f"The @sensitive decorator on {fn.__name__!r} {hint}"), - ) - runtime._emit_sdk_error( - err, - stage="sensitive_tool_extract", - workflow_id=workflow_id, - tool_name=fn.__name__, - ) - raise NullRunBlockedException( - workflow_id=workflow_id, - reason=err.reason, - tool_name=fn.__name__, - error_code="NR-B003", - user_action=err.user_action, - ) from exc + action_digest_hex: str = compute_action_digest(no_impact) - # ADR-008: prefer `on_transport_error` (raise classified from nullrun.breaker.exceptions import ( NullRunBlockedException, - NullRunDecision, # DEF-NR-TRANSPORT-CATCHFANIN-GAP (2026-09-10): umbrella arm - NullRunExecutionNotFoundError, # DEF-NR-EX01-REWRAP-LOSS (2026-09-10): pass-through arm - NullRunInfrastructureError, # DEF-NR-TRANSPORT-CATCHFANIN-GAP (2026-09-10): umbrella arm + NullRunDecision, + NullRunExecutionNotFoundError, + NullRunInfrastructureError, NullRunTransportError, - RateLimitError, # DEF-NR-R001-REWRAP-LOSS (2026-09-10): pass-through arm + RateLimitError, TransportErrorSource, ) fail_open = os.environ.get("NULLRUN_SENSITIVE_FAIL_OPEN", "").strip() == "1" - # DEFS-SDKEXEC-WORKFLOW-LABEL (2026-09-08): resolve the # *display* workflow_id via the runtime's precedence chain - # (contextvar → self.workflow_id → None) so the label reflects - # what the SDK actually sends on the wire (the API key's bound - # workflow, when the user hasn't explicitly opened a - # ``with workflow(...)`` block). Pre-fix this read only the - # contextvar; on every API-key-bound key without an explicit - # workflow block the displayed label was the literal sentinel - # ``"__nullrun_unknown__"``, which misleads operators reading - # the trace and the block message into thinking the gate was - # unable to identify the workflow. Sentinel stays as the last - # resort for legacy / never-bound keys. + # (contextvar → self.workflow_id → None). Sentinel stays as the + # last resort for legacy / never-bound keys. workflow_id = runtime._resolve_workflow_id(get_workflow_id()) or UNKNOWN_WORKFLOW_ID try: @@ -962,12 +813,6 @@ def _enforce_sensitive_tool( # returning a synthetic dict. The arm below converts the # typed error into NullRunBlockedException so the caller's # `except NullRunBlockedException` catches it uniformly. - # - # Thread the typed impact + digest through. When the - # decorator did NOT see an extractor, both are None and the - # runtime.execute() drops them from the payload; the - # backend then uses the approval_id-only grant consume - # (the legacy approval_id-only fallback). result = runtime.execute( fn.__name__, {"args": masked_args, "kwargs": masked}, @@ -977,58 +822,21 @@ def _enforce_sensitive_tool( tools=get_call_tools(), ) except NullRunExecutionNotFoundError: - # DEF-NR-EX01-REWRAP-LOSS (2026-09-10): pass-through arm. - # NullRunExecutionNotFoundError IS a NullRunTransportError - # (via NullRunBackendError -> NullRunTransportError), so the - # generic arm below would rewrap it as - # NullRunBlockedException(NR-B00X) and destroy the typed - # class + NR-EX01 catalog line. Cookbook code (and - # langgraph_openai_approval_demo.py) must be able to - # ``except NullRunExecutionNotFoundError`` for the - # documented regate_required=True recovery path. Re-raise - # BEFORE the NullRunBlockedException arm so the typed - # exception propagates unchanged. raise except RateLimitError: - # DEF-NR-R001-REWRAP-LOSS (2026-09-10): pass-through arm. - # RateLimitError IS a NullRunTransportError (its parent - # class) raised with source=GATEWAY_ERROR on a 429 wire - # response (RATE_LIMIT_EXCEEDED). Pre-fix the generic - # ``except NullRunTransportError as exc:`` arm below - # rewrote every TransportError as - # ``NullRunBlockedException(error_code="NR-B002", - # reason="policy engine unavailable: GATEWAY_ERROR")`` — - # losing ``exc.retry_after`` (gateway's Retry-After / - # ``retry_after_ms`` body field converted to seconds), - # ``exc.upgrade_url`` (plan-upgrade URL from 429 body), - # and ``exc.body`` (parsed 429 envelope). Cookbook code - # ``except RateLimitError`` would never match because the - # rewrap stripped the typed class. The user-facing - # catalog line also lost: NR-B002 says "Our service is - # temporarily unavailable. Please try again shortly." - # when the correct NR-R001 says "The NullRun backend - # rate-limited this API key. Wait ``retry_after`` seconds - # (or upgrade the plan) before retrying." Re-raise BEFORE - # the NullRunBlockedException arm so the typed exception - # propagates with error_code=NR-R001, retry_after, - # upgrade_url, and body intact. raise except NullRunBlockedException: # Real policy-block decision from the gateway — propagate as-is. raise except NullRunTransportError as exc: - # ADR-008: classified transport failure. Re-raise as + # ADR-008: classified transport failure. if fail_open: logger.warning( - f"sensitive tool pre-check unavailable for {fn.__name__!r}: " - f"{exc.source} on /{exc.endpoint}. NULLRUN_SENSITIVE_FAIL_OPEN=1 — body will run." + f"tool policy gate unavailable for {fn.__name__!r}: " + f"{exc.source} on /{exc.endpoint}. " + f"NULLRUN_SENSITIVE_FAIL_OPEN=1 — body will run." ) return - # Layer 1: stamp the source-specific error code so the - # caller can distinguish "backend is down" from "we tripped - # the local circuit breaker". Both are retryable in the - # sense that the body will run when the policy engine - # recovers, but the body still MUST NOT run now (fail-CLOSED). _code = { TransportErrorSource.NETWORK_ERROR: "NR-B001", TransportErrorSource.GATEWAY_ERROR: "NR-B002", @@ -1042,90 +850,36 @@ def _enforce_sensitive_tool( error_code=_code, user_action=( f"The NullRun policy engine is unreachable " - f"({exc.source.value}). The body of @sensitive " + f"({exc.source.value}). The body of " f"'{fn.__name__}' did NOT run (fail-CLOSED). " f"Set NULLRUN_SENSITIVE_FAIL_OPEN=1 to opt out for " f"tests / staging — production should leave it off." ), ) - # Layer 2: fire the on_error hook. The sensitive-tool - # path is where a transport failure becomes a hard - # deny — observability hooks should see it even if the - # user's except clause swallows the exception. runtime._emit_sdk_error( err, - stage="sensitive_tool", + stage="tool_policy_gate", workflow_id=workflow_id, tool_name=fn.__name__, extra={"transport_source": exc.source.value}, ) raise err from exc except NullRunDecision: - # DEF-NR-A003-REWRAP-LOSS (2026-09-10, broader scope): - # umbrella pass-through for typed Decision subclasses that - # reach here without hitting NullRunBlockedException (this - # decorator's natural block path) or NullRunTransportError - # (the generic rewrap above). Specifically: - # - NullRunChainError (NR-CH001) — chain lifetime / - # cross-org / Execution Graph parent-lineage - # rejections. Needs exc.chain_id, - # exc.parent_execution_id, exc.backend_code preserved. - # - NullRunWorkflowInactiveError (NR-W004) — soft-deleted - # workflow. Needs exc.workflow_id preserved. - # - NullRunConsumeOverbudgetError (NR-O001) — invariant - # violation. Needs exc.execution_id, - # exc.reserved_cents, exc.max_allowed_cents, - # exc.actual_cost_cents preserved. - # - WorkflowPausedException (NR-W003) — needs - # exc.workflow_id, exc.reason, exc.resume_after. - # Pre-fix the catch-all rewrap below stamped error_code - # NR-B001 on these and discarded every first-class - # attribute, blocking the cookbook recovery path for - # each. Re-raise BEFORE the catch-all to preserve the - # typed instance. + # DEF-NR-TRANSPORT-CATCHFANIN-GAP umbrella pass-through: + # NullRunChainError, NullRunWorkflowInactiveError, + # NullRunConsumeOverbudgetError, WorkflowPausedException — + # preserve first-class attributes for cookbook recovery. raise except NullRunInfrastructureError: - # DEF-NR-A003-REWRAP-LOSS (2026-09-10, broader scope): - # umbrella pass-through for typed Infrastructure - # subclasses that don't match NullRunBackendError, - # NullRunAuthenticationError, or NullRunTransportError - # above. Specifically: - # - NullRunAuthError (NR-A003) — typed 401 envelope. - # Needs exc.wire_code (API_KEY_REVOKED / - # API_KEY_EXPIRED / API_KEY_DISABLED / - # API_KEY_INVALID / API_KEY_MISSING / - # API_KEY_MALFORMED per v3.38) preserved so ops - # can branch on granular lifecycle state. The - # transport fan-in (transport.py:1294) already - # preserves this via NullRunAuthenticationError - # pass-through, but a refactor that reorders the - # transport arms would surface this here. - # - NullRunProtocolError (NR-P001) — wire-protocol - # mismatch. Needs the catalog line "Upgrade the SDK - # to a version that supports protocol - # X-NULLRUN-PROTOCOL: 4" to reach the cookbook. - # - NullRunRateLimitRedisError (NR-R002) — Redis - # outage for aggregate rate limit (fail-CLOSED). - # Needs the catalog line that distinguishes "Redis - # is down" from generic NR-B002. - # - NullRunConfigError (NR-Cxxx) — malformed config, - # typically surfaced by runtime.execute with bad - # env. Never rewrap a config error as a transient - # transport block — that's misleading. - # Pre-fix the catch-all stamped error_code NR-B001 on - # these and discarded wire_code (AuthError), - # protocol-version info (ProtocolError), and Redis - # source-of-failure (RateLimitRedisError). Re-raise - # BEFORE the catch-all. + # DEF-NR-TRANSPORT-CATCHFANIN-GAP umbrella pass-through: + # NullRunAuthError, NullRunProtocolError, + # NullRunRateLimitRedisError, NullRunConfigError — preserve + # first-class attributes. raise except Exception as exc: # noqa: BLE001 - # Any other exception is a transport / network / backend - # failure. Re-raise as NullRunBlockedException so the caller - # sees a uniform "this tool was denied" signal — they should - # not need to also catch httpx.ConnectError or similar. if fail_open: logger.warning( - f"sensitive tool pre-check unavailable for {fn.__name__!r}: " + f"tool policy gate unavailable for {fn.__name__!r}: " f"{exc}. NULLRUN_SENSITIVE_FAIL_OPEN=1 — body will run." ) return @@ -1136,24 +890,25 @@ def _enforce_sensitive_tool( error_code="NR-B001", user_action=( f"The NullRun policy engine raised an unexpected " - f"exception during the @sensitive pre-check of " + f"exception during the @protect pre-check of " f"'{fn.__name__}'. The body did NOT run. Check the " f"chained exception (raise ... from exc) for the " f"root cause." ), ) - # Layer 2: emit for the generic exception path too. - # (The NullRunTransportError path above already emits - # this covers the catch-all ``except Exception`` arm.) runtime._emit_sdk_error( err, - stage="sensitive_tool", + stage="tool_policy_gate", workflow_id=workflow_id, tool_name=fn.__name__, ) raise err from exc - # Defense in depth (ADR-008 Rule 1 + Rule 2): if `runtime.execute` + # Defense in depth: legacy fallback / classification audit. If + # the transport ever returns a synthetic dict whose + # decision_source marks a fallback, block per ADR-008 fail-CLOSED. + # This arm is preserved for defense in depth even though the + # typed transport-error arms above are the canonical path. if isinstance(result, dict): decision_source = result.get("decision_source", "") if isinstance(decision_source, str) and ( @@ -1168,15 +923,10 @@ def _enforce_sensitive_tool( ): if fail_open: logger.warning( - f"sensitive tool pre-check for {fn.__name__!r} returned " + f"tool policy gate for {fn.__name__!r} returned " f"{decision_source}; NULLRUN_SENSITIVE_FAIL_OPEN=1 — body will run." ) return - # Layer 1: stamp the source-specific code on the - # fallback block so cookbook code can distinguish - # between "the policy engine said block" (NR-T001 etc.) - # and "we blocked because the policy engine never - # answered" (NR-B001/B002). _code = { "NETWORK_ERROR": "NR-B001", "GATEWAY_ERROR": "NR-B002", @@ -1190,19 +940,15 @@ def _enforce_sensitive_tool( error_code=_code, user_action=( f"The NullRun policy engine returned a fallback " - f"({decision_source}) for @sensitive '{fn.__name__}'. " - f"The body did NOT run. Retry once the policy engine " - f"is back — or set NULLRUN_SENSITIVE_FAIL_OPEN=1 for " - f"tests / staging." + f"({decision_source}) for '{fn.__name__}'. The " + f"body did NOT run. Retry once the policy engine " + f"is back — or set NULLRUN_SENSITIVE_FAIL_OPEN=1 " + f"for tests / staging." ), ) - # Layer 2: emit the on_error hook with the fallback - # source as extra metadata so Sentry rules can - # distinguish "policy engine is down" from "we - # tripped the local circuit breaker". runtime._emit_sdk_error( err, - stage="sensitive_tool", + stage="tool_policy_gate", workflow_id=workflow_id, tool_name=fn.__name__, extra={"decision_source": decision_source}, @@ -1215,215 +961,6 @@ def _enforce_sensitive_tool( # (the happy path) just falls through and the body runs. -def sensitive( - fn: F | None = None, - *, - impact: Any = None, -) -> F: - """ - Mark a function as sensitive. `@protect` will pre-check - `runtime.execute(...)` before the body runs. - - .. deprecated:: - Bare ``@sensitive`` is deprecated as of SDK 0.18.1. Since - ``@protect`` now auto-attaches the same default tool_params - extractor that bare ``@sensitive`` used to install, and since - the business interpretation of those params belongs to NullRun - policy (not the SDK), the canonical pattern is now just - ``@protect``. Bare ``@sensitive`` still works in 0.18.x with a - ``DeprecationWarning`` and the legacy behaviour will be - removed in 0.19.x. - - The ``@sensitive(impact=...)`` factory form remains supported - as an explicit advanced API: it attaches a typed extractor - (``money_outflow(...)`` or a custom ``ToolParamsExtractor`` - map) and registers the tool for the server-side policy - path. New code does not need it; library authors wiring - approval rules into a custom runtime may still prefer it. - - This is the discoverable alternative to the lower-level - `runtime.add_sensitive_tool(fn.__name__)`. Chain with `@protect` - in either order (both work via `functools.wraps`); the - recommended form is `@sensitive` outside so the name is - registered before the wrapper is built: - - @nullrun.sensitive - @nullrun.protect - def charge_card(amount: int) -> str: - ... - - ``@sensitive(impact=money_outflow(...))`` attaches a typed - ``MoneyImpactExtractor`` to the function via the - ``_nullrun_extractor`` attribute. The wrapper reads it inside - ``_enforce_sensitive_tool`` to extract a typed - ``BusinessImpact`` + ``action_digest`` from the live call - arguments and forward them to /execute, so the backend can - stamp the approval row with the digest and refuse tampered - payloads on the post-approval re-check. - - @nullrun.sensitive(impact=money_outflow(argument="amount_cents")) - @nullrun.protect - def refund_customer(amount_cents: int, customer_id: str): - ... - - Args: - fn: the function to decorate. May be None when used with - keyword arguments (the ``@sensitive(impact=...)`` form). - impact: typed action extractor. Currently only - ``MoneyImpactExtractor`` (returned by - ``money_outflow(argument=...)``) is supported. - - Two forms are accepted: - - bare: ``@sensitive`` — fn must be the function being decorated. - **Deprecated** as of 0.18.1; emits ``DeprecationWarning``. - - factory: ``@sensitive(impact=...)`` — fn is None, returns a - decorator that closes over ``impact``. Still supported as - an advanced API. - - Both forms register the tool as sensitive in the runtime so the - ``_enforce_sensitive_tool`` pre-check fires. - """ - # Factory form: @sensitive(impact=...) returns a decorator that - if fn is None: - - def _attach_decorator(_fn: F) -> F: - if impact is not None: - _stamp_extractor_on_innermost(_fn, impact) - return _do_sensitive_register(_fn) - - return _attach_decorator # type: ignore[return-value] - - # Bare form: @sensitive. - # 0.18.1: bare `@sensitive` is deprecated. `@protect` already - # auto-attaches a default ToolParamsExtractor (see protect() above), - # so the bare form is a duplicate of capability that the user can - # get by writing just `@protect`. We keep the old behaviour - # (auto-attach + sensitive-tool registration) intact so this is a - # warning-only release; the special behaviour will be removed in - # 0.19.x. Users who need the sensitive-tool registration (which - # short-circuits to the server-side policy path) should switch to - # explicit ``@protect`` and call ``runtime.add_sensitive_tool(...)`` - # in their app bootstrap. - import warnings - - warnings.warn( - "Bare `@sensitive` is deprecated as of SDK 0.18.1: `@protect` " - "now auto-attaches the same default tool_params extractor, and " - "the business interpretation of those params belongs to NullRun " - "policy, not the SDK. Remove the bare `@sensitive` and rely on " - "`@protect` alone. The legacy behaviour will be removed in 0.19.", - DeprecationWarning, - stacklevel=2, - ) - if impact is not None: - _stamp_extractor_on_innermost(fn, impact) - return _do_sensitive_register(fn) - - -# Maximum depth for ``__wrapped__`` chain walks. The real chain is -# at most 3 deep (@sensitive factory + @protect + functools.wraps -# from @protect); the cap defends against pathological cycles. -_WRAPPED_CHAIN_MAX_HOPS = 32 - - -def _walk_wrapped_chain(fn: Any) -> Any: - """Yield each callable in ``fn``'s ``__wrapped__`` chain. - - Stops on ``None``, on a cycle (id already seen), or at - ``_WRAPPED_CHAIN_MAX_HOPS`` hops. The original ``fn`` is - always yielded first. - """ - seen: set[int] = set() - current: Any = fn - for _ in range(_WRAPPED_CHAIN_MAX_HOPS): - if current is None or id(current) in seen: - return - seen.add(id(current)) - yield current - current = getattr(current, "__wrapped__", None) - - -def _stamp_extractor_on_innermost(fn: F, impact: Any) -> None: - """Stamp ``_nullrun_extractor`` on the innermost callable in the chain. - - Setting the attribute on the innermost callable means the gate's - ``_enforce_sensitive_tool`` can read it from the bare user function - via a single ``getattr`` call — no chain walk needed. - """ - last: Any = None - for current in _walk_wrapped_chain(fn): - last = current - target = last if last is not None else fn - # `setattr` keeps mypy happy without a TYPE_CHECKING - # forward-reference declaration; ruff B010 is a stylistic - # preference (no functional risk here). - setattr(target, "_nullrun_extractor", impact) # noqa: B010 - - -def _find_extractor_in_chain(fn: Any) -> Any: - """Walk ``fn.__wrapped__`` looking for a stamped extractor. - - Used by ``_do_sensitive_register`` to detect an explicit - ``impact=tool_params({...})`` (or ``impact=money_outflow(...)``) - that was already stamped on the bare function by the - ``@sensitive`` factory form. Without the chain walk the - auto-attach path would see ``None`` on the @protect wrapper - and silently stamp its default ToolParamsExtractor on top, - breaking the user's explicit map. - """ - for current in _walk_wrapped_chain(fn): - ext = getattr(current, "_nullrun_extractor", None) - if ext is not None: - return ext - return None - - -def _do_sensitive_register(fn: F) -> F: - # If @sensitive was applied bare (no impact=...), auto-attach a - try: - from nullrun.extractor import ToolParamsExtractor, tool_params - - # Walk the __wrapped__ chain in case the explicit extractor - # was stamped on the bare function (by - # ``_stamp_extractor_on_innermost``) and we received the - # @protect-wrapped outer function as ``fn``. Without the - # chain walk, the auto-attach would silently overwrite - # the explicit extractor and break the user's - # ``impact=tool_params({...})`` map. - if _find_extractor_in_chain(fn) is None: - _stamp_extractor_on_innermost(fn, tool_params(include_all=True)) - except ImportError: - # The extractor module is loaded above us on every path - # we care about; this ImportError guard is defensive in - # case the SDK is shrunk (e.g. for a hypothetical - # tool-only build). Falling back to the legacy - # approval_id-only grant consume is the safe default -- - # the wire payload drops the business_impact field and the - # backend uses approval_id-only grant consume. - pass - - try: - # Use the same slot the @protect wrapper uses so the - # registration lands on the same runtime instance the - # wrapper will consult. Falling back to get_runtime - # would hit a different singleton and silently no-op in - # tests that build a custom runtime. - rt = _get_or_create_runtime() - rt.add_sensitive_tool(fn.__name__) - # 2026-07-24 (Root-cause fix): the runtime singleton - from nullrun.runtime import register_strict_mode_forced - - register_strict_mode_forced(fn.__name__) - except Exception as exc: - # Sensitive tool registration is part of the fail-CLOSED contract - raise RuntimeError( - f"@sensitive registration failed for {fn.__name__!r}: {exc}. " - "Cannot proceed without runtime; tool will be blocked until " - "NullRun initializes correctly." - ) from exc - return fn - - def reset() -> None: """ Reset NullRun runtime. Mainly for testing or when you need to diff --git a/src/nullrun/extractor.py b/src/nullrun/extractor.py deleted file mode 100644 index 6fbae74..0000000 --- a/src/nullrun/extractor.py +++ /dev/null @@ -1,1157 +0,0 @@ -"""BusinessImpact extraction — advanced API for @sensitive(impact=...). - -This module is the SDK-side counterpart of the backend's -``BusinessImpact`` discriminated union. It exposes a single -declarative API (``money_outflow(argument="...", units="...")``) -that: - -1. Binds the SDK call's positional/keyword arguments using - ``inspect.signature(...).bind(...)`` so positional and keyword - invocations look identical. -2. Pulls the named argument off the bound args. -3. Validates and converts the value to integer minor units - using the ``units`` discriminator, the ISO-4217 minor-unit - exponent for the currency, and the per-currency business cap - for agent safety. -4. Validates and builds a ``MoneyImpact``. -5. Computes the byte-identical ``action_digest`` the backend - expects (see ``nullrun.business_impact.compute_action_digest``). - -SDK 0.18.1: the canonical public entry point is ``@protect`` — -it auto-attaches a default ``ToolParamsExtractor`` on every -protected function so the wire payload carries -``tool_name + params`` without any second decorator. Bare -``@sensitive`` is deprecated. This module exists for the -``@sensitive(impact=...)`` advanced API: library authors who need -a typed ``BusinessImpact`` envelope (money flows, custom predicate -maps) and the SHA-256 ``action_digest`` for digest-bound approval. - -## Why this is its own helper, not part of ``@sensitive`` - -The ``@sensitive(impact=...)`` decorator chain is the integration -point, but the per-call impact extraction is data-driven and tested -independently. Keeping ``extractor.py`` as a pure helper avoids -the ``inspect.signature()`` cost on every sensitive call (the -binding result is cached after first extraction via Python's -``lru_cache``-friendly design) and makes the unit-discriminator -test matrix cheap to write without instantiating the full -``NullRunRuntime``. - -For the production flow, ``runtime.execute(...)`` reads the -extractor from the function's ``_nullrun_extractor`` attribute -(which ``@sensitive(impact=money_outflow(...))`` sets) and calls -``impact_for(...)`` automatically. - -## Why ``units`` is explicit, not a type discriminator - -The previous review explicitly rejected the -``int = minor, Decimal = major`` shortcut because the unit -semantics of a function argument should not flip silently when -the function signature is refactored. Concretely: - - @nullrun.sensitive(impact=nullrun.money_outflow(argument="amount")) - @nullrun.protect - def refund(amount: int) -> ... # 50 = 50 cents (minor units) - def refund(amount: Decimal) -> ... # 50 = $50.00 (5000 cents) - -If ``units`` were implicit-from-type, renaming ``amount``'s -annotation from ``int`` to ``Decimal`` would silently change the -operator-facing rule from "$0.50" to "$50.00". The explicit -``units="major" | units="minor"`` argument in the decorator -fixes the unit semantics at the call site so a future -signature refactor does not flip the meaning. - -## Float is rejected outright - -``Decimal`` exists precisely so that money code does not have -to deal with binary-floating-point surprises (``0.1 + 0.2 != -0.3`` in IEEE-754). The extractor therefore refuses ``float`` -values at the input level. The error includes a pointer to -the right alternative (``Decimal`` for major, ``int`` for minor) -so the operator can fix the call site without guessing. - -## Major-unit precision is validated, never rounded - -The first version of this module used banker's rounding -(``ROUND_HALF_EVEN``) to convert ``Decimal("50.99")`` to -``5099`` minor units. That decision was rejected in review: -banker's rounding silently drops sub-cent precision -(``Decimal("50.005")`` becomes ``5000`` minor units), which -is the exact bug class the explicit ``units`` discriminator -is designed to prevent. The current contract validates the -precision of the ``Decimal`` against the ISO-4217 minor-unit -exponent for the currency and raises ``InvalidMoneyPrecisionError`` -if the caller supplied more precision than the currency -supports. The caller can explicitly truncate with -``value.quantize(Decimal('1E-N'))`` to opt in to rounding; the -SDK never rounds silently. - -## Sign is validated - -A negative amount for either ``money_outflow`` (debit) or -``money_inflow`` (credit) is semantically incoherent. The -review pointed out that ``{"direction":"outflow", -"amount_minor":-5000}`` would silently fall through every -``op=gt`` predicate because ``-5000 > 5000`` is always False, -and the operator would never see a block. The current contract -rejects negative amounts with ``InvalidMoneyAmountError`` so the -``@protect`` wrapper can fail-CLOSED on the call site. If a -future variant needs negative amounts (e.g. refunds as negative -outflows) it can opt in via a future ``units="signed"`` -discriminator. - -## Overflow is bounded - -``i64`` can hold up to ``2**63 - 1 = 9_223_372_036_854_775_807`` -minor units (about $9.2 \u00d7 10\u00b9\u2076 for USD). The extractor checks -the converted value against this limit and raises -``InvalidMoneyAmountError`` if it would overflow. The check -uses ``int`` post-conversion so the operator sees the -offending amount, not just "too large". - -## Business cap is bounded - -The wire-format ``i64`` limit is a few hundred quadrillion -dollars, which is well above any sensible per-call debit. The -business cap (``_BUSINESS_CAP_MINOR`` table) is a much smaller -per-currency limit chosen so that any amount above the cap -goes through a separate risk path rather than being treated -as a normal call. The cap is policy, not correctness: a $1M -USD debit is technically valid on the wire, but for an agent -running a refund tool it almost certainly warrants a human -review. The cap is enforced as ``InvalidMoneyAmountError(reason="excessive")`` -with a clear "above the per-call business cap" message; the -``@protect`` wrapper upgrades the error to fail-CLOSED. - -## Float and ``bool`` are rejected - -``float`` is rejected because IEEE-754 surprises are the entire -reason ``Decimal`` exists. ``bool`` is rejected because ``bool`` -is a subclass of ``int`` in Python; without the explicit check, -``refund(amount=True)`` would silently treat ``True`` as -``1`` cent. - -## Currency is validated (whitelist + case) - -ISO-4217 minor-unit exponent lookup covers a small set of -codes by design. The ``normalize_currency`` helper rejects any -input that is not a 3-letter uppercase ISO-4217 code (e.g. -``"usd"``, ``"Usd"``, ``"USDX"``, ``""`` raise -``InvalidCurrencyError``). The SDK does NOT silently -upper-case the input because: - -- it would hide typos (``"usd"`` vs ``"USD"`` vs ``"Usd"`` - would all normalize to ``"USD"``, masking a typo in the - call site); -- ISO-4217 is a closed set of 3-letter uppercase codes, - anything else is wrong by definition; -- the error message names the offending input so the operator - can fix the call site. - -The whitelist is consulted by ``currency_minor_digits`` and -``business_cap_minor``; unknown codes are rejected with -``InvalidCurrencyError`` instead of falling back to a default. -This closes the conservative-fallback gap from the previous -hardening pass (``UNKNOWN`` was allowed but the operator -might never notice the typo). - -## Currency case rejection is enforced at construction time - -The ``MoneyImpactExtractor.__init__`` validates the currency -via ``normalize_currency``. Passing ``"usd"`` raises -``InvalidCurrencyError`` at decorator-application time, before -the tool is ever called. This is fail-CLOSED: a misconfigured -decorator never reaches runtime. -""" - -from __future__ import annotations - -import inspect -import logging -from collections.abc import Callable -from decimal import Decimal, InvalidOperation -from typing import Any - -from nullrun.business_impact import ( - INFLOW, - OUTFLOW, - BusinessImpact, - MoneyImpact, - ToolCallParams, - compute_action_digest, -) - -# Unit discriminators for the typed impact payload. -# -# ``minor`` = the value is already in minor units (cents, pence, -# satoshi-style). The SDK stores the value verbatim on the wire. -# This is the pre-Decimal path: a function declared -# ``def refund(amount_cents: int)`` already works in minor units -# so the operator just has to add the decorator and the wire -# shape does not change. -# -# ``major`` = the value is in major units (dollars, pounds, etc.) -# and the SDK converts to minor units via ``Decimal * 10**N`` -# where ``N = currency_minor_digits(currency)``. The -# ``MoneyImpact`` struct stores the result in minor units so -# the wire shape and the backend ``action_predicate`` shape -# are identical between the two paths. -# -# The discriminator is **explicit** in the decorator (not -# implicit from the type) so the unit semantics survive a -# refactor of the function signature. -UNIT_MINOR = "minor" -UNIT_MAJOR = "major" -UNITS = (UNIT_MINOR, UNIT_MAJOR) - - -# ISO-4217 minor-unit exponents for the currencies the SDK -# supports out of the box. The lookup is consulted by -# ``_to_minor_units`` to validate the precision of a -# ``Decimal`` in ``units="major"`` mode; a value with more -# fractional digits than the currency supports is a bug, not -# a rounding opportunity, and the SDK surfaces it as an -# ``InvalidMoneyPrecisionError`` so the operator can decide -# explicitly. -# -# Coverage is small by design: the SDK only enforces precision -# for currencies the form / wire shape already understands -# (USD/EUR/GBP/CHF/CAD/AUD = 2 fractional digits, JPY = 0, -# KWD/BHD/OMR = 3). Adding a new currency to the wire contract -# is a one-line change in ``_CURRENCY_MINOR_DIGITS`` *and* -# ``_BUSINESS_CAP_MINOR``. -_CURRENCY_MINOR_DIGITS = { - # 2 fractional digits (cents, pence, centimes) - "USD": 2, - "EUR": 2, - "GBP": 2, - "CHF": 2, - "CAD": 2, - "AUD": 2, - # 0 fractional digits (yen) - "JPY": 0, - # 3 fractional digits (fils) - "KWD": 3, - "BHD": 3, - "OMR": 3, -} - - -# Per-currency business cap (in minor units). Above this -# threshold the extractor raises ``InvalidMoneyAmountError`` -# with ``reason="excessive"`` so the call goes through a -# separate risk path rather than being treated as a normal -# call. The cap is policy, not correctness: a $1M USD debit -# is technically valid on the wire (well within ``i64``), but -# for an agent running a refund tool it almost certainly -# warrants a human review. -# -# Caps are chosen as round numbers above any plausible single -# transaction but well below the wire-format ``i64`` limit so -# the ``@protect`` wrapper can branch on -# ``reason="excessive"`` without confusing it with -# ``reason="overflow"`` (a real wire-format overflow). -# -# To opt out of the cap on a per-extractor basis, set -# ``enforce_business_cap=False`` in ``MoneyImpactExtractor.__init__``. -_BUSINESS_CAP_MINOR = { - # $1,000,000.00 USD per call (one million dollars) - "USD": 100_000_000, - "EUR": 100_000_000, - "GBP": 100_000_000, - "CHF": 100_000_000, - "CAD": 100_000_000, - "AUD": 100_000_000, - # 100,000,000 JPY (one hundred million yen) - "JPY": 100_000_000, - # 100,000.000 KWD / BHD / OMR (one hundred thousand, three - # decimal digits each) - "KWD": 100_000_000, - "BHD": 100_000_000, - "OMR": 100_000_000, -} - - -# Hard upper bound for the converted ``amount_minor``. The -# wire format is ``i64``; values exceeding ``2**63 - 1`` would -# silently overflow on the backend side. The constant is -# checked AFTER conversion so the operator sees the offending -# amount, not just "too large". -_I64_MAX = (1 << 63) - 1 - - -# ----- Dedicated error types -------------------------------------- -# -# ``InvalidMoneyPrecisionError`` -- the caller supplied more -# fractional digits than the currency supports (e.g. -# ``Decimal("50.005")`` for USD). The error carries -# ``currency``, ``allowed``, ``received`` so a UI or test -# harness can format a specific message. -# -# ``InvalidMoneyAmountError`` -- generic money-amount -# invariant violation: negative amounts, overflow, -# non-finite Decimals, or amounts above the per-currency -# business cap. The error carries ``currency`` (when known) -# and ``reason`` (a string discriminator). -# -# ``InvalidCurrencyError`` -- the supplied currency is not a -# 3-letter uppercase ISO-4217 code the SDK supports. The -# error carries the offending input so the operator can fix -# the call site. -# -# All three inherit from ``ValueError`` so the existing -# ``except ValueError`` callers in ``runtime.py`` continue to -# work; the subclasses let a careful caller branch on the -# type. - - -class InvalidMoneyPrecisionError(ValueError): - """Sub-precision rejected: the supplied Decimal has more - fractional digits than the currency supports. - - Attributes: - currency: the ISO-4217 code the extractor was called - with. - allowed: the number of fractional digits the currency - supports (e.g. 2 for USD). - received: the offending Decimal as a string (so the - caller sees exactly what was passed). - received_digits: the number of fractional digits the - offending Decimal actually had. - """ - - def __init__( - self, - currency: str, - allowed: int, - received: str, - received_digits: int, - ) -> None: - self.currency = currency - self.allowed = allowed - self.received = received - self.received_digits = received_digits - msg = ( - f"{currency} supports at most {allowed} fractional " - f"digit(s); got {received} ({received_digits}). " - f"Either truncate explicitly with " - f"``value.quantize(Decimal('1E-{allowed}'))`` " - f"before passing to money_outflow, or change currency." - ) - super().__init__(msg) - - -class InvalidMoneyAmountError(ValueError): - """Generic money-amount invariant violation. - - Attributes: - currency: the ISO-4217 code the extractor was called - with (may be empty if the error happened before - currency dispatch). - reason: short string discriminator (``"negative"``, - ``"overflow"``, ``"non_finite"``, ``"excessive"``). - Lets a UI or test harness branch without parsing - the message. - """ - - def __init__( - self, - reason: str, - detail: str, - currency: str = "", - ) -> None: - self.reason = reason - self.currency = currency - msg = detail if not currency else f"[{currency}] {detail}" - super().__init__(msg) - - -class InvalidCurrencyError(ValueError): - """The supplied currency is not a 3-letter uppercase - ISO-4217 code the SDK supports. - - Attributes: - received: the offending currency string (so the - operator sees exactly what was passed). - """ - - def __init__(self, received: str, detail: str) -> None: - self.received = received - msg = f"currency={received!r}: {detail}" - super().__init__(msg) - - -def normalize_currency(currency: str) -> str: - """Validate and return the ISO-4217 currency code. - - The SDK does NOT silently upper-case the input because: - - - it would hide typos (``"usd"`` vs ``"USD"`` vs ``"Usd"`` - would all normalize to ``"USD"``, masking a typo in the - call site); - - ISO-4217 is a closed set of 3-letter uppercase codes, - anything else is wrong by definition; - - the error message names the offending input so the - operator can fix the call site. - - Raises ``InvalidCurrencyError`` for any input that is not - a 3-letter uppercase ISO-4217 code the SDK supports. - """ - if not isinstance(currency, str): - raise InvalidCurrencyError( - str(currency), - "currency must be a string", - ) - if len(currency) != 3: - raise InvalidCurrencyError( - currency, - f"currency must be a 3-letter ISO-4217 code; got length {len(currency)}", - ) - if not currency.isupper() or not currency.isalpha(): - raise InvalidCurrencyError( - currency, - "currency must be 3 uppercase ASCII letters (ISO-4217)", - ) - if currency not in _CURRENCY_MINOR_DIGITS: - raise InvalidCurrencyError( - currency, - f"currency is not in the supported ISO-4217 whitelist " - f"(supported: {sorted(_CURRENCY_MINOR_DIGITS.keys())})", - ) - return currency - - -def currency_minor_digits(currency: str) -> int: - """Return the number of fractional digits for ``currency``. - - Calls ``normalize_currency`` so the caller cannot pass an - unknown code; previously this function silently fell back - to 2 digits for unknown codes, which masked typos like - ``"USDX"`` or ``"usd"``. - - Raises ``InvalidCurrencyError`` for any input that is not - in the whitelist. - """ - return _CURRENCY_MINOR_DIGITS[normalize_currency(currency)] - - -def business_cap_minor(currency: str) -> int: - """Return the per-call business cap (in minor units) for ``currency``. - - The cap is policy, not correctness: a debit at the cap is - technically valid on the wire but should go through a - separate risk path. Callers that need to opt out (e.g. - batch settlement tools) can pass - ``enforce_business_cap=False`` to ``MoneyImpactExtractor``. - - Raises ``InvalidCurrencyError`` for any input that is not - in the whitelist. - """ - return _BUSINESS_CAP_MINOR[normalize_currency(currency)] - - -def _decimal_has_more_fractional_digits(value: Decimal, allowed: int) -> bool: - """Return True iff ``value`` has more fractional digits than - ``allowed``. - - The check uses ``value % 1`` so that a value like - ``Decimal("50.00")`` (which ``as_tuple()`` reports as having - two fractional digits) is correctly classified as an - integer-valued decimal with zero effective fractional - digits. ``Decimal("50.005")`` has a non-zero fractional - part and is rejected. - """ - if allowed < 0: - raise ValueError(f"allowed fractional digits must be >= 0, got {allowed}") - if not value.is_finite(): - raise InvalidMoneyAmountError( - reason="non_finite", - detail=f"Decimal must be finite, got {value}", - ) - fractional = value - value.to_integral_value(rounding="ROUND_DOWN") - if fractional == 0: - return False - exponent = value.as_tuple().exponent - if isinstance(exponent, int): - return abs(exponent) > allowed - return False - - -def _count_fractional_digits(value: Decimal) -> int: - """Return the number of fractional digits in ``value``.""" - exponent = value.as_tuple().exponent - if isinstance(exponent, int): - return max(0, abs(exponent)) - return 0 - - -def _check_overflow(amount_minor: int, currency: str) -> None: - """Raise ``InvalidMoneyAmountError`` if ``amount_minor`` - exceeds the wire-format ``i64`` upper bound. - """ - if amount_minor > _I64_MAX: - raise InvalidMoneyAmountError( - reason="overflow", - currency=currency, - detail=( - f"amount_minor={amount_minor} exceeds i64::MAX={_I64_MAX}; " - f"either the input amount is too large for the wire " - f"format or the currency conversion factor is wrong." - ), - ) - - -def _check_business_cap(amount_minor: int, currency: str, enforce: bool) -> None: - """Raise ``InvalidMoneyAmountError`` if ``amount_minor`` - exceeds the per-currency business cap. - - The cap is policy, not correctness: a debit at the cap is - technically valid on the wire but should go through a - separate risk path. ``enforce=False`` skips the check for - callers that need to opt out (e.g. batch settlement tools - that already have a human-in-the-loop approval flow). - """ - if not enforce: - return - cap = _BUSINESS_CAP_MINOR[currency] - if amount_minor > cap: - raise InvalidMoneyAmountError( - reason="excessive", - currency=currency, - detail=( - f"amount_minor={amount_minor} exceeds the per-call " - f"business cap={cap} minor units for {currency}; " - f"send the call through the explicit human-approval " - f"path instead of the auto-decision flow." - ), - ) - - -def _to_minor_units( - value: int | Decimal, - units: str, - currency: str, - enforce_business_cap: bool = True, -) -> int: - """Convert a Decimal-or-int value to integer minor units. - - See module docstring for the full contract. The - ``enforce_business_cap`` flag is passed through from - ``MoneyImpactExtractor`` so callers that need to opt out - (batch settlement) can do so without bypassing the rest - of the validation. - """ - if units == UNIT_MINOR: - if isinstance(value, bool) or not isinstance(value, int): - if isinstance(value, Decimal) and not isinstance(value, bool): - # Caller has already pre-quantized; the SDK does - # not change the value. If the Decimal has a - # fractional part (e.g. ``0.05``) we surface a - # TypeError rather than silently truncate. - if _decimal_has_more_fractional_digits(value, 0): - raise TypeError( - f"money_outflow(argument={value!r}, units='minor'): " - f"refusing to round {value!r} to integer minor units; " - f"either pass an int (e.g. int({value!r})) or set units='major'." - ) - converted = int(value) - else: - raise TypeError( - f"money_outflow(units='minor') requires int or Decimal; " - f"got {type(value).__name__}: {value!r}" - ) - else: - converted = value - if converted < 0: - raise InvalidMoneyAmountError( - reason="negative", - currency=currency, - detail=( - f"money_outflow(units='minor') rejected negative " - f"amount {converted!r}; a negative amount would " - f"silently fall through every op=gt predicate " - f"because negative < positive is always False." - ), - ) - _check_overflow(converted, currency) - _check_business_cap(converted, currency, enforce_business_cap) - return converted - - if units == UNIT_MAJOR: - if isinstance(value, bool) or not isinstance(value, Decimal): - raise TypeError( - f"money_outflow(units='major') requires Decimal; " - f"got {type(value).__name__}: {value!r}. " - f"For int minor units, set units='minor' or use money_inflow(...) " - f"with units='minor'." - ) - if value < 0: - raise InvalidMoneyAmountError( - reason="negative", - currency=currency, - detail=( - f"money_outflow(units='major') rejected negative " - f"amount {value!r}; a negative amount would " - f"silently fall through every op=gt predicate " - f"because negative < positive is always False." - ), - ) - allowed = currency_minor_digits(currency) - if _decimal_has_more_fractional_digits(value, allowed): - raise InvalidMoneyPrecisionError( - currency=currency, - allowed=allowed, - received=str(value), - received_digits=_count_fractional_digits(value), - ) - converted = int(value * (Decimal(10) ** allowed)) - _check_overflow(converted, currency) - _check_business_cap(converted, currency, enforce_business_cap) - return converted - - raise ValueError(f"unknown units={units!r}; expected one of: {UNITS}") - - -class MoneyImpactExtractor: - """Declarative money-impact extractor. - - ``units`` discriminator semantics: - - - ``units="minor"`` (default): the bound argument is - already in minor units. ``int`` is the canonical type; - ``Decimal`` is accepted if it is already integer-valued. - ``float`` is rejected outright. - - ``units="major"``: the bound argument is a Decimal in - major units. The SDK converts to minor units via - ``Decimal * 10**currency_minor_digits(currency)`` after - validating that the Decimal's precision matches the - currency's ISO-4217 minor-unit exponent. ``float`` and - ``int`` are rejected outright. - - The ``enforce_business_cap`` flag (default ``True``) gates - the per-currency cap. Set to ``False`` for batch settlement - tools that already have a human-in-the-loop approval flow - and need to bypass the cap. - - The discriminator is **explicit** rather than implicit from - the type. A future signature refactor (``int`` -> ``Decimal`` - or vice versa) does not silently flip the meaning of the - number. Operators reading the code see the unit in the - decorator argument, not in the type annotation. - """ - - def __init__( - self, - argument: str, - direction: str = OUTFLOW, - currency: str = "USD", - units: str = UNIT_MINOR, - extractor_id: str = "nullrun.money.path", - extractor_version: str = "1", - enforce_business_cap: bool = True, - ) -> None: - if direction not in (OUTFLOW, INFLOW): - raise ValueError(f"direction must be {OUTFLOW!r} or {INFLOW!r}, got {direction!r}") - if units not in UNITS: - raise ValueError(f"units must be one of {UNITS}, got {units!r}") - # ``normalize_currency`` raises ``InvalidCurrencyError`` - # if the input is not a 3-letter uppercase ISO-4217 - # code in the whitelist. The constructor fails-CLOSED: - # a misconfigured decorator (``currency="usd"`` typo) - # never reaches runtime. - currency = normalize_currency(currency) - self.argument = argument - self.direction = direction - self.currency = currency - self.units = units - self.extractor_id = extractor_id - self.extractor_version = extractor_version - self.enforce_business_cap = enforce_business_cap - - def impact_for( - self, - fn: Callable[..., Any], - args: tuple[Any, ...], - kwargs: dict[str, Any], - ) -> BusinessImpact: - """Bind the call and pull ``self.argument`` out of the bound args. - - Raises: - TypeError: when ``self.argument`` is not a named - parameter, or when the supplied value is not a - Decimal / int in the unit discriminator the - constructor was called with. - InvalidMoneyPrecisionError: when ``units="major"`` - and the supplied Decimal has more fractional - digits than the currency supports. - InvalidMoneyAmountError: when the supplied amount - is negative, non-finite, exceeds the wire-format - ``i64`` upper bound, or exceeds the per-currency - business cap. - """ - sig = inspect.signature(fn) - try: - bound = sig.bind(*args, **kwargs) - except TypeError as exc: - raise TypeError( - f"MoneyImpactExtractor.argument={self.argument!r} " - f"failed to bind call to {fn!r}: {exc}" - ) from exc - bound.apply_defaults() - if self.argument not in bound.arguments: - raise TypeError( - f"MoneyImpactExtractor expects argument {self.argument!r} " - f"on {fn!r}; call did not provide it" - ) - value = bound.arguments[self.argument] - - amount_minor = _to_minor_units( - value, - units=self.units, - currency=self.currency, - enforce_business_cap=self.enforce_business_cap, - ) - - impact = MoneyImpact( - direction=self.direction, - amount_minor=amount_minor, - currency=self.currency, - extractor_id=self.extractor_id, - extractor_version=self.extractor_version, - ) - impact.validate() - return BusinessImpact(impact=impact) - - -def money_outflow( - argument: str, - currency: str = "USD", - units: str = UNIT_MINOR, - extractor_id: str = "nullrun.money.path", - extractor_version: str = "1", - enforce_business_cap: bool = True, -) -> MoneyImpactExtractor: - """Shorthand constructor used by ``@sensitive(impact=money_outflow(...))``. - - ``currency`` must be a 3-letter uppercase ISO-4217 code in - the whitelist (USD/EUR/GBP/CHF/CAD/AUD/JPY/KWD/BHD/OMR). - ``"usd"``, ``"Usd"``, ``"USDX"`` all raise - ``InvalidCurrencyError`` at decorator-application time. - - ``units`` defaults to ``"minor"`` for backward compatibility - with the pre-Decimal path. New code that passes Decimal - amounts in major units should pass ``units="major"`` explicitly. - - ``enforce_business_cap`` defaults to ``True`` so any debit - above the per-currency cap goes through the explicit - human-approval path. Set to ``False`` for batch settlement - tools that already have a human-in-the-loop approval flow. - """ - return MoneyImpactExtractor( - argument=argument, - direction=OUTFLOW, - currency=currency, - units=units, - extractor_id=extractor_id, - extractor_version=extractor_version, - enforce_business_cap=enforce_business_cap, - ) - - -# ============================================================================ -# ToolParameters extractor -# ============================================================================ -# -# The ToolParamsExtractor is the SDK-side mirror of the backend's -# ``BusinessImpact::ToolCall(ToolCallParams)`` variant. It captures a -# free-form argument bag from the live call and ships it as -# ``kind: "tool_call"`` on the /execute wire so the backend can match -# the params against ToolParameters Approval Rules (ValueMatcher: -# Equals / OneOf / NumericRange / Regex / Exists; TriggerLogic: Any / -# All / DNF groups). -# -# Why this exists alongside MoneyImpactExtractor: -# - The Money variant answers one question ("how much money?") and -# is matched against MoneyAmount predicates. -# - The ToolCall variant answers a different question ("which args?") -# and is matched against ToolParameters predicates. -# - Same wire envelope (``BusinessImpact``), same digest contract, same -# fail-CLOSED semantics at extraction time -- only the -# ``impact_for(...)`` body differs. -# -# Why ``include_all=True`` is the default (and not opt-in): -# - Operators adopting ToolParameters Approval Rules need their -# tools to ship args without rewriting every decorator site. -# - SDK 0.18.1+: ``@protect`` auto-attaches this extractor on every -# protected function (see ``protect()`` in decorators.py), so the -# behaviour is "every protected tool ships its args by default". -# - The legacy bare ``@sensitive`` (no impact=...) is deprecated in -# 0.18.1+; it auto-attached this extractor too, so it was already -# equivalent to ``@protect`` alone. It still works in 0.18.x with a -# ``DeprecationWarning`` and will be removed in 0.19.x. -# - Users with sensitive args (e.g. raw PANs, secrets) who want to -# opt out pass ``@sensitive(impact=tool_params(include_all=False))`` -# explicitly — the advanced API. -# -# Why ``param_extractors`` is an explicit map (not a glob): -# - The operator-facing rule references param names -# ("user_id == 42") not arg names ("uid", "userId" -- which the -# SDK may rename mid-refactor). An explicit map decouples the -# rule name from the function signature. -# - Without it, a Python refactor of ``uid`` -> ``user_id`` would -# silently break every ToolParameters rule without warning. - -_TOOL_CALL_EXTRACTOR_ID = "nullrun.tool_call.path" -_TOOL_CALL_EXTRACTOR_VERSION = "1" - -# Per-value size cap for wire-shipped params. Implemented as a -# truncation, not a rejection: oversize values get a deterministic -# suffix so the operator can tell from the wire payload that the -# value was bounded (rather than receiving a value that silently -# drops without trace). 1024 bytes is an implementation safety -# limit, not part of the public SDK contract -- the backend can -# reject any value it doesn't want; the SDK's job is to never -# build an unbounded payload that could blow past per-request -# transport limits. -_TOOL_PARAM_VALUE_MAX_BYTES = 1024 -_TOOL_PARAM_TRUNCATION_MARKER = "...[truncated:{} bytes]" - - -class ToolParamsExtractor: - """ToolParameters impact extractor. - - Captures the live call's kwargs into a free-form JSON object - that the backend matches against ToolParameters Approval - Rules. The wire shape mirrors ``ToolCallParams`` in - ``nullrun.business_impact`` and the backend - ``BusinessImpact::ToolCall`` variant in - ``backend/src/proxy/gate/business_impact.rs:62-307``. - - Three extraction modes (mutually exclusive in priority order): - - 1. ``param_extractors`` set: explicit {rule_param: arg_name} - map. Only the listed args are extracted; everything else - is dropped. Use this when the rule name diverges from the - function arg name (``{"user_id": "uid"}``). - 2. ``include_all=True`` (default): capture every kwarg as-is. - Use this when rule names match arg names. - 3. ``include_all=False`` and no map: empty ``params``. Rare; - for tools that take no args but should still be eligible - for ToolCall-kind Approval Rules. - - Args dropped from ``params``: - - positional args (only kwargs survive; positional binding - would require the operator to know Python's arg-order - semantics, which is fragile across refactors) - - values that fail ``_validate_param_value`` (e.g. ``float`` - which can't round-trip through JSON losslessly) - - values that survived ``_safe_kwargs`` masking (i.e. - ``***`` sentinels for PAN, password, etc.). Sending a - masked sentinel to the operator would never match a - real rule -- the rule predicate sees ``"***"`` and never - the real value. So masking happens BEFORE this extractor - runs and the masked value is filtered out here. - """ - - __slots__ = ( - "param_extractors", - "include_all", - "extractor_id", - "extractor_version", - # 0.18.1: marker stamped by ``@protect`` auto-attach so - # ``_enforce_sensitive_tool`` can distinguish "SDK installed - # this for tooling reasons" from "developer opted in via - # ``@sensitive(impact=...)``". Auto-attached extractors do - # NOT trigger the policy gate; explicit ones do. - "_nullrun_auto_attached", - ) - - def __init__( - self, - param_extractors: dict[str, str] | None = None, - *, - include_all: bool = True, - extractor_id: str = _TOOL_CALL_EXTRACTOR_ID, - extractor_version: str = _TOOL_CALL_EXTRACTOR_VERSION, - ) -> None: - if param_extractors is not None and include_all is False: - # The two modes are mutually exclusive. Setting both is - # almost certainly a typo -- fail-CLOSED at decorator- - # application time rather than silently dropping rules. - raise ValueError( - "ToolParamsExtractor: param_extractors and include_all=False are mutually exclusive" - ) - self.param_extractors = param_extractors - self.include_all = include_all - self.extractor_id = extractor_id - self.extractor_version = extractor_version - - def impact_for( - self, - fn: Callable[..., Any], - args: tuple[Any, ...], - kwargs: dict[str, Any], - ) -> BusinessImpact: - """Extract a ToolCall BusinessImpact from the live call. - - Args: - fn: the decorated function (used only for ``fn.__name__`` - on the wire; positional arg shape is not consulted). - args: positional args (dropped; only kwargs are extracted). - kwargs: keyword args from the live call. May have been - PII-masked by ``_safe_kwargs`` before reaching here; - masked values (any ``str`` that equals the - ``MASK_SENTINEL`` value used by the masking layer) - are filtered out before the wire. - - Returns: - ``BusinessImpact(impact=ToolCallParams(...))`` ready to - serialise to wire via ``to_wire_dict()``. - - Raises: - ValueError: if the resulting ``ToolCallParams`` fails - validation (bad ``tool_name``, ``params`` key too - long, f64 value, unsupported type). - """ - # Filter masked values BEFORE building ToolCallParams. The - # masking layer uses a fixed sentinel string (e.g. "***"); - # sending that to the backend would never match a real - # rule, so we silently drop it. This matches the - # pre-existing rationale that masked audit-log entries - # exist for compliance, not for runtime matching. - # Note: the decorator wrapper above us already masked the - # positional args (via ``_safe_args``) and the kwargs (via - # ``_safe_kwargs``); we just filter kwargs against the - # same sentinel value. - params = self._extract_params(kwargs) - - params_obj = ToolCallParams( - tool_name=fn.__name__, - params=params, - extractor_id=self.extractor_id, - extractor_version=self.extractor_version, - ) - params_obj.validate() - return BusinessImpact(impact=params_obj) - - def _extract_params(self, kwargs: dict[str, Any]) -> dict[str, Any]: - """Pull the right kwargs into the wire-shape ``params`` dict. - - Three branches, in priority order: - 1. ``param_extractors`` set: explicit {rule_param: arg_name} - 2. ``include_all=True``: every kwarg - 3. neither: empty dict - - Each value is filtered through ``_safe_for_wire``: only - JSON-roundtrippable types survive, and PII-masked - sentinels are dropped. Surviving values are then bounded - (``_bound_value``) — oversize scalars get a deterministic - truncation marker; nested containers are walked with a - cycle guard so a Python object graph with self-references - cannot raise RecursionError out of the extractor. - """ - result: dict[str, Any] = {} - dropped: dict[str, int] = {} - if self.param_extractors is not None: - for rule_param, arg_name in self.param_extractors.items(): - if arg_name not in kwargs: - continue - value = kwargs[arg_name] - if not _safe_for_wire(value): - _record_dropped(dropped, value) - continue - result[rule_param] = _bound_value(value) - elif self.include_all: - for k, v in kwargs.items(): - if not _safe_for_wire(v): - _record_dropped(dropped, v) - continue - result[k] = _bound_value(v) - if dropped: - # Aggregate per extraction -- one DEBUG line, never one - # per dropped field. Names and values are NOT logged; - # only the type name and count. Operators debugging - # "why isn't my rule matching" can enable DEBUG to see - # which types their tool's args are being filtered as. - _logger.debug( - "ToolParamsExtractor dropped %d values: %s", - sum(dropped.values()), - ", ".join(f"{type_name}={count}" for type_name, count in sorted(dropped.items())), - ) - return result - - -def _safe_for_wire(value: Any) -> bool: - """Return True if ``value`` can land on the wire unmodified. - - Two reasons to drop a value: - 1. It's a PII-masked sentinel (``"***"`` or similar) -- the - operator would see a placeholder, never the real value, - so the rule predicate can't match. - 2. It's an unsupported type that the canonical-JSON layer - can't round-trip (``float``, ``set``, custom objects). - The extractor validates each surviving value through - ``_validate_param_value`` which catches f64 explicitly. - - This is the wire-safety mirror of the backend's - ``ToolCallParams::validate()`` and ``_check_value_kind``. - Doing the check here keeps the SDK's "what to ship" logic - in one place (this module), so the wire shape is - documented and testable without crossing the network. - """ - if value is None: - return True - if isinstance(value, bool): - return True - if isinstance(value, int): - return True - if isinstance(value, str): - # Drop PII-masked sentinels. The masking layer uses the - # literal "***" string; a rule that matches "***" would - # be a confused rule. Real values that happen to equal - # "***" are vanishingly rare; if the operator runs into - # it, they can rename the value. - if value == "***": - return False - return True - if isinstance(value, (list, tuple)): - return True - if isinstance(value, dict): - return True - # float, set, custom objects -- rejected at this layer so we - # never build a ToolCallParams that the backend would reject. - return False - - -# Module-level logger for extraction diagnostics. -_logger = logging.getLogger("nullrun.extractor") - - -def _record_dropped(bucket: dict[str, int], value: Any) -> None: - """Bucket a dropped value by its type name into the per-call aggregate. - - The bucket is local to one ``_extract_params`` call -- the - aggregate DEBUG line is emitted once per extraction, never once - per dropped field. ``bucket`` is mutated in place; the caller - emits the log line after the walk completes. - """ - bucket[type(value).__name__] = bucket.get(type(value).__name__, 0) + 1 - - -def _bound_string(value: str) -> str: - """Bound a string value to ``_TOOL_PARAM_VALUE_MAX_BYTES`` bytes. - - Returns the value unchanged when it fits. Otherwise truncates - to ``max_bytes - len(marker)`` and appends a deterministic - marker showing how many bytes were dropped. The marker is - sized so the returned string never exceeds the cap: - - _TOOL_PARAM_TRUNCATION_MARKER = "...[truncated:N bytes]" - - with N = number of bytes that were dropped (NOT the original - length). Operators can grep for ``...[truncated:`` in the - audit log to spot values that needed bounding. - """ - encoded = value.encode("utf-8", errors="replace") - if len(encoded) <= _TOOL_PARAM_VALUE_MAX_BYTES: - return value - marker = _TOOL_PARAM_TRUNCATION_MARKER.format( - len(encoded) - _TOOL_PARAM_VALUE_MAX_BYTES - ) - marker_bytes = len(marker.encode("utf-8")) - head = encoded[: _TOOL_PARAM_VALUE_MAX_BYTES - marker_bytes] - # Decode back to str so the returned value is still a Python - # ``str`` and round-trips through the canonical-JSON layer - # the same way the original would have. ``errors="replace"`` - # is fine here because the marker itself is pure ASCII. - return head.decode("utf-8", errors="replace") + marker - - -def _bound_value(value: Any, _seen: set[int] | None = None) -> Any: - """Bound a surviving value to the per-value size cap. - - Strings get a deterministic truncation suffix (see - ``_bound_string``). Nested ``list`` / ``tuple`` / ``dict`` get - walked recursively with a cycle guard so a Python object - graph with self-references returns the partial walk instead - of raising ``RecursionError`` -- a guard that was implicit - when extraction was opt-in (developer had to opt into the - failure mode) but became mandatory when extraction became - the default for every ``@protect`` call. - - Other JSON-safe types (``int``, ``bool``, ``None``) are - inherently bounded and pass through unchanged. - """ - if isinstance(value, str): - return _bound_string(value) - if isinstance(value, (list, tuple)): - if _seen is None: - _seen = set() - if id(value) in _seen: - # Cycle: stop the walk and return what we have so far. - # Returning the (possibly partial) container is safer - # than raising -- the operator gets to see the partial - # structure and the backend can reject or accept. - return [] - _seen = _seen | {id(value)} - return [_bound_value(item, _seen) for item in value] - if isinstance(value, dict): - if _seen is None: - _seen = set() - if id(value) in _seen: - return {} - _seen = _seen | {id(value)} - return {k: _bound_value(v, _seen) for k, v in value.items()} - # int / bool / None are inherently bounded. - return value - - -def tool_params( - param_extractors: dict[str, str] | None = None, - *, - include_all: bool = True, -) -> ToolParamsExtractor: - """Shorthand constructor used by ``@sensitive(impact=tool_params(...))``. - - Every ``@protect`` tool auto-attaches a - ``ToolParamsExtractor(include_all=True)`` (see ``protect()`` - in decorators.py — SDK 0.18.1+), so most users never need to - call this function explicitly. The factory below is for two - opt-in cases: - - 1. Explicit ``{rule_param: arg_name}`` mapping when the - rule name diverges from the function arg name:: - - @protect - @sensitive(impact=tool_params({"user_id": "uid"})) - def delete_user(uid: int): ... - - 2. Strict opt-out from auto-capture (rare; for tools whose - every kwarg is a secret the operator must never see):: - - @protect - @sensitive(impact=tool_params(include_all=False)) - def handle_secret(token: str): ... - - SDK 0.18.1: bare ``@sensitive`` is deprecated — ``@protect`` - already auto-attaches the default extractor, so the bare form - contributed nothing the canonical form does not. The - ``@sensitive(impact=tool_params(...))`` factory form remains - the advanced API for typed impact. - - Args: - param_extractors: explicit ``{rule_param: arg_name}`` map. - When set, only those args are captured under - ``rule_param`` keys. ``include_all`` is ignored. - include_all: when True (default), capture every kwarg. - Ignored when ``param_extractors`` is set. - - Returns: - ``ToolParamsExtractor`` ready to be passed to - ``@sensitive(impact=...)`` or stamped on a function by the - auto-attach path. - """ - return ToolParamsExtractor( - param_extractors=param_extractors, - include_all=include_all, - ) diff --git a/src/nullrun/instrumentation/auto.py b/src/nullrun/instrumentation/auto.py index f135e20..465455e 100644 --- a/src/nullrun/instrumentation/auto.py +++ b/src/nullrun/instrumentation/auto.py @@ -172,7 +172,6 @@ def _openai_extractor(body: bytes, status: int) -> ExtractedUsage | None: "completion_tokens": completion, "total_tokens": total, "model": payload.get("model"), - # Audit 2026-06-29 (unified fingerprint): the upstream # chat-completion id (``payload["id"]``, e.g. # ``"chatcmpl-Dw7288WJI4bBDFyQ4DnZvhPUKfaZo"`` for OpenAI) is # the tightest discriminator for collapsing the sibling @@ -203,7 +202,6 @@ def _openai_extractor(body: bytes, status: int) -> ExtractedUsage | None: # --------------------------------------------------------------------------- -# D2.5 (Audit 2026-06-29): unified LLM-call fingerprint # --------------------------------------------------------------------------- # The httpx transport and the LangChain callback both observe the same # real LLM call, but until this commit they computed fingerprints from @@ -311,7 +309,6 @@ def _anthropic_extractor(body: bytes, status: int) -> ExtractedUsage | None: "completion_tokens": out, "total_tokens": inp + out, "model": payload.get("model"), - # Audit 2026-06-29 (unified fingerprint): Anthropic message id # e.g. ``"msg_01HXYZ..."``. See _openai_extractor comment. "id": payload.get("id"), "cache_read_tokens": int(usage.get("cache_read_input_tokens", 0) or 0), @@ -376,7 +373,6 @@ def _gemini_extractor(body: bytes, status: int) -> ExtractedUsage | None: "completion_tokens": completion, "total_tokens": total or (prompt + completion), "model": payload.get("modelVersion"), - # Audit 2026-06-29 (unified fingerprint): Gemini doesn't # currently surface a stable response id at the top level # fall back to ``None`` and rely on model+provider to # disambiguate. See _openai_extractor for the rationale. @@ -483,7 +479,6 @@ def _cohere_extractor(body: bytes, status: int) -> ExtractedUsage | None: "completion_tokens": out, "total_tokens": total, "model": payload.get("model"), - # Audit 2026-06-29 (unified fingerprint): Cohere v2 doesn't # surface a stable response id at the top level; rely on # model+provider for disambiguation. See _openai_extractor. "id": payload.get("id") or payload.get("generation_id"), @@ -639,7 +634,6 @@ def _bedrock_extractor(body: bytes, status: int) -> ExtractedUsage | None: "completion_tokens": out, "total_tokens": total, "model": payload.get("modelId") or payload.get("model"), - # Audit 2026-06-29 (unified fingerprint): Bedrock InvokeModel # response carries ``id`` at the top level (e.g. # ``"msg_01ABC..."`` for Anthropic-on-Bedrock, ``"cmpl-..."`` # for Mistral-on-Bedrock). Falls back to ``None`` when the @@ -764,7 +758,6 @@ def _check_kill_before_send(runtime: Any, request: httpx.Request) -> None: state = runtime._remote_state_for(workflow_id) if hasattr(runtime, "_remote_state_for") else getattr(runtime, "_remote_states", {}).get(workflow_id, {}) state_name = state.get("state", "Normal") if state_name == "Killed": - # 2026-09-08: typed kill signal (NR-W002). Cookbook # code can `except NullRunWorkflowKilledError`; legacy # `except WorkflowKilledInterrupt` still matches (subclass). from nullrun.breaker.exceptions import NullRunWorkflowKilledError @@ -821,7 +814,6 @@ def handle_request(self, request: httpx.Request) -> httpx.Response: return self._inner.handle_request(request) response = self._inner.handle_request(request) try: - # P0-3: bounded read — never buffer more than # MAX_RESPONSE_BYTES for tracking purposes. Above the cap # we skip tracking (the user still gets the full body via # the rebuilt response below). The body still needs to @@ -873,7 +865,6 @@ def _emit( body: bytes, status: int, ) -> None: - # 2026-06-28 (Issue 2 fix): if the extractor returned ``None`` # for ``model`` (response body lacked the field — observed for # some OpenAI Responses-API and Anthropic streaming edge cases) # fall back to the model name embedded in the request body. The @@ -928,7 +919,6 @@ async def handle_async_request(self, request: httpx.Request) -> httpx.Response: return await self._inner.handle_async_request(request) response = await self._inner.handle_async_request(request) try: - # P0-3: bounded read (see sync path for full rationale). body = await _aread_body_with_cap(response, MAX_RESPONSE_BYTES) if body is None: # 0.9.0: emit llm_call with metadata.streaming_skipped: true @@ -971,7 +961,6 @@ def _emit( body: bytes, status: int, ) -> None: - # F-29 (UI-UX-AUDIT 2026-08-14): mirror the sync path's # request-body model fallback (lines 882-885) so async # Anthropic / OpenAI streaming clients without ``usage.model`` # don't silently zero-bill. ``_extract_model_from_request_body`` @@ -1155,7 +1144,6 @@ def _fingerprint_for_event_dict(event: dict[str, Any]) -> str: _httpx_patched = False _httpx_lock = threading.Lock() # separate locks for the langchain / langgraph -# patch functions. The pre-fix code did ``if _x_patched: # return True`` and ``getattr(SomeClass, "_nullrun_patched" # False)`` without a lock — two threads racing through # ``auto_instrument`` simultaneously could both pass the early @@ -1174,7 +1162,6 @@ def _fingerprint_for_event_dict(event: dict[str, Any]) -> str: # first runtime — silently losing track calls from later test runs. _orig_sync_init: Callable[..., Any] | None = None _orig_async_init: Callable[..., Any] | None = None -# Audit 2026-06-29 (reset_for_tests gap): stash the originals of the # class methods we wrap so reset_for_tests can put them back. Without # this, a second test pass with `_langchain_patched = False` would # double-wrap `BaseCallbackManager.__init__`, and similarly for @@ -1230,7 +1217,6 @@ def _wrap_async_init(self: httpx.AsyncClient, *args: Any, **kwargs: Any) -> None _httpx_patched = True logger.info("httpx auto-instrumentation installed (sync + async)") - # Audit 2026-06-29 (init-ordering hazard): the class-level # __init__ patch only wraps httpx.Clients created AFTER it is # installed. If a user does # @@ -1330,7 +1316,6 @@ def patch_langchain_callback(runtime: Any) -> bool: """Install NullRunCallback into the LangChain callback manager so all LLM calls (including mock providers) flow through it. Idempotent. - #47: the pre-fix code did ``if _langchain_patched: return`` and ``getattr(BaseCallbackManager, "_nullrun_patched", False)`` without a lock; two threads racing through ``auto_instrument`` simultaneously could both pass the early check, then both @@ -1355,7 +1340,6 @@ def patch_langchain_callback(runtime: Any) -> bool: return True _orig_init = BaseCallbackManager.__init__ - # Audit 2026-06-29 (reset_for_tests gap): stash the original # on a module-level so reset_for_tests can put it back. # Without this, a second test pass with `_langchain_patched # = False` would double-wrap. @@ -1390,7 +1374,6 @@ def _wrap_init(self: Any, *args: Any, **kwargs: Any) -> None: # D4b: patch_chat_model_invoke — defensive callback injection at the LLM # boundary. # --------------------------------------------------------------------------- -# Audit 2026-06-29 (silent zero-billing): the previous # ``patch_langchain_callback`` only wrapped ``BaseCallbackManager.__init__``. # When the user instantiates ``ChatOpenAI(...)`` *before* ``nullrun.init`` # (a common pattern — see SDK examples), the ``ChatOpenAI`` object keeps @@ -1555,7 +1538,6 @@ def patch_openai_agents(runtime: Any) -> bool: _orig_run = Runner.run _orig_run_sync = getattr(Runner, "run_sync", None) - # Audit 2026-06-29 (reset_for_tests gap): stash originals so # reset_for_tests can restore them. Without this, a second # test pass with `_agents_patched = False` would double-wrap # Runner.run / Runner.run_sync. @@ -1623,7 +1605,6 @@ def _emit_from_agents_result(runtime: Any, result: Any) -> None: name = (tc.get("function") or {}).get("name") if name: tool_names.append(name) - # Audit 2026-06-28 (SDK↔backend wire): ``span.get("model")`` # used to be put on the wire as-is — when the agents SDK # didn't populate the span's ``model`` field (some # custom tracer configs), this shipped ``model=None`` → @@ -1699,7 +1680,6 @@ def patch_langgraph_compiled(runtime: Any) -> bool: importable. #47: same fix as ``patch_langchain_callback`` — the - pre-fix code read the patched flag and the class-level marker without a lock, so two threads racing through ``auto_instrument`` could both fall through to ``Pregel.invoke = _wrap_invoke`` and double-wrap the class. @@ -1828,7 +1808,6 @@ def auto_instrument(runtime: Any) -> bool: paths = [ safe_patch("httpx", lambda: patch_httpx(runtime)), safe_patch("langchain_callback", lambda: patch_langchain_callback(runtime)), - # D4b (2026-06-29): belt-and-suspenders callback injection at # the BaseChatModel.invoke boundary. Ensures NullRunCallback # fires even when the user creates the LLM BEFORE init and # the BaseCallbackManager.__init__ patch is somehow bypassed @@ -1918,7 +1897,6 @@ def reset_for_tests() -> None: _orig_pregel_stream = None _orig_pregel_ainvoke = None _orig_pregel_astream = None - # D4b (2026-06-29): restore BaseChatModel.invoke/ainvoke/stream/astream # if we patched them, otherwise the next test pass would double-wrap. if _orig_chat_model_invoke is not None: try: @@ -1936,7 +1914,6 @@ def reset_for_tests() -> None: _orig_chat_model_astream = None global _chat_model_invoke_patched _chat_model_invoke_patched = False - # Audit 2026-06-29 (reset_for_tests gap): pre-fix the function # reset the *_patched flag for langchain_callback and openai_agents # but did NOT restore the wrapped class methods. A second test # pass with `auto._langchain_patched = False; auto_instrument(r)` @@ -1973,7 +1950,6 @@ def reset_for_tests() -> None: DEDUP_LRU_MAX = 4096 # 4096 entries give a 410ms dedup window at 10K events/sec -# P0-3: streaming-OOM cap. Pre-fix, the sync transport # called ``response.read `` and the async transport called # ``await response.aread `` — both buffer the ENTIRE response body # in memory. For an OpenAI streaming completion with max_tokens=8192 @@ -2123,7 +2099,6 @@ def _emit_streaming_skipped( # denominator (``llm_call_count``) stays accurate. When ``model`` # is ``None`` the backend's into_track_request_v2 gate may log a # ``cost_pipeline_missing_model_total`` warning, but that's the - # same noise a streaming-skipped response produced pre-0.9.1 and # is preferable to silently dropping the event and skewing # coverage_pct. The ``metadata.streaming_skipped: True`` flag # tells the backend this is a known-skipped emission, not a @@ -2144,7 +2119,6 @@ def _emit_streaming_skipped( "tracked": False, "streaming_skipped": True, }, - # Audit 2026-06-29 (unified fingerprint): use the # shared ``_fingerprint_for_llm_call`` helper so this # ghost emission also collapses with any sibling # emission the LangChain callback produces for the @@ -2153,7 +2127,6 @@ def _emit_streaming_skipped( # provider pair still gives a deterministic key that # matches the callback's emission for the same call # when the callback has the model but not the id. - # (The pre-fix ``_fingerprint_for(host, b"<...>", 0)`` # sentinel produced a unique-per-path key that # collided with NOTHING.) "_fingerprint": _fingerprint_for_llm_call( diff --git a/src/nullrun/instrumentation/auto_requests.py b/src/nullrun/instrumentation/auto_requests.py index 1810914..dc4fa88 100644 --- a/src/nullrun/instrumentation/auto_requests.py +++ b/src/nullrun/instrumentation/auto_requests.py @@ -176,8 +176,12 @@ def patch_requests(runtime: Any) -> bool: with _requests_lock: if _requests_patched: return True + # `requests` is an optional dep. The bare ``try: import + # requests`` is the cheapest availability probe; the + # `_requests_patched` guard above short-circuits repeated + # patch calls on a hot path. try: - import requests # type: ignore[import-not-found] + import requests # noqa: F401 -- availability probe only except ImportError: logger.debug("requests not installed; auto-instrumentation skipped") return False diff --git a/src/nullrun/instrumentation/autogen.py b/src/nullrun/instrumentation/autogen.py index c65d335..bdcc7f4 100644 --- a/src/nullrun/instrumentation/autogen.py +++ b/src/nullrun/instrumentation/autogen.py @@ -101,7 +101,6 @@ def _wrap_create(self: Any, *args: Any, **kwargs: Any) -> Any: getattr(usage, "total_tokens", 0) or 0 ) or (prompt + completion) if prompt or completion or total: - # Audit 2026-06-28 (SDK↔backend wire): model # used to come only from ``self.model`` with a # bare ``None`` fallback — if the autogen client # didn't expose a ``model`` attribute (some @@ -118,7 +117,6 @@ def _wrap_create(self: Any, *args: Any, **kwargs: Any) -> Any: # model id, may differ from request if the # server aliased) # 3. None — let the runtime-level warning log - # (added 2026-06-28 in runtime.py:track ) # surface which path produced the gap. model = ( getattr(self, "model", None) @@ -152,7 +150,6 @@ def _wrap_create(self: Any, *args: Any, **kwargs: Any) -> Any: OpenAIChatCompletionClient._nullrun_patched = True # type: ignore[attr-defined] except ImportError: # autogen-agentchat present but autogen-ext not installed — - # spans still work; usage capture silently skipped. pass BaseChatAgent._nullrun_patched = True # type: ignore[attr-defined] diff --git a/src/nullrun/instrumentation/langgraph.py b/src/nullrun/instrumentation/langgraph.py index c784aeb..9fd7e82 100644 --- a/src/nullrun/instrumentation/langgraph.py +++ b/src/nullrun/instrumentation/langgraph.py @@ -287,7 +287,6 @@ def extract_usage_from_response(response: Any, provider: str, model: str) -> dic usage["cache_write_tokens"] = int(cache_write) or 0 prompt_details = raw.get("prompt_tokens_details") or {} if isinstance(prompt_details, dict) and prompt_details.get("cached_tokens"): - # OpenAI's prefix-cached prompt hits — best-effort merge. usage["cache_read_tokens"] = int(prompt_details.get("cached_tokens") or 0) completion_details = raw.get("completion_tokens_details") or {} if isinstance(completion_details, dict) and completion_details.get("reasoning_tokens"): @@ -412,7 +411,6 @@ def __init__(self, runtime: Any | None = None) -> None: self._active_runs: OrderedDict[str, SpanContext] = OrderedDict() self._active_runs_max: int = _ACTIVE_RUNS_MAX - # F-28 (UI-UX-AUDIT 2026-08-14): protect ``_active_runs`` with # a reentrant lock so concurrent callbacks on multi-threaded # LangChain runners (and free-threaded CPython PEP 703 builds) # cannot interleave ``on_chain_start`` / ``on_chain_end`` in a @@ -438,7 +436,6 @@ def _register_active_run(self, run_id: str, ctx: SpanContext) -> None: If the dict is at capacity, evict the oldest-inserted entry and log a warning so operators can detect chain-end drops. """ - # F-28 (UI-UX-AUDIT 2026-08-14): the cap-check + eviction + # insertion must be atomic against ``_end_run`` on a different # thread, otherwise two threads can both pass the cap check # and one eviction races the other insert (the dict grows @@ -496,7 +493,6 @@ def on_llm_start(self, serialized: Any, prompts: Any, **kwargs: Any) -> None: parent_ctx: SpanContext | None = None if parent_run_id: - # F-28 (UI-UX-AUDIT 2026-08-14): the lookup is a single # ``.get()`` (no nested acquire), but we still hold the # lock so a concurrent ``_register_active_run`` / # ``_end_run`` cannot observe a partial state where the @@ -513,7 +509,6 @@ def on_llm_start(self, serialized: Any, prompts: Any, **kwargs: Any) -> None: ctx = create_root_span() self._register_active_run(str(run_id), ctx) - # DEFS-SDKEXEC-LLM-RESERVATION (2026-09-08): pair the LLM # span with a server-minted reservation so the matching # llm_call cost event emitted by ``on_llm_end`` lands on # ``/track_single`` instead of being dropped by @@ -528,8 +523,6 @@ def on_llm_start(self, serialized: Any, prompts: Any, **kwargs: Any) -> None: # ``_route_track`` will then drop the matching llm_call # cost event (v3.66.2 alignment — backend rejects batched # llm_call events without a reservation with 503 - # BUDGET_RECHECK_FAILED). This matches the pre-fix - # behaviour because pre-fix the SDK also had no reservation # at this site (no /check round-trip happened on the LLM # span) and the llm_call cost event was dropped the same # way. We swallow ``WorkflowKilledInterrupt`` / @@ -628,7 +621,6 @@ def on_llm_end(self, response: Any, **kwargs: Any) -> None: f"usage={usage}, has_usage={usage['has_usage']}" ) - # Audit 2026-06-29 (unified fingerprint): derive the same # fingerprint the httpx transport computes for the same # call, so the dedup LRU at runtime.track collapses the # two emissions to a single wire event. Both observers feed @@ -715,7 +707,6 @@ def on_llm_end(self, response: Any, **kwargs: Any) -> None: # Stripped at the wire boundary by _WIRE_STRIP_FIELDS — # kept here for in-process dedup + test introspection. "raw_usage": usage["raw_usage"], - # Audit 2026-06-29 (unified fingerprint): use the # same helper the httpx transport calls so the dedup # LRU at runtime.track collapses the sibling # emission for the same real LLM call. Pre-fix this @@ -732,7 +723,6 @@ def on_llm_end(self, response: Any, **kwargs: Any) -> None: logger.info(f"NullRun track event: {event}") - # 2026-07-12 (multi-agent span attachment): the per-LLM-call # cost event must carry the parent chain's `trace_id` so the # backend's unified SELECT can JOIN `cost_summary` by it. # `on_llm_start` already stored the SpanContext under the @@ -752,7 +742,6 @@ def on_llm_end(self, response: Any, **kwargs: Any) -> None: # upcoming tree-renderer that wants to walk children by # the parent's trace bucket. llm_run_id = kwargs.get("run_id") - # F-28 (UI-UX-AUDIT 2026-08-14): the lookup must hold the # lock so a concurrent ``_end_run`` cannot pop the entry # between this ``.get()`` and the (later) ``_end_run`` at # the bottom of this method — that race produced the @@ -894,7 +883,6 @@ def _begin_run( """ parent_ctx: SpanContext | None = None if parent_run_id: - # F-28 (UI-UX-AUDIT 2026-08-14): same orphan-span race as # ``on_llm_start``. The subsequent ``_register_active_run`` # call already acquires the lock; the reentrant ``RLock`` # lets us hold it across BOTH the lookup AND the @@ -927,7 +915,6 @@ def _begin_run( def _end_run(self, run_id: Any, error: str | None = None) -> None: if run_id is None: return - # F-28 (UI-UX-AUDIT 2026-08-14): the pop must be atomic so a # concurrent ``_register_active_run`` cannot INSERT an entry # for the same ``run_id`` between this pop and the # ``runtime.track_event`` call below — the freshly-inserted @@ -970,7 +957,6 @@ def _extract_node_name(serialized: Any, default: str) -> str: # --------------------------------------------------------------------------- -# Audit 2026-06-28 (SDK↔backend wire): model_name on the callback path # --------------------------------------------------------------------------- # Pre-fix: ``on_llm_end`` pulled ``model_name`` exclusively from # ``kwargs['invocation_params']`` with a hard fallback to the literal @@ -1085,7 +1071,6 @@ def _extract_model_from_response(response: Any) -> str | None: # operator can correlate the wire warning back to a specific # response shape. # - # Audit 2026-06-29 (silent zero-billing): the previous version # emitted a single DEBUG line with only the response type. That # was insufficient when the operator needed to see *which* of # the four fallback steps almost-but-didn't match. We now dump diff --git a/src/nullrun/instrumentation/llama_index.py b/src/nullrun/instrumentation/llama_index.py index 64e28c8..215dd70 100644 --- a/src/nullrun/instrumentation/llama_index.py +++ b/src/nullrun/instrumentation/llama_index.py @@ -50,7 +50,6 @@ def on_chat_end(event: Any) -> None: total = int(usage.get("total_tokens", 0) or 0) or (prompt + completion) if not (prompt or completion or total): return - # Audit 2026-06-28 (SDK↔backend wire): model used to come # only from ``event.response.model`` with a bare ``None`` # fallback — mock providers and some adapters don't # populate ``.model`` on ChatResponse, which sent diff --git a/src/nullrun/integrations/fastapi.py b/src/nullrun/integrations/fastapi.py index 5e35fdd..a281510 100644 --- a/src/nullrun/integrations/fastapi.py +++ b/src/nullrun/integrations/fastapi.py @@ -71,11 +71,8 @@ def chat(message: str) -> str: """ from __future__ import annotations -from collections.abc import Callable - from fastapi import FastAPI, Request from fastapi.responses import JSONResponse -from starlette.requests import Request as StarletteRequest from nullrun.breaker.exceptions import ( NullRunDecision, @@ -107,35 +104,9 @@ def chat(message: str) -> str: _KILL_STATUS = 503 -# Locale negotiation helpers -LocaleResolver = Callable[[Request], str] - - -def _default_locale_resolver(request: Request) -> str: - """Parse ``Accept-Language`` and return a 2-letter locale code. - - Falls back to ``"en"`` when the header is missing or malformed. - Only the first supported subtag is returned (``en-US`` → ``en``). - """ - header = request.headers.get("accept-language", "") - if not header: - return "en" - first = header.split(",", 1)[0].strip() - first = first.split(";", 1)[0].strip() - primary = first.split("-", 1)[0].strip().lower() - return primary or "en" - - -def _resolve_locale(request: Request, resolver: LocaleResolver | None) -> str: - if resolver is None: - return _default_locale_resolver(request) - try: - return resolver(request) or "en" - except Exception: - # Resolver bugs must not break error responses. Degrade to the - # default and continue — the user still gets a clean message - # just not in their preferred locale. - return "en" +# Locale negotiation helpers removed — the catalog is English-only and +# ``format_user_message`` no longer takes a ``locale=`` kwarg. Reserved +# for a future locale-pack release if/when a non-English catalog lands. def _build_headers(exc: BaseException) -> dict[str, str]: @@ -180,13 +151,12 @@ async def _decision_handler( End-user-facing — the ``user_message`` field is safe to display verbatim to the user that triggered the request. """ - locale = _resolve_locale(request, _LOCALE_RESOLVER) status = _DECISION_STATUS.get(exc.error_code, _DEFAULT_DECISION_STATUS) return JSONResponse( status_code=status, content={ "error_code": exc.error_code, - "user_message": format_user_message(exc, locale=locale), + "user_message": format_user_message(exc), "category": "decision", "retryable": exc.retryable, }, @@ -204,12 +174,11 @@ async def _infrastructure_handler( failure (generic "service unavailable"), but ``error_code`` lets the operator triage without parsing the user's response. """ - locale = _resolve_locale(request, _LOCALE_RESOLVER) return JSONResponse( status_code=_DEFAULT_INFRASTRUCTURE_STATUS, content={ "error_code": exc.error_code, - "user_message": format_user_message(exc, locale=locale), + "user_message": format_user_message(exc), "category": "infrastructure", "retryable": exc.retryable, }, @@ -243,9 +212,8 @@ class NullRunMiddleware: register the middleware by hand. """ - def __init__(self, app, *, locale_resolver: LocaleResolver | None = None) -> None: + def __init__(self, app) -> None: self.app = app - self.locale_resolver = locale_resolver async def __call__(self, scope, receive, send) -> None: # Lifespan and websocket scopes — pass through unmodified. @@ -269,13 +237,11 @@ async def safe_send(message) -> None: except WorkflowKilledInterrupt as exc: if response_started: raise # headers already sent — re-raise and let the connection drop - request = StarletteRequest(scope, receive) - locale = _resolve_locale(request, self.locale_resolver) response = JSONResponse( status_code=_KILL_STATUS, content={ "error_code": exc.error_code, - "user_message": format_user_message(exc, locale=locale), + "user_message": format_user_message(exc), "category": "killed", }, headers=_build_headers(exc), @@ -283,48 +249,25 @@ async def safe_send(message) -> None: await response(scope, receive, send) -# Module-level resolver — set by:func:`install` and read by the -# FastAPI exception handlers. The middleware gets its own copy via -# its constructor (Starlette instantiates middleware via -# ``add_middleware``, which does not let us pass per-request state). -_LOCALE_RESOLVER: LocaleResolver | None = None - - -def install( - app: FastAPI, - *, - locale_resolver: LocaleResolver | None = None, -) -> None: +def install(app: FastAPI) -> None: """Register NullRun exception handlers + kill middleware on a FastAPI app. Idempotent — calling ``install`` twice on the same app replaces - the handlers with the latest configuration. The middleware uses - the resolver that was passed at the most recent ``install`` call. + the handlers with the latest configuration. Args: app: The FastAPI application to instrument. - locale_resolver: Optional callable ``(request) -> str`` - returning a 2-letter locale code. Defaults to parsing - ``Accept-Language``. Example:: - from fastapi import FastAPI, Request + from fastapi import FastAPI import nullrun from nullrun.integrations.fastapi import install nullrun.init(api_key="...") - app = FastAPI + app = FastAPI() install(app) - - # Custom resolver: read locale from a session cookie. - install( - app - locale_resolver=lambda req: req.cookies.get("locale", "en") - ) """ - global _LOCALE_RESOLVER - _LOCALE_RESOLVER = locale_resolver # Exception handlers for Exception subclasses. Starlette dispatches # by isinstance, so registering the more specific categories first @@ -338,7 +281,7 @@ def install( # so we add the kill middleware AFTER exception handlers — actually # it doesn't matter here because the exception handlers and the # middleware handle disjoint exception classes. - app.add_middleware(NullRunMiddleware, locale_resolver=locale_resolver) + app.add_middleware(NullRunMiddleware) __all__ = ["install", "NullRunMiddleware"] diff --git a/src/nullrun/messages.py b/src/nullrun/messages.py index 134cd8f..2ac0c12 100644 --- a/src/nullrun/messages.py +++ b/src/nullrun/messages.py @@ -36,7 +36,7 @@ # Imported under ``TYPE_CHECKING`` so this module stays importable # without pulling in the exception hierarchy (which itself depends # on transport / runtime modules). - from nullrun.breaker.exceptions import NullRunError + pass # --------------------------------------------------------------------------- @@ -107,7 +107,6 @@ # 2. Client-side timeout path — WS push went silent for # ``approval_timeout_seconds`` without an operator decision. # Cookbook pattern: do NOT retry the same approval_id; request a - # fresh row and re-/gate. Pre-fix (2026-09-08), the catalog was # missing NR-A012 entirely, so ``format_user_message`` fell # through to ``FALLBACK_MESSAGE = "Something went wrong. Please # try again."`` — exactly what @@ -145,7 +144,6 @@ # the next /gate will mint a fresh reservation against the current # available budget. "NR-B006": "Your request couldn't be completed because the available capacity changed. Please try again.", - # NR-B007: removed 2026-09-10. NullRunBudgetThrottleError was a # zombie class — never raised on a wire or runtime path # (runtime.py:2116 raises WorkflowPausedException on # decision=="throttle"). Catalog entry removed to keep @@ -156,7 +154,6 @@ # copy is generic because the cause is operator-side accounting; # user should retry (a fresh /gate will recompute the reservation). "NR-O001": "Your request couldn't be completed due to a usage accounting discrepancy. Please try again.", - # ── B.1 (2026-09-10): MCP umbrella + APPROVAL_DB typed arms. # Three MCP umbrella codes (ADR-013, frozen-dormant) and the # single NR-A016 typed class for the six APPROVAL_DB_* sibling # codes. NR-A016 wording is intentionally close to the generic @@ -274,7 +271,7 @@ def get_user_message(code: str) -> str: return DEFAULT_MESSAGES.get(code, FALLBACK_MESSAGE) -def format_user_message(exc: BaseException | object, locale: str = "en") -> str: +def format_user_message(exc: BaseException | object) -> str: """Render a NullRun exception as a user-facing string. This is the function host code should call when it wants to show @@ -285,9 +282,6 @@ def format_user_message(exc: BaseException | object, locale: str = "en") -> str: Args: exc: A NullRun exception (or any object exposing ``error_code``). - locale: DEPRECATED — reserved for a future locale-pack release. Currently ignored; the catalog is English-only. - non-``"en"`` value falls back to the English message. The - parameter is reserved for future locale packs. Returns: User-facing string. Always non-empty and safe to display. diff --git a/src/nullrun/observability/__init__.py b/src/nullrun/observability/__init__.py index d2128bf..41a65eb 100644 --- a/src/nullrun/observability/__init__.py +++ b/src/nullrun/observability/__init__.py @@ -91,7 +91,6 @@ class RuntimeMetrics: cost_limit_exceeded: int = 0 timeouts: int = 0 loop_detections: int = 0 - # 2026-08-13 (sprint handoff Bug #4): counter for the # fail-OPEN paths in ``check_workflow_budget``. Incremented on # three sites (cache-enabled exception, cache-disabled exception, # synthetic FALLBACK decision_source). Operators alert on @@ -100,7 +99,6 @@ class RuntimeMetrics: # counter, the failure mode was invisible at INFO log level on # the FALLBACK path. gate_fail_open_total: int = 0 - # 2026-08-20 (v0.16.0, backend v3.66.2 alignment): v1/v2 path # `/track/batch` with `llm_call` events missing `reservation_id` # is now fail-CLOSED at the backend (whole-batch 503 BUDGET_RECHECK_FAILED). # Pre-0.16.0 the SDK fell back to that path for calls without a diff --git a/src/nullrun/observability/error_hooks.py b/src/nullrun/observability/error_hooks.py index 622be59..57c544d 100644 --- a/src/nullrun/observability/error_hooks.py +++ b/src/nullrun/observability/error_hooks.py @@ -39,7 +39,7 @@ import time from collections.abc import Callable from dataclasses import dataclass, field -from typing import Any, Optional +from typing import Any logger = logging.getLogger(__name__) @@ -55,8 +55,7 @@ # track — POST /api/v1/track (event ingest) # gate — POST /api/v1/gate (legacy pre-flight) # check — POST /api/v1/check (budget pre-flight) -# sensitive_tool — @sensitive pre-check -# org_status — get_org_status +# org_status — get_org_status # ws — WebSocket control-plane message handling # transport — generic transport-layer raise STAGES: tuple[str, ...] = ( @@ -93,7 +92,8 @@ class ErrorContext: workflow_id: str | None = None # Tool that triggered the error, or ``None`` for non-tool - # errors. Set on @sensitive / @protect / track_tool raises. + # errors. Set on @protect raises and on transport-layer + # tool-call failures. tool_name: str | None = None # First 10 characters of the api key in use, or ``None`` if @@ -132,7 +132,6 @@ def __post_init__(self) -> None: # The callback type. Sync only — Layer 2 design discussion -# 2026-06-24: async hooks in except blocks are awkward (no # running event loop to await on), and the SDK surface is # already sync. Revisit if/when a real async use case appears. ErrorHook = Callable[["Any", ErrorContext], None] @@ -142,7 +141,6 @@ def __post_init__(self) -> None: # from one thread and fired from another (e.g. register at app # startup, fire from a transport background thread). # -# The hot path is has_hooks(), which previously took an # RLock.acquire on every call (100+ raises/min in a busy agent # is enough to show up in profiles). We now keep the hook list # under the same RLock but expose has_hooks() as a lock-free diff --git a/src/nullrun/observability/status.py b/src/nullrun/observability/status.py index 7d92c62..fa56f51 100644 --- a/src/nullrun/observability/status.py +++ b/src/nullrun/observability/status.py @@ -51,11 +51,9 @@ from __future__ import annotations import logging -import time from collections import deque from dataclasses import dataclass, field -from datetime import datetime, timezone -from typing import Any, Optional +from datetime import datetime logger = logging.getLogger(__name__) @@ -115,7 +113,6 @@ class WorkflowState: whether the body will run on the next call. CP1 fix (2026-06-26): the backend WsWorkflowState enum has 5 - variants, not 3 — Flagged and Tripped were previously silently treated as Normal. The SDK now handles all 5 explicitly in ``runtime.check_control_plane``; this dataclass reflects the full set so the operator-facing status mirrors reality. diff --git a/src/nullrun/toolbox/__init__.py b/src/nullrun/toolbox/__init__.py index d7c89b9..5bac016 100644 --- a/src/nullrun/toolbox/__init__.py +++ b/src/nullrun/toolbox/__init__.py @@ -6,7 +6,7 @@ the low-level patches (httpx, OpenAI v1+ attribute path, auto mode) the `toolbox/` package ships opinionated wrappers that combine instrumentation + cost enforcement + workflow scoping for the most -common agent runtimes (LangGraph, LlamaIndex, etc.). +common agent runtimes (MCP, etc.). The split keeps the curated public surface (`nullrun.init` `nullrun.protect`, `nullrun.track_*`) discoverable in `dir(nullrun)` @@ -16,6 +16,5 @@ from __future__ import annotations __all__ = [ - "langgraph", "mcp", ] diff --git a/src/nullrun/toolbox/langgraph.py b/src/nullrun/toolbox/langgraph.py index 58c7c6b..23c3461 100644 --- a/src/nullrun/toolbox/langgraph.py +++ b/src/nullrun/toolbox/langgraph.py @@ -1,30 +1,13 @@ """ LangGraph toolbox helpers for NullRun. -DEPRECATED for auto-instrumentation use cases. +The previous ``wrapper(app, runtime=None)`` escape hatch was removed — +``nullrun.init_or_die()`` (or ``@nullrun.protect`` on the agent +function) auto-patches ``langgraph.pregel.Pregel`` via +``nullrun.instrumentation.auto.patch_langgraph_compiled``, which +covers every supported LangGraph invocation path. -For typical LangGraph usage, ``nullrun.init_or_die()`` (or just -``@nullrun.protect`` on the agent function) auto-patches -``langgraph.pregel.Pregel`` via -``nullrun.instrumentation.auto.patch_langgraph_compiled`` — the same -callback injection that this ``wrapper()`` performs manually. The -auto-patch is the canonical path; users should NOT need to call -``wrapper(graph)`` themselves. - -This ``wrapper()`` remains as an **escape hatch** for three narrow -cases where the auto-patch cannot run or where the user needs -explicit control: - - 1. Tests with custom runtimes (the auto-patch binds to the active - runtime; a test fixture that swaps runtimes mid-flight may need - the wrapper to attach to the new runtime directly). - 2. Apps where ``Pregel`` is imported BEFORE ``nullrun.init_or_die()`` - AND the import side-effects register a non-Pregel transport - that the auto-patch cannot reach. - 3. Manual control over which ``NullRunCallback`` instance is - attached (rare; the default singleton is usually correct). - -For the canonical path (90%+ of users), omit this wrapper:: +Migrate to the auto-patch:: from nullrun import init_or_die, protect @@ -33,93 +16,4 @@ @protect def my_agent(prompt): return graph.invoke({"messages": [("user", prompt)]}) - -If you must use this wrapper explicitly (escape hatch):: - - from nullrun import init_or_die - from nullrun.toolbox.langgraph import wrapper - - runtime = init_or_die() - graph = build_my_graph() - graph = wrapper(graph, runtime=runtime) - result = graph.invoke({"messages": [("user", "hi")]}) - -Why this lives in ``toolbox/``, not ``instrumentation/``: - - ``instrumentation/`` ships the generic, low-level patches - (httpx, OpenAI v1+ attribute path, LangChain callback class, - Pregel class-method wrap). These are reusable building blocks - and run automatically on ``init_or_die()``. - - ``toolbox/langgraph.py`` ships an opinionated one-call wrapper - that mutates a specific ``app`` instance in place. It is no - longer the recommended path for typical usage. - -The previous location ``nullrun.instrumentation.langgraph.instrument`` -has been removed. Users who imported it should switch to -``nullrun.toolbox.langgraph.wrapper`` (escape hatch only) or rely -on the auto-patch (canonical path). """ -from __future__ import annotations - -import logging -from typing import Any - -from nullrun.instrumentation.langgraph import NullRunCallback -from nullrun.runtime import NullRunRuntime, get_runtime - -logger = logging.getLogger(__name__) - - -def wrapper(app: Any, runtime: Any | None = None) -> Any: - """ - Wrap a compiled LangGraph app with NullRun tracking. - - .. deprecated:: - For typical usage, rely on the auto-patch in - ``nullrun.init_or_die()`` / ``@nullrun.protect``. This wrapper - is an escape hatch for the narrow cases documented in the - module docstring (custom runtime, Pregel imported before init, - manual callback control). - - Every ``app.invoke(...)`` and ``app.stream(...)`` call gets a - ``NullRunCallback`` attached so the runtime sees the LLM - usage for cost accounting and policy enforcement. - - Args: - app: A compiled LangGraph ``StateGraph`` (anything with - ``.invoke`` and ``.stream``). - runtime: Optional ``NullRunRuntime``. Defaults to the - module-level singleton from ``get_runtime()``. - - Returns: - The same ``app`` object, with ``.invoke`` and ``.stream`` - wrapped in place. The callback is added to LangChain's - ``config["callbacks"]`` list per call, so multiple - wrappers compose without colliding. - """ - rt: NullRunRuntime = runtime or get_runtime() - callback = NullRunCallback(runtime=rt) - original_invoke = getattr(app, "invoke", None) - original_stream = getattr(app, "stream", None) - - if original_invoke is not None: - def wrapped_invoke(input: Any, config: Any | None = None, **kwargs: Any) -> Any: - if config is None: - config = {} - if "callbacks" not in config: - config["callbacks"] = [] - config["callbacks"].append(callback) - return original_invoke(input, config, **kwargs) - app.invoke = wrapped_invoke - - if original_stream is not None: - def wrapped_stream(input: Any, config: Any | None = None, **kwargs: Any) -> Any: - if config is None: - config = {} - if "callbacks" not in config: - config["callbacks"] = [] - config["callbacks"].append(callback) - return original_stream(input, config, **kwargs) - app.stream = wrapped_stream - - logger.info("LangGraph app wrapped with NullRun tracking") - return app diff --git a/src/nullrun/toolbox/mcp.py b/src/nullrun/toolbox/mcp.py index bc39bac..4e1ed95 100644 --- a/src/nullrun/toolbox/mcp.py +++ b/src/nullrun/toolbox/mcp.py @@ -58,7 +58,6 @@ from typing import Any from nullrun.context import ( - get_call_mcp_annotations, set_mcp_tool_context, ) @@ -156,13 +155,12 @@ def __init__( mcp_client: Any, cache_seconds: int = DEFAULT_CACHE_SECONDS, list_tools: Callable[[], Iterable[Any]] | None = None, - # v3.53 audit #5 — ``runtime`` is optional but RECOMMENDED. # When provided, ``call_tool`` routes the invocation through # ``runtime.execute(...)`` (the /api/v1/execute gate endpoint) # BEFORE the underlying MCP client is called, so the operator's # tool-block / budget / approval policies apply to MCP tool # calls just like they do to local functions decorated with - # ``@protect`` / ``@sensitive``. + # ``@protect``. # # When ``runtime`` is None the adapter falls back to the legacy # contextvar-only path (``set_mcp_tool_context``) so callers @@ -202,7 +200,6 @@ def __init__( # call_tool, refreshed every ``cache_seconds``. self._cache: dict[str, _CachedTool] = {} self._cached_at: float = 0.0 - # v3.53 audit #5 — optional runtime for gate enforcement on # every ``call_tool``. When provided, ``call_tool`` blocks on # ``runtime.execute(...)`` returning decision="block" so a # permissive MCP server cannot bypass the operator's @@ -317,7 +314,6 @@ def call_tool( legacy contextvar-only path — the call proceeds without any /api/v1/execute round-trip and the next ``@protect``-decorated wrapper picks up the contextvar on its next ``/check`` request. - This preserves back-compat for callers who already wire MCP calls inside ``@protect``-decorated functions. Returns the underlying client's result (when allowed). @@ -363,7 +359,6 @@ def call_tool( # when assembling the next /check request. set_mcp_tool_context(tool_class=tool_class, annotations=annotations) - # v3.53 audit #5 — when a runtime is wired, run the gate # synchronously BEFORE invoking the MCP client. This closes # the silent bypass where the agentic loop called # ``adapter.call_tool`` directly without a ``@protect`` @@ -372,19 +367,21 @@ def call_tool( # unless an ``approval_id`` is supplied) — both short-circuit # to the call site without touching ``self._mcp_client``. # - # The runtime is opt-in for back-compat: pre-v3.53 callers # who relied on the contextvar-only path continue to work. # New integrations should pass ``runtime=`` so the # tool-block / budget / approval policies actually apply. if self._runtime is not None: execute_input = arguments if arguments is not None else {} + # 0.18.2: there is no ``mode=`` opt-out — every MCP + # tool call routed through the runtime contacts + # /api/v1/execute unconditionally. The pre-refactor + # ``mode="strict"`` was removed because it was the only + # way to make the audit-grade path actually apply to + # non-sensitive MCP tools; with the inline-mode bypass + # gone, that exemption is no longer possible. execute_result = self._runtime.execute( tool_name=tool_name, input_data=execute_input, - # Strict mode forces /api/v1/execute even for - # non-sensitive MCP tools — the audit flag is that - # MCP calls previously ran without ANY gate check. - mode="strict", ) decision = execute_result.get("decision") if decision == "block": diff --git a/src/nullrun/transport.py b/src/nullrun/transport.py index 5754b84..2930e69 100644 --- a/src/nullrun/transport.py +++ b/src/nullrun/transport.py @@ -16,7 +16,6 @@ import time import uuid import weakref -from collections import OrderedDict from collections.abc import Callable from dataclasses import dataclass from typing import TYPE_CHECKING, Any, cast @@ -69,7 +68,6 @@ # `X-NULLRUN-PROTOCOL: ` with 400. Bump must be coordinated with backend # `proxy::http::gate::protocol` and `/api/v1/capabilities`. # -# v4 (2026-08-31, ADR-037 Slice B): ADDITIVE — /gate response now echoes # the SDK-supplied `action_digest` and a `policy_hash` slot (always None # today; Slice D wires per-request computation). Wire-additive: v3 SDKs # parsing the response simply ignore the new fields; v4 SDKs parsing a @@ -287,13 +285,11 @@ def _retry_with_backoff( ) raise err if result.status_code >= 500 and retry_on_5xx and attempt < max_retries: - # NR-006: treat 5xx as transient infra failure and retry. # Convert to HTTPStatusError so the except branch catches # it as a retryable condition. After retry exhaustion # the helper returns the last response (see below). result.raise_for_status() elif result.status_code >= 500 and not retry_on_5xx: - # Pre-NR-006 behaviour: 5xx without ``retry_on_5xx`` # raises HTTPStatusError so the caller (e.g. # ``Transport.execute``) can run its fallback logic # after retry exhaustion produces BreakerTransportError. @@ -306,7 +302,6 @@ def _retry_with_backoff( # returns a synthetic block; /track batch inspects status # directly). Calling ``raise_for_status()`` here would force # every caller into the except path and retry a permanent - # error — the audit's NR-006 PIN 3 pins this non-retry # contract. return result @@ -362,7 +357,6 @@ def _retry_with_backoff( time.sleep(actual_delay) - # Retry exhaustion. NR-006 path: if the caller opted into # ``retry_on_5xx`` and the failure mode was 5xx, return the # last response so the caller can synthesize a fallback # (e.g. ``Transport.check`` returns the legacy synthetic-block @@ -392,7 +386,6 @@ class FallbackMode: block agent execution, but behavior must be defined and logged. """ - # Block if Gateway unavailable. v3.53 audit #4 — DEFAULT for # ``Transport.execute()`` and ``ExecuteConfig.fallback_mode``. # Per CLAUDE.md §4 "DEFAULT: fail-CLOSED для всех enforcement # путей", the /execute enforcement path must not silently allow @@ -434,12 +427,9 @@ class FlushConfig: class ExecuteConfig: """Configuration for execute (strict mode) behavior.""" - # Fallback mode when Gateway is unavailable. v3.53 audit #4 — - # default is STRICT (fail-CLOSED on enforcement) per CLAUDE.md §4. - # Pre-v3.53 the default was PERMISSIVE which silently allowed - # local execution on transport failure; that was fail-OPEN on the - # primary enforcement path (Transport.execute → /api/v1/execute). - fallback_mode: str = FallbackMode.STRICT + # Fallback mode when Gateway is unavailable. Default is STRICT + # (fail-CLOSED on enforcement) per CLAUDE.md §4. + fallback_mode: FallbackMode = FallbackMode.STRICT # Gateway timeout in seconds timeout: float = 5.0 # Max retries for execute calls @@ -657,7 +647,6 @@ def _persist_to_wal(self) -> None: try: with open(tmp_path, "a") as f: for event in self._buffer: - # 2026-07-24 (Decimal serialization): same default=str as f.write(json.dumps(event, default=str) + "\n") f.flush() os.fsync(f.fileno()) @@ -1104,7 +1093,6 @@ def execute( tool: str, input_data: dict[str, Any], mode: str = "auto", - # v3.53 audit #4 — default flipped from PERMISSIVE to STRICT # to match CLAUDE.md §4 ("DEFAULT: fail-CLOSED для всех # enforcement путей"). /execute is the primary enforcement # point (see docstring) — when the gateway is unreachable the @@ -1112,11 +1100,12 @@ def execute( # intentionally want fail-OPEN on this path (dev / test # harnesses without a live engine) must opt in by passing # ``fallback_mode=FallbackMode.PERMISSIVE`` explicitly. - fallback_mode: str = FallbackMode.STRICT, + fallback_mode: FallbackMode = FallbackMode.STRICT, operation_id: str | None = None, approval_id: str | None = None, - # Typed-impact + digest-bound approval. Forwarded when @sensitive(impact=...) - # built them so the backend can stamp the approval row with the digest. + # Typed-impact + digest-bound approval. Forwarded when the + # gate built them so the backend can stamp the approval row + # with the digest. business_impact: dict[str, Any] | None = None, action_digest: str | None = None, # Tool-call argument bag forwarded on /execute so the gate can compute @@ -1128,8 +1117,7 @@ def execute( # aggregate. Without this, TB-1 fails closed with `no_tools_field` # whenever the workflow has an active `policy.tool_patterns` block. # Populated by `runtime.execute` from the `get_call_tools()` contextvar - # when the caller invoked `set_call_context(tools=...)` (or the - # `_enforce_sensitive_tool` decorator did so on their behalf). + # when the caller invoked `set_call_context(tools=...)`. tools: tuple[str, ...] | None = None, on_transport_error: Callable[[Exception], dict[str, Any]] | None = None, ) -> dict[str, Any]: @@ -1150,9 +1138,7 @@ def execute( method's caller is the single source of truth for ``execution_id`` selection. - Prior to DEF-SDKK-022 the comment here claimed "/execute MUST be called rather than /gate" — that contract was the legacy - pre-2026-09-04 shape. The post-fix shape is "/execute MUST be preceded by /gate for the same execution_id" — the budget pre-flight (Transport.check, /api/v1/gate) is the binding registrar; /execute is the policy decision that re-uses it. @@ -1164,13 +1150,14 @@ def execute( tool: Tool to execute input_data: Tool input mode: Execution mode ("auto", "inline", "strict") - fallback_mode: What to do if Gateway unavailable + fallback_mode: :class:`FallbackMode` enum (STRICT or + PERMISSIVE). Default STRICT (fail-CLOSED on transport + failure per CLAUDE.md §4). operation_id: Optional idempotency key on_transport_error: Optional callback invoked on BreakerTransportError. When set, the callback's return value is returned verbatim; otherwise - the request falls through to fallback_mode. The decorator's - _enforce_sensitive_tool sets this to convert the error into a - NullRunBlockedException (fail-CLOSED). + the request falls through to fallback_mode. The gate sets this + to convert the error into a NullRunBlockedException (fail-CLOSED). Returns: Dict with: @@ -1234,7 +1221,6 @@ def do_execute_request() -> httpx.Response: elif response.status_code >= 400: # 4xx — don't retry. # - # 2026-09-10 (NR-SDK-A015-SURFACE): before the fix, # this branch dropped the wire envelope on the floor # and synthesised a generic ``{"decision": "block", # "explanation": "Gateway returned 409"}`` dict. That @@ -1260,8 +1246,8 @@ def do_execute_request() -> httpx.Response: # `APPROVAL_REPLAY_REJECTED` → NR-A015 / `` # NullRunApprovalReplayRejectedError``) — and raise # the typed exception so the @protect / - # @sensitive / runtime.execute() exception arms - # propagate the right class up to the caller. + # runtime.execute() exception arms propagate the + # right class up to the caller. # # Fall through to the synthetic block shape if the # envelope is unrecognised (plaintext body, malformed @@ -1271,7 +1257,7 @@ def do_execute_request() -> httpx.Response: # Exception — it never silently swallows a 4xx. try: raise _parse_v3_error_envelope(response, "execute") - except NullRunApprovalReplayRejectedError as exc: + except NullRunApprovalReplayRejectedError: # The exact case the user reported: the operator # approved, the SDK polled /execute again, and # the backend's atomic consume_approved UPDATE @@ -1284,32 +1270,31 @@ def do_execute_request() -> httpx.Response: # instead of the fallback. metrics.inc_transport("execute_block_replay_rejected") raise - except NullRunBlockedException as exc: + except NullRunBlockedException: # All other typed blocks from the dispatch — # budget, rate, tool, approval-deny, etc. # Re-raise for the @protect / runtime.execute # arms to handle. metrics.inc_transport("execute_block_typed") raise - except NullRunBackendError as exc: + except NullRunBackendError: # 5xx-classified envelope parsed as a typed # backend error (shouldn't normally land here # because the helper maps 5xx to GATEWAY_ERROR # via NullRunTransportError, but stays # defensive). Re-raise. raise - except NullRunAuthenticationError as exc: + except NullRunAuthenticationError: # 401 envelope parsed as auth error — surface # directly so the caller can react. raise - except NullRunTransportError as exc: + except NullRunTransportError: # Transport-classified (network, breaker) — not # a real 4xx, but helper may return one if the # envelope shape is ambiguous. Re-raise so the # on_transport_error arm sees it. raise - except NullRunDecision as exc: - # DEF-NR-TRANSPORT-CATCHFANIN-GAP (2026-09-10): + except NullRunDecision: # umbrella pass-through for typed Decision # subclasses NOT in the NullRunBlockedException # MRO. Specifically: @@ -1338,8 +1323,7 @@ def do_execute_request() -> httpx.Response: # matches by MRO specificity. metrics.inc_transport("execute_block_decision_typed") raise - except NullRunInfrastructureError as exc: - # DEF-NR-TRANSPORT-CATCHFANIN-GAP (2026-09-10): + except NullRunInfrastructureError: # umbrella pass-through for typed # Infrastructure subclasses NOT in the # NullRunBackendError / NullRunAuthenticationError / @@ -1426,7 +1410,7 @@ def do_execute_request() -> httpx.Response: # All attempts failed - apply fallback mode. metrics.inc_transport("fallback_mode_activations") - if fallback_mode == FallbackMode.STRICT: + if fallback_mode == FallbackMode.STRICT: # type: ignore[comparison-overlap] return { "decision": "block", "decision_source": DecisionSource.FALLBACK, @@ -1434,11 +1418,11 @@ def do_execute_request() -> httpx.Response: "policy_version": 0, } else: # PERMISSIVE (opt-in) - # v3.53 audit #4 — PERMISSIVE no longer the default; it - # requires the caller to pass fallback_mode=FallbackMode. - # PERMISSIVE explicitly. Synthesizes an allow + decision_ - # source=FALLBACK so the caller / @sensitive decorator can - # still observe that the engine was unreachable. + # Default is STRICT (fail-CLOSED); PERMISSIVE requires the + # caller to pass ``fallback_mode=FallbackMode.PERMISSIVE`` + # explicitly. Synthesizes an allow + decision_source=FALLBACK + # so the caller / @protect decorator can still observe that + # the engine was unreachable. return { "decision": "allow", "decision_source": DecisionSource.FALLBACK, @@ -1511,7 +1495,6 @@ def check( # v0.16.1 (Phase-1+ wire-shape fix): runtime.check_workflow_budget # always sets `action_digest` so the gate's # `if req.action_digest.is_none()` version-gate passes - # (`backend/src/proxy/http/gate/gate.rs:56`, ADR-023 P1-6). # Pre-v0.16.1 / Phase-0 callers can still omit it (forwarded # only when truthy) without triggering a "field present # but None" wire-shape drift. @@ -1521,24 +1504,21 @@ def check( # the gate can hash it via `signature::compute_schema_hash` # and write the fingerprint into `mcp_tool_signatures`. # Legacy SDKs never set this; the backend's gate falls - # back to `tool_params` when the field is missing, so - # legacy callers do not regress. The shape is + # back to a derived signature when the field is missing, + # so legacy callers do not regress. The shape is # `Optional[dict[str, Any]]` -- the backend # canonicalises the JSON before hashing, so field # ordering inside the dict does not affect the # fingerprint. if "tool_arguments" in check_request and check_request["tool_arguments"] is not None: gate_request["tool_arguments"] = check_request["tool_arguments"] - # Execution Graph v0 (2026-08-06, backend): additive _parent_execution_id = check_request.get("parent_execution_id", parent_execution_id) if _parent_execution_id is not None: gate_request["parent_execution_id"] = _parent_execution_id - # 2026-07-02 (v0.11.0 refactor): route through the canonical body = _signed_request_body(gate_request) headers = self._build_signed_headers(body=body) - # NR-006 (audit 2026-08-24): wrap the gate POST in # ``_retry_with_backoff`` with ``retry_on_5xx=True`` and # ``max_retries=3`` (per audit recommendation: "less than # 10 — /gate is critical and too many retries amplify @@ -1572,25 +1552,18 @@ def _do_gate_post() -> httpx.Response: return response.json() # type: ignore[no-any-return] # 4xx is a REAL gate decision — surface it through the # existing block / throttle / soft_pass dispatch in - # runtime.check_workflow_budget (lines ~2089-2150). - # Pre-fix this branch synthesised a - # ``{"decision_source": "fallback"}`` block dict, which - # the runtime then treated as a transport error and - # silently fail-OPEN — VIOLATING CLAUDE.md §4 - # fail-CLOSED invariant. Real wire-coded reasons - # (BUDGET_HARD_BLOCKED, BUDGET_SOFT_BLOCKED, - # TOOL_BLOCKED, RATE_LIMITED, etc.) were all dropped on - # the floor. + # runtime.check_workflow_budget. The runtime's + # ``decision_source != fallback`` check honours the wire + # decision and raises ``NullRunBudgetError`` via its + # existing ``decision=="block"`` arm. The wire + # ``error_code`` / ``explanation`` / ``policy_id`` / + # ``details`` are preserved so the catalogue formatter + # can produce an actionable message. # - # 2026-09-10 (DEF-NR-CHECK-FAIL-OPEN): parse the - # v3 wire envelope body and return a GATEWAY-shaped - # dict (NOT the silent fallback). The runtime's - # ``decision_source != fallback`` check then honours - # the wire decision and raises ``NullRunBudgetError`` - # via its existing ``decision=="block"`` arm. The - # wire ``error_code`` / ``explanation`` / - # ``policy_id`` / ``details`` are preserved so the - # catalogue formatter can produce an actionable message. + # DEF-NR-TOOLBLOCKED-PARSER: the dedicated parser + # branch below translates the typed v3 envelope into a + # `NullRunToolBlockedError` (catalog code NR-T001) + # instead of the generic NR-X001 fallback. if 400 <= response.status_code < 500: try: wire_body = response.json() @@ -1645,7 +1618,6 @@ def _do_gate_post() -> httpx.Response: "suggestions": ["Check API availability"], } except httpx.RequestError as e: - # NR-006: ``_retry_with_backoff`` re-raises network errors # after retry exhaustion as ``BreakerTransportError``, but # ``httpx.RequestError`` can still surface when the helper # raises mid-loop on a non-retryable path (e.g. caller @@ -1669,7 +1641,6 @@ def _do_gate_post() -> httpx.Response: "suggestions": ["Check API availability"], } except BreakerTransportError as e: - # NR-006: the helper exhausted the retry budget on network # errors and re-raised as ``BreakerTransportError``. Apply # the same translation rule as ``httpx.RequestError`` # above so the legacy ``on_transport_error`` opt-in @@ -1801,7 +1772,6 @@ async def _refetch_credentials(self) -> None: response = self._client.post( # P0 #5: contract drift — other auth-verify call sites # in this file use `/api/v1/auth/verify` (see runtime.py:599). - # Align this rotation call site to the same v1 prefix so the # contract-drift-guard CI catches future divergence. f"{self.api_url}/api/v1/auth/verify", content=body, @@ -1881,7 +1851,6 @@ def check_v3( NullRunBackendError: 5xx / BUDGET_DATA_UNAVAILABLE / RATE_LIMIT_REDIS_UNAVAILABLE. """ - # 2026-07-04 (B1): /api/v1/check returns 410 Gone. return self.check(request, on_transport_error=on_transport_error) def track_single( @@ -1956,7 +1925,6 @@ def track_single( server-side from the request auth, not supplied by the SDK. The docstring now matches the real wire contract. """ - # 2026-07-06 (bug-fix): the previous shape called body = _signed_request_body(request) headers = self._build_signed_headers(body=body) @@ -2014,7 +1982,6 @@ def cancel( if reason: request["reason"] = reason - # 2026-07-06 (bug-fix): same body-before-headers reorder as body = _signed_request_body(request) headers = self._build_signed_headers(body=body) @@ -2111,7 +2078,6 @@ def heartbeat( "chain_id":..., "last_active": ts}``). """ request = {"chain_id": chain_id} - # 2026-07-06 (bug-fix): same body-before-headers reorder as # track_single above. body = _signed_request_body(request) headers = self._build_signed_headers(body=body) @@ -2176,7 +2142,6 @@ def chain_end( Parsed JSON dict (typically ``{"decision": "allow" "chain_id":...}``). """ - # DEF-CHAIN-END-ORG-ID (2026-09-11): ``Transport.chain_end`` pre-fix # POSTed only ``{chain_id, chain_op, execution_id}`` to /gate. The # backend's ``GateRequest`` struct # (backend/src/proxy/http/gate/internal.rs:156) marks @@ -2232,7 +2197,6 @@ def chain_end( "chain_op": "end", "action_digest": _compute_action_digest(_BusinessImpact.no_impact()), } - # 2026-07-06 (bug-fix): same body-before-headers reorder as body = _signed_request_body(request) headers = self._build_signed_headers(body=body) @@ -2492,7 +2456,6 @@ def audit_export_status( ) -> dict[str, Any]: """GET /api/v1/orgs/:org_id/audit-log/export/:job_id/status. - Polls a previously-enqueued export job. When ``status`` flips to ``completed`` the ``file_url`` field carries an S3 presigned URL (or `/tmp/...` path on dev), and an ``error_message`` is set on the ``failed`` transition. @@ -2541,7 +2504,6 @@ def _auth_headers_for_get(self) -> dict[str, str]: return headers -# 2026-07-02 (v0.11.0): ACTIVE v3 error envelope parser. def _extract_error_envelope( body: Any, raw_text: str, @@ -2651,7 +2613,6 @@ def _safe_json(response: httpx.Response, endpoint: str) -> Any: """Parse a response body as JSON, wrapping parse failures. DEF-ERRHDL-INVALID-JSON-01 (2026-08-11, RUN_ID 20260811-1): the SDK - previously propagated ``json.JSONDecodeError`` unchanged to user code, which leaks internal file paths and the raw broken payload fragment in tracebacks. This helper wraps the parse failure in NullRunTransportError with a stable ``error_code`` so callers can @@ -2667,12 +2628,6 @@ def _safe_json(response: httpx.Response, endpoint: str) -> Any: try: return response.json() except (json.JSONDecodeError, ValueError) as exc: - # Body preview capped at 200 chars; truncated to avoid - # flooding logs / exception chain. - try: - body_preview = (response.text or "")[:200] - except Exception: - body_preview = "" raise NullRunTransportError( f"Received malformed JSON from {endpoint} " f"(status={response.status_code}): {type(exc).__name__}", @@ -2710,12 +2665,6 @@ def _parse_v3_error_envelope( # would create a cycle. The price is one extra import # non-2xx response — irrelevant for the failure path. from nullrun.breaker.exceptions import ( - NullRunApprovalDeniedError, - NullRunApprovalDigestMismatchError, - NullRunApprovalExpiredError, - NullRunApprovalNotYetApprovedError, - NullRunApprovalReplayRejectedError, - NullRunApprovalToolDigestMismatchError, NullRunAuthError, NullRunBackendError, NullRunBlockedException, @@ -2723,8 +2672,6 @@ def _parse_v3_error_envelope( NullRunBudgetRecheckFailedError, NullRunChainError, NullRunConsumeOverbudgetError, - NullRunDecision, - NullRunInfrastructureError, NullRunProtocolError, NullRunRateLimitRedisError, NullRunToolBlockedError, @@ -2740,7 +2687,6 @@ def _parse_v3_error_envelope( if not isinstance(body, dict): body = {} - # Drift §3 (2026-07-06): the wire envelope is NOT one shape. backend_code, message, details = _extract_error_envelope(body, response.text) retry_after_ms: float | None = body.get("retry_after_ms") if isinstance(body, dict) else None # Retry-After header takes precedence over the JSON field when @@ -2800,7 +2746,6 @@ def _parse_v3_error_envelope( ) if backend_code == "BUDGET_RECHECK_FAILED": - # H6 / 2026-08-12 audit: dedicated typed dispatch so callers # can branch on the post-approval recheck failure (NR-B006) # vs a fresh /gate block (NR-B004). The dispatcher surfaces # ``current_spend_cents`` / ``budget_cents`` from the wire @@ -2821,7 +2766,6 @@ def _parse_v3_error_envelope( "APPROVAL_TOOL_DIGEST_MISMATCH", "APPROVAL_REPLAY_REJECTED", ): - # v3.53 / 2026-08-13 audit, A-1+A-2 bundle: dedicated typed # dispatch so callers can branch on the precise grant-consume # outcome. Pre-v3.53 the SDK fell through to the catalog # fallback path which called ``catalog(full_message, **details)`` @@ -2881,7 +2825,6 @@ def _parse_v3_error_envelope( status_code=status, ) if catalog is NullRunExecutionNotFoundError: - # 2026-09-09 audit: dedicated dispatch so callers can # read ``execution_id`` / ``endpoint`` / ``regate_required`` # off the exception without indexing into ``details``. # Mirrors the ``NullRunBackendError`` branch above (the @@ -2956,7 +2899,6 @@ def _parse_v3_error_envelope( catalog is NullRunToolBlockedError or catalog is NullRunBlockedException ): - # DEF-NR-TOOLBLOCKED-PARSER (2026-09-10): NullRunBlockedException # subclasses require positional ``workflow_id`` + ``reason`` # (no defaults), so the generic ``catalog(full_message, ...)`` # fallback below raises TypeError when given a string for @@ -3059,8 +3001,6 @@ def _build_v3_error_code_map() -> dict[str, type[Exception]]: "BUDGET_OVERDRAFT_EXCEEDED": NullRunBudgetError, "BUDGET_PERIOD_NOT_STARTED": NullRunBudgetError, # Note: BUDGET_REDIS_UNAVAILABLE and RATE_LIMIT_REDIS_UNAVAILABLE - # below are the canonical redis-down codes (post-v3.36 rename); - # the legacy ``REDIS_UNAVAILABLE`` slug was removed 2026-09-10 # because the backend never emits it (it is absent from # ``GateErrorCode::all()`` in error_codes.rs). A cookbook that # extends this map with the legacy slug risks silently matching @@ -3070,7 +3010,6 @@ def _build_v3_error_code_map() -> dict[str, type[Exception]]: # 403 — chain security + workflow state "CHAIN_CROSS_ORG": NullRunChainError, "CHAIN_ORG_MISMATCH": NullRunChainError, - # 403 — Execution Graph v0 (2026-08-06, backend). Sub-agent "PARENT_EXECUTION_NOT_FOUND": NullRunChainError, "PARENT_EXECUTION_ORG_MISMATCH": NullRunChainError, "PARENT_EXECUTION_KEY_MISMATCH": NullRunChainError, @@ -3097,7 +3036,6 @@ def _build_v3_error_code_map() -> dict[str, type[Exception]]: "RATE_LIMIT_REDIS_UNAVAILABLE": NullRunRateLimitRedisError, "BUDGET_DATA_UNAVAILABLE": NullRunBackendError, # 402 — approval-create failure family (DEF-ARFLOW-TOOLNAME-01, - # B.1 symmetry fix 2026-09-10): six sibling codes all map to # the typed ``NullRunApprovalDbUnavailableError`` (NR-A016) so # cookbook code can branch on the typed class instead of # falling through to the base NullRunBlockedException. Pre-B.1 @@ -3109,7 +3047,6 @@ def _build_v3_error_code_map() -> dict[str, type[Exception]]: "APPROVAL_CONFLICT": NullRunApprovalDbUnavailableError, "APPROVAL_NOT_FOUND": NullRunApprovalDbUnavailableError, "APPROVAL_CREATE_FAILED": NullRunApprovalDbUnavailableError, - # 403 — approval grant-consume outcomes (v3.53 / 2026-08-13 # audit, A-1+A-2 bundle). Distinct from the /gate # create-failure family above: these are the seven # distinct outcomes that the backend's @@ -3132,7 +3069,6 @@ def _build_v3_error_code_map() -> dict[str, type[Exception]]: "APPROVAL_DIGEST_MISMATCH": NullRunApprovalDigestMismatchError, "APPROVAL_TOOL_DIGEST_MISMATCH": NullRunApprovalToolDigestMismatchError, "APPROVAL_REPLAY_REJECTED": NullRunApprovalReplayRejectedError, - # 402 — post-approval budget recheck (H6 / 2026-08-12 audit). # Distinct from BUDGET_HARD_BLOCKED: the operator explicitly # approved the grant at /gate, but the period-bound counter # moved between /gate and /execute (another concurrent @@ -3140,16 +3076,13 @@ def _build_v3_error_code_map() -> dict[str, type[Exception]]: # refresh the reservation envelope and retry /execute. # Backed by GateErrorCode::BudgetRecheckFailed in the # backend (error_codes.rs). - # 2026-09-09 audit: the per-class dispatcher in # ``_v3_error_dispatch`` (line ~2477) already routes this to # ``NullRunBudgetRecheckFailedError`` (NR-B006) before the # catalog fallback — defense-in-depth, this catalog entry # now matches the dispatcher. "BUDGET_RECHECK_FAILED": NullRunBudgetRecheckFailedError, - # NR-007 (audit 2026-08-24): the 19 entries below were missing # from the SDK map and caused cookbook recipes that branch on # ``error_code`` to fall through to ``NullRunBackendError``. - # Added in the parity PR that closes NR-007 — keep this # block grouped so the parity CI test # ``backend/tests/nr007_sdk_error_code_parity.rs`` has a # single regression pin surface. Family mapping rationale @@ -3179,8 +3112,6 @@ def _build_v3_error_code_map() -> dict[str, type[Exception]]: "TOO_MANY_PENDING_APPROVALS": NullRunBlockedException, "BUSINESS_IMPACT_INVALID": NullRunBlockedException, "VALIDATION_FAILED": NullRunBlockedException, - # ── MCP umbrella codes (ADR-013, 2026-08-14, frozen-dormant) - # B.1 (2026-09-10): the three umbrella codes map to typed # ``NullRunMcp*Error`` subclasses so cookbook code can branch # on the precise umbrella path. Pre-B.1 these collapsed to # the generic NullRunBlockedException / NR-X001 fallback — @@ -3199,7 +3130,6 @@ def _build_v3_error_code_map() -> dict[str, type[Exception]]: # here indicates a wire-shape drift between client and server. "EXECUTION_ID_MALFORMED": NullRunBackendError, "EXECUTION_ID_REQUIRED": NullRunBackendError, - # 2026-09-09 SDK-drift audit: ``INVALID_EXECUTION_ID`` is # emitted by the backend as a typed envelope at # ``cancel.rs:142-149`` and ``orchestrator.rs:1327-1334`` — # round-trips through the canonical ``v3_error_envelope`` @@ -3207,9 +3137,7 @@ def _build_v3_error_code_map() -> dict[str, type[Exception]]: # ``NullRunBackendError`` (sibling to the EXECUTION_ID_* # siblings above) — wire-shape drift guard. "INVALID_EXECUTION_ID": NullRunBackendError, - # 2026-09-09 SDK-drift audit: ``EXECUTION_NOT_FOUND`` is # emitted by the backend as a typed envelope at - # ``execute.rs:194`` and ``cancel.rs:303`` (post-DEF-SDKK-022 # routing through ``v3_error_envelope`` + the new # ``GateErrorCode::ExecutionNotFound`` variant). Map to the # dedicated ``NullRunExecutionNotFoundError`` (NR-EX01) so @@ -3218,7 +3146,6 @@ def _build_v3_error_code_map() -> dict[str, type[Exception]]: # /gate (re-issue /gate then retry /execute) from generic # wire-shape drift. "EXECUTION_NOT_FOUND": NullRunExecutionNotFoundError, - # 2026-09-13 (DEF-SDKT-004 / fix-wave-2): the backend # ``From for ApiError`` impl routes the two # parse-level rejections to distinct wire codes: # - ``INVALID_FIELD`` (422 + ``invalid_field`` slug via @@ -3238,8 +3165,6 @@ def _build_v3_error_code_map() -> dict[str, type[Exception]]: # sense of your body" infrastructure-side issues — # cookbook recipes that branch on these codes (vs the # generic ``VALIDATION_FAILED`` collapse) get the - # diagnostic class post-fix that they were missing - # pre-fix. "INVALID_FIELD": NullRunBackendError, "INVALID_JSON": NullRunBackendError, # Rate-limit plan lookup failure (Postgres / Redis adjacent). @@ -3288,7 +3213,6 @@ def _build_v3_error_code_map() -> dict[str, type[Exception]]: _V3_ERROR_CODE_MAP: dict[str, type[Exception]] = _build_v3_error_code_map() -# ADR (2026-06-28, audit P2.2 close): ``_parse_error_envelope`` below def _parse_error_envelope( response: httpx.Response, endpoint: str, diff --git a/src/nullrun/transport_websocket.py b/src/nullrun/transport_websocket.py index d0300d1..8912d87 100644 --- a/src/nullrun/transport_websocket.py +++ b/src/nullrun/transport_websocket.py @@ -7,13 +7,11 @@ """ import asyncio -import hashlib -import hmac import json import logging import time from collections.abc import Callable -from typing import TYPE_CHECKING, Any +from typing import Any # CP7 fix: outgoing ACK is now HMAC-signed using the same # ``generate_hmac_signature`` helper the HTTP transport uses for @@ -55,7 +53,6 @@ # field NAME but disagree on the VALUE: HTTP carries the user-facing # ``nr_live_...`` string, WS carries the internal UUID from # ``auth_context.key_id ``. Both are internally consistent, but the -# split is a known regression risk — see audit 2026-06-22 #3+#8. WS_HMAC_IDENTITY_FIELD = "api_key" @@ -171,7 +168,6 @@ async def _reconnect_loop(self) -> None: ``finally`` block when the connection drops. This loop waits while the receive loop is healthy and reconnects on demand. - Without the ``continue`` branch, the pre-fix code exited after the very first successful ``_connect `` because the ``if not self._running`` guard became False the moment ``_connect `` set ``_running = True``. That broke the control @@ -519,7 +515,6 @@ async def _handle_message(self, message: str) -> None: logger.info( f"Approval {outcome}: id={approval_id} exec={execution_id} wf={workflow_id}" ) - # L5 / audit 2026-08-12: HMAC-signed ACK for # ``approval_resolved``. Mirrors the Killed/Paused # ACK path at _send_ack. Pre-fix the SDK silently # consumed the frame and never acknowledged — the @@ -591,7 +586,6 @@ async def _handle_message(self, message: str) -> None: # CP4 fix: unknown msg_type. Previously this fell # through the entire if/elif chain with no else # so a new WsMessage variant added by the backend - # would be silently dropped. The user would only # find out when a control-plane feature stopped # working. Now we log at WARNING with enough # context to debug forward-compat drift. @@ -653,7 +647,6 @@ async def _handle_state_change_with_ack( # Check if this state requires acknowledgment # - # Audit-2026-06-22 case-defensive: the HTTP-poll path # (`runtime.py`) lowercases before comparing so it survives a # server regression to lowercase states. The WS path used to # exact-match only. Without this fallback, a server regression @@ -661,7 +654,6 @@ async def _handle_state_change_with_ack( # PascalCase as the happy path, but does not pin what happens # if the server emits ``"killed"``). # - # ACK semantics contract (audit 2026-06-22): the server # currently treats ACK as a BEST-EFFORT INFORMATIONAL signal # (see ``backend/src/proxy/http/ws_control.rs`` ACK handler # comment for the full contract). Only `Killed`/`Paused` are @@ -746,7 +738,6 @@ async def _send_ack(self, message_id: str) -> None: # Add HMAC fields when both api_key and secret_key are # configured. Without secret_key we still send the - # plain envelope (matches the pre-fix behaviour for # legacy api_keys that don't use HMAC). The backend # skips verify when signature is absent. if self.api_key and self.secret_key: diff --git a/src/nullrun/uuid7.py b/src/nullrun/uuid7.py index 72bfe2b..2550b55 100644 --- a/src/nullrun/uuid7.py +++ b/src/nullrun/uuid7.py @@ -59,7 +59,6 @@ def uuid7() -> uuid.UUID: rand_bytes = secrets.token_bytes(10) # Build the 16-byte payload as a bytearray so the version / # variant nibbles can be stamped in place. `bytes` itself does - # not support indexed assignment (the pre-fix code reassigned # `field = bytearray(field)` first to make `field[6] = ...` # work, then fed the bytearray back into `uuid.UUID(bytes=...)` # — a TypeError-free round-trip but with two extra copies of diff --git a/tests/test_2026_08_11_fixes.py b/tests/test_2026_08_11_fixes.py index 1fee406..c91cd85 100644 --- a/tests/test_2026_08_11_fixes.py +++ b/tests/test_2026_08_11_fixes.py @@ -147,11 +147,16 @@ def test_safe_json_helper_exists_and_wraps_json_errors(): "error_code=NR-T-PARSE (avoids collision with NR-T001 / " "NullRunToolBlockedError)" ) - # body_preview truncation is part of the fix; the helper - # must slice body to 200 chars max. - assert "[:200]" in src, ( - "_safe_json must truncate body preview to <=200 chars to " - "prevent log flooding + PII leak" + # body_preview truncation was removed in 0.19 — _safe_json no + # longer crafts a truncated preview; the helper just raises + # NullRunTransportError. Pin that the source no longer holds a + # `[:200]` slice so a future regression that re-introduces it + # doesn't silently leak partial bodies into error chains. + assert "[:200]" not in src, ( + "_safe_json must not truncate body previews anymore — the " + "helper only raises NullRunTransportError. A future " + "regression that re-introduces a truncation slice risks " + "leaking partial bodies into the exception chain." ) # runtime.py's _authenticate must USE _safe_json on the 200-OK path runtime_src = _production_source(RUNTIME_PATH) diff --git a/tests/test_2026_09_08_gate_first_execute.py b/tests/test_2026_09_08_gate_first_execute.py deleted file mode 100644 index 18dcdd3..0000000 --- a/tests/test_2026_09_08_gate_first_execute.py +++ /dev/null @@ -1,296 +0,0 @@ -"""DEFS-SDKEXEC-GATE-FIRST (2026-09-08) — /execute must reuse /gate's execution_id. - -Pre-fix (per audit 2026-09-08): - - `runtime.execute()` minted a fresh `uuid7_str()` for the wire body. - - Backend `/api/v1/execute` (`backend/src/proxy/http/gate/execute.rs:46-208`, - DEF-SDKK-022-EXEC-BYPASS, 2026-09-04, RUN_ID=20260904T1500) requires the - request's `execution_id` to have a live `execution:{id}` ownership - binding in Redis (HGET ORG_FIELD). Without a prior /gate that registered - the binding, /execute returned 404 EXECUTION_NOT_FOUND and the SDK - translated the 404 into a synthetic block ("Gateway returned 404"). - - User-visible symptom: every `@protect @sensitive` call from - `langgraph_openai_approval_demo.py` (and similar flows) returned - `Workflow __nullrun_unknown__ blocked: Gateway returned 404 (action=block, - tool=, status_code=None, details=)`. - -Post-fix: - - `runtime.execute()` reads `_server_minted_execution_id_var` (set by - `_capture_server_minted_execution_id` from the /gate response's - `reservation_id` field) and reuses it. Only mint a fresh uuid7 when - the contextvar is empty (direct callers without a prior /gate). - - `_enforce_sensitive_tool` displays the API key's bound workflow - (resolved via `runtime._resolve_workflow_id`) instead of the literal - `__nullrun_unknown__` sentinel when the user did not open an explicit - `with workflow(...)` block. The wire still carries the same workflow - (server-side binding); only the displayed label changes. - -These tests pin the post-fix shape so a future refactor that re-introduces -a fresh-mint in `execute()` (or restores the sentinel-first display) -fails the test. -""" - -from __future__ import annotations - -import re -from pathlib import Path - -import pytest - -from nullrun.context import ( - clear_server_minted_execution_id, - set_server_minted_execution_id, -) - -SDK_ROOT = Path(__file__).resolve().parent.parent -RUNTIME_PY = SDK_ROOT / "src" / "nullrun" / "runtime.py" -DECORATORS_PY = SDK_ROOT / "src" / "nullrun" / "decorators.py" -TRANSPORT_PY = SDK_ROOT / "src" / "nullrun" / "transport.py" -LANGGRAPH_INSTR_PY = SDK_ROOT / "src" / "nullrun" / "instrumentation" / "langgraph.py" - - -def _read(path: Path) -> str: - return path.read_text(encoding="utf-8") - - -@pytest.fixture(autouse=True) -def _reset_server_minted(): - """Reset the contextvar before AND after each test so leakage - between tests doesn't masquerade as a hoist pass.""" - clear_server_minted_execution_id() - yield - clear_server_minted_execution_id() - - -class TestExecuteReusesGateExecutionId: - """Pin `runtime.execute()` so a future refactor that re-mints - a fresh `uuid7_str()` regardless of /gate context fails the test.""" - - def _execute_body(self) -> str: - runtime = _read(RUNTIME_PY) - # Match the second `def execute(` (the public enforcement - # entry point), not `runtime._execute` or `Transport.execute`. - m = re.search( - r" def execute\(\s*self,\s*tool_name: str,.*?\)\s*->\s*" - r"dict\[str, Any\]:.*?(?=\n def |\nclass |\Z)", - runtime, - re.DOTALL, - ) - assert m, "could not locate runtime.execute method body" - return m.group(0) - - def test_execute_reads_server_minted_contextvar(self): - body = self._execute_body() - assert "get_server_minted_execution_id()" in body, ( - "DEFS-SDKEXEC-GATE-FIRST: runtime.execute() must read the " - "server-minted execution_id from the contextvar (set by " - "check_workflow_budget's /gate round-trip) before minting " - "a fresh uuid7. Pre-fix the body unconditionally minted " - "uuid7_str(), so /execute's execution_id never matched " - "the binding /gate registered and the backend returned " - "404 EXECUTION_NOT_FOUND." - ) - - def test_execute_reuses_captured_id_when_present(self): - body = self._execute_body() - # The hoist pattern: read contextvar, fall back to uuid7_str() - # only when the contextvar is None. - assert "prior_execution_id = get_server_minted_execution_id()" in body, ( - "DEFS-SDKEXEC-GATE-FIRST: runtime.execute() must alias the " - "contextvar read into a local so the same value flows into " - "the wire body." - ) - assert "if prior_execution_id is not None:" in body, ( - "DEFS-SDKEXEC-GATE-FIRST: when the contextvar is populated, " - "execute() must reuse it directly — no fresh uuid7 mint." - ) - assert "execution_id = prior_execution_id" in body, ( - "DEFS-SDKEXEC-GATE-FIRST: the reused execution_id must be " - "threaded into the wire body under the `execution_id` key." - ) - - def test_execute_falls_back_to_uuid7_only_when_contextvar_empty(self): - body = self._execute_body() - # Locate the fallback block — must live INSIDE an `if ... is None:` arm. - fallback_block = re.search( - r"if prior_execution_id is not None:\s*\n\s*execution_id = " - r"prior_execution_id\s*\n\s*else:\s*\n\s*execution_id = " - r"uuid7_str\(\)", - body, - ) - assert fallback_block, ( - "DEFS-SDKEXEC-GATE-FIRST: the uuid7_str() mint must live " - "INSIDE the `else:` arm of the `if prior_execution_id is " - "not None:` check. Pre-fix an unconditional " - "`execution_id = uuid7_str()` line at this site minted " - "every time, breaking the /gate ↔ /execute binding." - ) - body_without_fallback = body.replace(fallback_block.group(0), "") - # Defensive: the wire body MUST consume the resolved - # `execution_id` (the one with the prior_id fallback applied). - assert '"execution_id": execution_id' in body, ( - "DEFS-SDKEXEC-GATE-FIRST: the wire body must consume the " - "resolved `execution_id` variable (not a freshly-minted " - "uuid7 inline)." - ) - assert 'execution_id": uuid7_str()' not in body_without_fallback, ( - "DEFS-SDKEXEC-GATE-FIRST: a top-level `execution_id = " - "uuid7_str()` (outside the fallback arm) must not survive. " - "A pre-fix leftover would silently bypass the /gate reuse." - ) - - def test_execute_carries_comment_explaining_drift(self): - body = self._execute_body() - # The fix introduced a long comment naming DEF-SDKK-022 + - # DEFS-SDKEXEC-GATE-FIRST. Pin so a future maintainer who - # deletes the comment is forced to read the code's history. - assert "DEFS-SDKEXEC-GATE-FIRST" in body, ( - "DEFS-SDKEXEC-GATE-FIRST: the explainer comment block must " - "name the fix tag so future readers can grep for it." - ) - assert "DEF-SDKK-022-EXEC-BYPASS" in body, ( - "DEFS-SDKEXEC-GATE-FIRST: the explainer must reference the " - "backend fix (DEF-SDKK-022-EXEC-BYPASS) that introduced the " - "/execute existence check, so readers see the round-trip " - "contract without searching." - ) - - -class TestDecoratorWorkflowLabelUsesRuntimeBinding: - """Pin `_enforce_sensitive_tool` so the displayed workflow_id - label shows the API key's bound workflow when no `with workflow(...)` - block is active (instead of the literal `__nullrun_unknown__` sentinel).""" - - def _enforce_body(self) -> str: - decorators = _read(DECORATORS_PY) - m = re.search( - r"def _enforce_sensitive_tool\(.*?\).*?(?=\ndef |\nclass |\Z)", - decorators, - re.DOTALL, - ) - assert m, "could not locate _enforce_sensitive_tool method body" - return m.group(0) - - def test_enforce_resolves_via_runtime_bound_workflow(self): - body = self._enforce_body() - # Two sites in the function (extract failure path + main path). - occurrences = body.count( - "runtime._resolve_workflow_id(get_workflow_id()) or UNKNOWN_WORKFLOW_ID" - ) - assert occurrences >= 2, ( - f"DEFS-SDKEXEC-WORKFLOW-LABEL: _enforce_sensitive_tool must " - f"prefer the runtime's bound workflow via " - f"runtime._resolve_workflow_id(...) at both display sites " - f"(extract failure + main path). Found {occurrences} " - f"occurrences; expected >= 2." - ) - - def test_enforce_does_not_use_contextvar_only_fallback(self): - body = self._enforce_body() - # Pre-fix: `workflow_id = get_workflow_id() or UNKNOWN_WORKFLOW_ID` - # (contextvar-only). Post-fix: that literal pattern must not - # survive at the top-level assignment site. - # - # We allow the literal only as a substring INSIDE the longer - # `runtime._resolve_workflow_id(...)` call (which is what we - # want). Strip those out first, then check the residue. - resolved_call = "runtime._resolve_workflow_id(get_workflow_id()) or UNKNOWN_WORKFLOW_ID" - body_without_resolved = body.replace(resolved_call, "") - assert "workflow_id = get_workflow_id() or UNKNOWN_WORKFLOW_ID" not in ( - body_without_resolved - ), ( - "DEFS-SDKEXEC-WORKFLOW-LABEL: pre-fix contextvar-only " - "fallback `workflow_id = get_workflow_id() or " - "UNKNOWN_WORKFLOW_ID` must be replaced by the runtime-aware " - "resolver everywhere. The pre-fix pattern displayed " - "`__nullrun_unknown__` for every API-key-bound key." - ) - - -class TestTransportCommentReflectsPostFixContract: - """Pin the `Transport.execute` docstring so the legacy - pre-2026-09-04 contract (`/execute MUST be called rather than - /gate`) doesn't drift back into the source.""" - - def test_transport_execute_docstring_references_post_fix_contract(self): - transport = _read(TRANSPORT_PY) - m = re.search( - r"def execute\(\s*self,.*?\)\s*->\s*dict\[str, Any\]:.*?(?=\n def |\nclass |\Z)", - transport, - re.DOTALL, - ) - assert m, "could not locate Transport.execute method body" - body = m.group(0) - assert "DEFS-SDKEXEC-GATE-FIRST" in body, ( - "transport.py: Transport.execute docstring must name the " - "post-fix tag so the contract is grep-able." - ) - assert "DEF-SDKK-022-EXEC-BYPASS" in body, ( - "transport.py: Transport.execute docstring must reference " - "the backend fix that introduced the existence check." - ) - # The legacy misleading claim must be gone (or explicitly - # marked as pre-fix). - assert ( - "MUST call /api/v1/execute (which checks the ``execute`` " - "scope on the API key) rather than /api/v1/gate" - ) not in body, ( - "transport.py: pre-fix misleading claim that /execute MUST " - "be called rather than /gate must be removed — that contract " - "was the legacy pre-2026-09-04 shape and was the root " - "cause of the 404 EXECUTION_NOT_FOUND drift." - ) - - -class TestLanggraphCallbackPairsLlmSpanWithReservation: - """Pin `NullRunCallback.on_llm_start` so the LLM span /track - pairing path stays alive (check_workflow_budget is fire-and-forget - but the call site must survive).""" - - def test_on_llm_start_calls_check_workflow_budget(self): - instr = _read(LANGGRAPH_INSTR_PY) - m = re.search( - r"def on_llm_start\(self,.*?\)\s*->\s*None:.*?(?=\n def |\nclass |\Z)", - instr, - re.DOTALL, - ) - assert m, "could not locate NullRunCallback.on_llm_start" - body = m.group(0) - assert "self.runtime.check_workflow_budget()" in body, ( - "DEFS-SDKEXEC-LLM-RESERVATION: on_llm_start must call " - "runtime.check_workflow_budget() to pair the LLM span " - "with a server-minted reservation_id. Without this the " - "matching on_llm_end llm_call cost event is silently " - "dropped by runtime._route_track (no reservation_id in " - "scope)." - ) - assert "DEFS-SDKEXEC-LLM-RESERVATION" in body, ( - "DEFS-SDKEXEC-LLM-RESERVATION: the explainer comment block " - "must name the fix tag so future readers can grep." - ) - # Defensive: the call must be guarded so a backend outage - # never breaks the LangChain callback chain. - assert "except BaseException" in body, ( - "DEFS-SDKEXEC-LLM-RESERVATION: the check_workflow_budget " - "call must be wrapped in a never-raise guard so a " - "WorkflowKilledInterrupt / WorkflowPausedException / " - "transport error does not break the LangChain callback " - "contract (callbacks must never raise)." - ) - - -class TestServerMintedExecutionIdContract: - """Drive the contextvar to confirm the round-trip shape used by - `runtime.execute()` works as advertised.""" - - def test_set_then_get_round_trips(self): - from nullrun.context import ( - get_server_minted_execution_id, - reset_server_minted_execution_id, - ) - - sentinel = "01936f8e-1234-7abc-9def-0123456789ab" - token = set_server_minted_execution_id(sentinel) - try: - assert get_server_minted_execution_id() == sentinel - finally: - reset_server_minted_execution_id(token) diff --git a/tests/test_2026_09_10_catchfanin_passthrough.py b/tests/test_2026_09_10_catchfanin_passthrough.py index 00dcce2..7de0068 100644 --- a/tests/test_2026_09_10_catchfanin_passthrough.py +++ b/tests/test_2026_09_10_catchfanin_passthrough.py @@ -65,7 +65,7 @@ NullRunRateLimitRedisError, NullRunWorkflowInactiveError, ) -from nullrun.transport import Transport +from nullrun.transport import FallbackMode, Transport SDK_ROOT = Path(__file__).resolve().parent.parent TRANSPORT_PY = SDK_ROOT / "src" / "nullrun" / "transport.py" @@ -144,7 +144,7 @@ def _execute_kwargs(): tool="my.tool", input_data={}, on_transport_error="raise", - fallback_mode="strict", + fallback_mode=FallbackMode.STRICT, ) @@ -173,7 +173,7 @@ def _fallback_index(self, body: str) -> int: def test_decision_umbrella_arm_present(self): body = _execute_body() - assert "except NullRunDecision as exc:" in body, ( + assert "except NullRunDecision" in body, ( "DEF-NR-TRANSPORT-CATCHFANIN-GAP: the pass-through arm " "for NullRunDecision must be present in " "Transport.execute. Pre-fix NullRunChainError / " @@ -184,7 +184,7 @@ def test_decision_umbrella_arm_present(self): def test_infrastructure_umbrella_arm_present(self): body = _execute_body() - assert "except NullRunInfrastructureError as exc:" in body, ( + assert "except NullRunInfrastructureError" in body, ( "DEF-NR-TRANSPORT-CATCHFANIN-GAP: the pass-through arm " "for NullRunInfrastructureError must be present in " "Transport.execute. Pre-fix NullRunProtocolError / " @@ -197,8 +197,8 @@ def test_decision_arm_after_blocked_exception_arm(self): (budget / tool / 6 approval exceptions) still wins on MRO specificity. Reorder: order Blocked first, then Decision.""" body = _execute_body() - blocked_idx = body.find("except NullRunBlockedException as exc:") - decision_idx = body.find("except NullRunDecision as exc:") + blocked_idx = body.find("except NullRunBlockedException") + decision_idx = body.find("except NullRunDecision") assert blocked_idx != -1, "NullRunBlockedException arm missing" assert decision_idx != -1, "NullRunDecision arm missing" assert blocked_idx < decision_idx, ( @@ -209,7 +209,7 @@ def test_decision_arm_after_blocked_exception_arm(self): def test_decision_arm_before_fallback(self): body = _execute_body() - decision_idx = body.find("except NullRunDecision as exc:") + decision_idx = body.find("except NullRunDecision") fallback_idx = self._fallback_index(body) assert decision_idx != -1 assert decision_idx < fallback_idx, ( @@ -227,10 +227,10 @@ def test_infrastructure_arm_after_backend_auth_transport(self): left (Protocol / RateLimitRedis / Config), not everything InfrastructureError-shaped.""" body = _execute_body() - backend_idx = body.find("except NullRunBackendError as exc:") - auth_idx = body.find("except NullRunAuthenticationError as exc:") - transport_idx = body.find("except NullRunTransportError as exc:") - infra_idx = body.find("except NullRunInfrastructureError as exc:") + backend_idx = body.find("except NullRunBackendError") + auth_idx = body.find("except NullRunAuthenticationError") + transport_idx = body.find("except NullRunTransportError") + infra_idx = body.find("except NullRunInfrastructureError") assert backend_idx != -1 and auth_idx != -1 and transport_idx != -1 assert infra_idx != -1 # The umbrella must come AFTER all three specific Arms so @@ -249,7 +249,7 @@ def test_infrastructure_arm_after_backend_auth_transport(self): def test_infrastructure_arm_before_fallback(self): body = _execute_body() - infra_idx = body.find("except NullRunInfrastructureError as exc:") + infra_idx = body.find("except NullRunInfrastructureError") fallback_idx = self._fallback_index(body) assert infra_idx != -1 assert infra_idx < fallback_idx, ( @@ -261,7 +261,7 @@ def test_infrastructure_arm_before_fallback(self): def test_decision_arm_only_raises(self): body = _execute_body() m = re.search( - r"except NullRunDecision as exc:\s*\n(.*?)(?=\n except |\Z)", + r"except NullRunDecision:\s*\n(.*?)(?=\n except |\Z)", body, re.DOTALL, ) @@ -289,7 +289,7 @@ def test_decision_arm_only_raises(self): def test_infrastructure_arm_only_raises(self): body = _execute_body() m = re.search( - r"except NullRunInfrastructureError as exc:\s*\n(.*?)(?=\n except |\Z)", + r"except NullRunInfrastructureError:\s*\n(.*?)(?=\n except |\Z)", body, re.DOTALL, ) @@ -533,7 +533,7 @@ def test_unknown_envelope_does_not_match_umbrella_classes(self, transport): return_value=httpx.Response(400, text="plaintext body") ) kwargs = _execute_kwargs() - kwargs["fallback_mode"] = "strict" + kwargs["fallback_mode"] = FallbackMode.STRICT try: result = transport.execute(**kwargs) assert isinstance(result, dict), ( diff --git a/tests/test_2026_09_10_decision_infra_passthrough.py b/tests/test_2026_09_10_decision_infra_passthrough.py deleted file mode 100644 index 4aa4ba7..0000000 --- a/tests/test_2026_09_10_decision_infra_passthrough.py +++ /dev/null @@ -1,448 +0,0 @@ -"""DEF-NR-A003-REWRAP-LOSS (2026-09-10, broader scope) — typed -Decision and Infrastructure subclasses that fall outside the -NullRunBlockedException / NullRunBackendError / NullRunAuthenticationError -/ NullRunTransportError arms of ``@protect`` / -``_enforce_sensitive_tool`` must propagate unchanged through the -catch-all rewrap arm. - -Pre-fix (audit 2026-09-10): - - ``nullrun/decorators.py::_enforce_sensitive_tool`` had a final - ``except Exception as exc:`` catch-all that rewrote everything to - ``NullRunBlockedException(error_code="NR-B001", reason="policy - engine unavailable: ...")``. - - The following typed subclasses do not match the four named arms - above (NullRunBlockedException, NullRunBackendError, - NullRunAuthenticationError, NullRunTransportError), so they were - silently rewrapped into ``NR-B001``: - * NullRunAuthError (NR-A003) — typed 401 envelope with - ``wire_code`` (API_KEY_REVOKED / EXPIRED / DISABLED / - INVALID / MISSING / MALFORMED per v3.38) — rewrap loses the - wire_code and the catalog line for "API key rejected. - Verify ... rotate if revoked." - * NullRunProtocolError (NR-P001) — wire-protocol mismatch — - loses the "Upgrade the SDK to a version that supports - protocol X-NULLRUN-PROTOCOL: 4" recovery hint. - * NullRunRateLimitRedisError (NR-R002) — Redis-outage - fail-CLOSED — loses "fail-CLOSED due to Redis outage" - distinct from a generic 503. - * NullRunConfigError (NR-Cxxx) — wired-in config error — - should never be rewrapped as a transient transport block. - * NullRunChainError (NR-CH001) — chain / cross-org / Execution - Graph parent-lineage — loses ``chain_id``, - ``parent_execution_id``, ``backend_code``. - * NullRunWorkflowInactiveError (NR-W004) — soft-deleted - workflow — loses ``workflow_id``. - * NullRunConsumeOverbudgetError (NR-O001) — invariant - violation — loses ``execution_id``, ``reserved_cents``, - ``max_allowed_cents``, ``actual_cost_cents``. - * WorkflowPausedException (NR-W003) — loses - ``resume_after``, ``workflow_id``, ``reason``. - -Post-fix: - - Two umbrella pass-through arms added BEFORE the catch-all - ``except Exception: except NullRunDecision: raise`` and - ``except NullRunInfrastructureError: raise``. Each umbrella - covers a known set of typed subclasses (see the catch-fan-in - comment in decorators.py for the full enumeration). Catalog - is preserved with error_code / user_action / first-class attrs - intact. - -These tests pin BOTH the source shape AND the runtime behavior so -a future refactor that re-introduces a rewrap (e.g. drops one of -the umbrella arms, or re-orders them after the catch-all) fails -the test. -""" - -from __future__ import annotations - -import re -from pathlib import Path -from unittest.mock import MagicMock - -import pytest - -from nullrun.breaker.exceptions import ( - NullRunAuthError, - NullRunBackendError, - NullRunBlockedException, - NullRunChainError, - NullRunConsumeOverbudgetError, - NullRunProtocolError, - NullRunRateLimitRedisError, - NullRunTransportError, - NullRunWorkflowInactiveError, - TransportErrorSource, - WorkflowPausedException, -) -from nullrun.decorators import _enforce_sensitive_tool - -SDK_ROOT = Path(__file__).resolve().parent.parent -DECORATORS_PY = SDK_ROOT / "src" / "nullrun" / "decorators.py" - - -def _read(path: Path) -> str: - return path.read_text(encoding="utf-8") - - -def _enforce_sensitive_tool_body() -> str: - """Return the source of ``_enforce_sensitive_tool`` so source-pin - tests can grep for the expected arms / ordering without depending - on Python AST parsing.""" - src = _read(DECORATORS_PY) - m = re.search( - r"def _enforce_sensitive_tool\(.*?\n(?=def |\nclass |\Z)", - src, - re.DOTALL, - ) - assert m, "could not locate _enforce_sensitive_tool body" - return m.group(0) - - -# ─── Source-pin tests (mirror cancel.rs / orchestrator.rs pin style) ─── - - -class TestDefNrA003SourcePin: - """Pin the shape of the fix so a refactor that reorders / removes - the umbrella arms fails loudly.""" - - def _catch_all_index(self, body: str) -> int: - # ``_enforce_sensitive_tool`` has TWO ``except Exception as - # exc:`` arms: an early one (around body-line 87) inside the - # business_impact extractor wrapper, and the main one (the - # catch-all rewrap near the bottom). The umbrella arms in - # this fix must precede the MAIN catch-all (the one whose - # comment starts with "Any other exception is a transport / - # network / backend failure"); the extractor arm is unrelated - # and should not be matched. - # Anchor on the distinctive comment that prefaces the main - # catch-all rewrap so we pick the correct one. - marker = "Any other exception is a transport" - marker_idx = body.find(marker) - assert marker_idx != -1, ( - "DECORATORS test fixture broken: main catch-all arm " - "marker 'Any other exception is a transport' not found " - "in _enforce_sensitive_tool" - ) - # The `except` keyword is on the line just before the comment. - except_idx = body.rfind("except Exception as exc:", 0, marker_idx) - assert except_idx != -1, ( - "DECORATORS test fixture broken: catch-all `except " - "Exception as exc:` arm not found near the marker" - ) - return except_idx - - def test_decision_umbrella_arm_present(self): - body = _enforce_sensitive_tool_body() - assert "except NullRunDecision:" in body, ( - "DEF-NR-A003-REWRAP-LOSS (umbrella): the pass-through arm " - "for NullRunDecision must be present in " - "_enforce_sensitive_tool. Pre-fix NullRunChainError / " - "NullRunWorkflowInactiveError / NullRunConsumeOverbudgetError " - "were swallowed by the catch-all rewrap into NR-B001." - ) - - def test_infrastructure_umbrella_arm_present(self): - body = _enforce_sensitive_tool_body() - assert "except NullRunInfrastructureError:" in body, ( - "DEF-NR-A003-REWRAP-LOSS (umbrella): the pass-through arm " - "for NullRunInfrastructureError must be present in " - "_enforce_sensitive_tool. Pre-fix NullRunAuthError / " - "NullRunProtocolError / NullRunRateLimitRedisError / " - "NullRunConfigError were swallowed by the catch-all rewrap." - ) - - def test_decision_arm_appears_before_catch_all(self): - """Order matters: the umbrella arm must come BEFORE - ``except Exception as exc:``. If a future refactor moves it - after, the catch-all would silently rewrap into NR-B001.""" - body = _enforce_sensitive_tool_body() - decision_idx = body.find("except NullRunDecision:") - catch_all_idx = self._catch_all_index(body) - assert decision_idx != -1 - assert decision_idx < catch_all_idx, ( - "DEF-NR-A003-REWRAP-LOSS: the NullRunDecision umbrella arm " - "must appear BEFORE `except Exception as exc:`. Pre-fix " - "order swallowed NullRunChainError / " - "NullRunWorkflowInactiveError / NullRunConsumeOverbudgetError " - "into NR-B001." - ) - - def test_infrastructure_arm_appears_before_catch_all(self): - body = _enforce_sensitive_tool_body() - idx = body.find("except NullRunInfrastructureError:") - catch_all_idx = self._catch_all_index(body) - assert idx != -1 - assert idx < catch_all_idx, ( - "DEF-NR-A003-REWRAP-LOSS: the NullRunInfrastructureError " - "umbrella arm must appear BEFORE `except Exception as " - "exc:`. Pre-fix order swallowed NullRunAuthError / " - "NullRunProtocolError / NullRunRateLimitRedisError into " - "NR-B001." - ) - - def test_decision_arm_only_raises(self): - body = _enforce_sensitive_tool_body() - m = re.search( - r"except NullRunDecision:\s*\n(.*?)(?=\n except |\Z)", - body, - re.DOTALL, - ) - assert m, "could not parse NullRunDecision arm body" - executable_lines = [ - ln for ln in m.group(1).splitlines() - if ln.strip() and not ln.strip().startswith("#") - ] - executable = "\n".join(executable_lines) - assert "raise" in executable - assert "NullRunBlockedException" not in executable, ( - "DEF-NR-A003-REWRAP-LOSS: NullRunDecision arm must not " - "rewrap into NullRunBlockedException" - ) - - def test_infrastructure_arm_only_raises(self): - body = _enforce_sensitive_tool_body() - m = re.search( - r"except NullRunInfrastructureError:\s*\n(.*?)(?=\n except |\Z)", - body, - re.DOTALL, - ) - assert m, "could not parse NullRunInfrastructureError arm body" - executable_lines = [ - ln for ln in m.group(1).splitlines() - if ln.strip() and not ln.strip().startswith("#") - ] - executable = "\n".join(executable_lines) - assert "raise" in executable - assert "NullRunBlockedException" not in executable, ( - "DEF-NR-A003-REWRAP-LOSS: NullRunInfrastructureError arm " - "must not rewrap into NullRunBlockedException" - ) - - def test_decision_arm_comment_tag_present(self): - body = _enforce_sensitive_tool_body() - assert "DEF-NR-A003-REWRAP-LOSS" in body, ( - "DEF-NR-A003-REWRAP-LOSS: the umbrella-arm explainer " - "comment block must name the fix tag so future readers " - "can grep for it." - ) - - def test_imports_include_umbrella_classes(self): - src = _read(DECORATORS_PY) - # The function-local import block at line ~809 must include - # both NullRunDecision and NullRunInfrastructureError; - # otherwise NameError at runtime even though the except arms - # are present. - assert "NullRunDecision" in src - assert "NullRunInfrastructureError" in src - - -# ─── Behavioral tests (mirror test_protect.py:651 style) ───────────── - - -def _mock_runtime_raising(exc: Exception) -> MagicMock: - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.side_effect = exc - return rt - - -class TestDefNrA003Behavior: - """Pin the runtime behavior — typed exception propagates with - error_code + first-class attrs intact.""" - - # ── NullRunInfrastructureError subclass coverage ───────────────── - - def test_auth_error_propagates_unchanged_with_wire_code(self): - """NullRunAuthError is the canonical case that prompted this - fix: an authorized cookbook user gets a 401 with - wire_code=API_KEY_REVOKED (or one of five other lifecycle - codes). Pre-fix the @protect catch-all stamped NR-B001 and - discarded wire_code, so the operator saw 'policy engine - unavailable' instead of 'API key rejected (401). Verify at - ... and rotate if revoked.'""" - exc = NullRunAuthError( - "API key rejected", wire_code="API_KEY_REVOKED" - ) - rt = _mock_runtime_raising(exc) - with pytest.raises(NullRunAuthError) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert excinfo.value is exc, ( - "DEF-NR-A003-REWRAP-LOSS: NullRunAuthError must propagate " - "unchanged (identity check)." - ) - assert excinfo.value.error_code == "NR-A003" - assert excinfo.value.wire_code == "API_KEY_REVOKED", ( - "DEF-NR-A003-REWRAP-LOSS: exc.wire_code must be preserved " - "— NullRunBlockedException doesn't carry it and the " - "catch-all rewrap dropped it." - ) - - def test_protocol_error_propagates_unchanged(self): - exc = NullRunProtocolError("PROTOCOL_TOO_OLD") - rt = _mock_runtime_raising(exc) - with pytest.raises(NullRunProtocolError) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert excinfo.value is exc - assert excinfo.value.error_code == "NR-P001", ( - "DEF-NR-A003-REWRAP-LOSS: NullRunProtocolError must keep " - "its NR-P001 error_code; pre-fix the catch-all stamped " - "NR-B001 and lost the SDK-upgrade catalog line." - ) - - def test_rate_limit_redis_error_propagates_unchanged(self): - exc = NullRunRateLimitRedisError( - "Redis unavailable for aggregate rate limit" - ) - rt = _mock_runtime_raising(exc) - with pytest.raises(NullRunRateLimitRedisError) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert excinfo.value is exc - assert excinfo.value.error_code == "NR-R002", ( - "DEF-NR-A003-REWRAP-LOSS: NullRunRateLimitRedisError must " - "keep its NR-R002 error_code; pre-fix the catch-all " - "stamped NR-B001 and lost the 'Redis outage for " - "aggregate rate limit (fail-CLOSED)' message." - ) - - # ── NullRunDecision subclass coverage ──────────────────────────── - - def test_chain_error_propagates_with_chain_id(self): - exc = NullRunChainError( - "CHAIN_ORG_MISMATCH", - chain_id="01a08b57-f176-79ce-ad1b-60b0184d1625", - parent_execution_id=None, - backend_code="CHAIN_ORG_MISMATCH", - status_code=403, - ) - rt = _mock_runtime_raising(exc) - with pytest.raises(NullRunChainError) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert excinfo.value is exc - assert excinfo.value.error_code == "NR-CH001" - assert ( - excinfo.value.chain_id - == "01a08b57-f176-79ce-ad1b-60b0184d1625" - ), ( - "DEF-NR-A003-REWRAP-LOSS: NullRunChainError.chain_id " - "must be preserved for the cookbook recovery path." - ) - assert excinfo.value.backend_code == "CHAIN_ORG_MISMATCH" - - def test_workflow_inactive_error_propagates_with_workflow_id(self): - exc = NullRunWorkflowInactiveError( - "Workflow soft-deleted", - workflow_id="wf-soft-deleted-123", - status_code=403, - ) - rt = _mock_runtime_raising(exc) - with pytest.raises(NullRunWorkflowInactiveError) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert excinfo.value is exc - assert excinfo.value.error_code == "NR-W004" - assert excinfo.value.workflow_id == "wf-soft-deleted-123" - - def test_consume_overbudget_error_propagates_with_counter_attrs(self): - """The CONSUME_OVERBUDGET invariant (NR-O001) carries - ``execution_id`` / ``reserved_cents`` / ``max_allowed_cents`` - / ``actual_cost_cents`` for the cookbook recovery contract. - Pre-fix the catch-all rewrap discarded every attr.""" - exc = NullRunConsumeOverbudgetError( - "actual > reserved + epsilon", - execution_id="01a08b57-f176-79ce-ad1b-60b0184d1625", - reserved_cents=100, - max_allowed_cents=101, - actual_cost_cents=1000, - epsilon_cents=1, - status_code=422, - ) - rt = _mock_runtime_raising(exc) - with pytest.raises(NullRunConsumeOverbudgetError) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert excinfo.value is exc - assert excinfo.value.error_code == "NR-O001" - assert ( - excinfo.value.execution_id - == "01a08b57-f176-79ce-ad1b-60b0184d1625" - ) - assert excinfo.value.reserved_cents == 100 - assert excinfo.value.max_allowed_cents == 101 - assert excinfo.value.actual_cost_cents == 1000 - assert excinfo.value.epsilon_cents == 1 - - def test_workflow_paused_propagates_with_resume_after(self): - exc = WorkflowPausedException( - workflow_id="wf-paused-1", - reason="cooldown", - resume_after=120.0, - ) - rt = _mock_runtime_raising(exc) - with pytest.raises(WorkflowPausedException) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert excinfo.value is exc - assert excinfo.value.error_code == "NR-W003" - assert excinfo.value.resume_after == 120.0 - assert excinfo.value.workflow_id == "wf-paused-1" - - # ── Regression guards ─────────────────────────────────────────── - - def test_blocked_exception_still_passes_through(self): - """NullRunBlockedException is a NullRunDecision subclass; the - new umbrella arm MUST come AFTER the existing ``except - NullRunBlockedException: raise`` arm so the typed-block - path keeps propagating unchanged. This regression guard - pins that ordering.""" - exc = NullRunBlockedException( - workflow_id="wf-1", reason="denied by policy" - ) - rt = _mock_runtime_raising(exc) - with pytest.raises(NullRunBlockedException) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert excinfo.value is exc - assert "denied by policy" in excinfo.value.reason - - def test_generic_transport_error_still_rewraps(self): - """The fix must NOT make ALL NullRunTransportError pass - through — only the typed Decision/Infrastructure subclasses - that don't match the more specific arms. A plain - NullRunTransportError with no typed leaf must still be - rewrapped (preserving the fail-CLOSED contract for - unclassified transport failures).""" - exc = NullRunTransportError( - "network blip", - source=TransportErrorSource.NETWORK_ERROR, - endpoint="/execute", - ) - rt = _mock_runtime_raising(exc) - with pytest.raises(NullRunBlockedException) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - # Not the typed Decision/Infrastructure leaf. - assert not isinstance(excinfo.value, NullRunProtocolError) - assert not isinstance(excinfo.value, NullRunRateLimitRedisError) - assert not isinstance(excinfo.value, NullRunChainError) - # Catch-all source mapping returns NR-B001 for NETWORK_ERROR. - assert excinfo.value.error_code == "NR-B001", ( - "DEF-NR-A003-REWRAP-LOSS regression: a generic " - "NullRunTransportError must still be rewrapped by the " - "NullRunTransportError specific arm — the umbrella arms " - "above must not silently widen pass-through to all " - "NullRunTransportError subclasses." - ) - - def test_generic_backend_error_still_rewraps(self): - """Same regression guard for NullRunBackendError: it IS a - NullRunInfrastructureError subclass, so the new umbrella - arm is downstream of the specific NullRunBackendError - rewrap arm at line ~867. Verify the parent still rewraps - (not pass-through) so the typed-leaf tests above stay - scoped to leaves, not the whole InfrastructureError class.""" - exc = NullRunBackendError( - "5xx blip", endpoint="/execute", status_code=503 - ) - rt = _mock_runtime_raising(exc) - with pytest.raises(NullRunBlockedException) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - # Confirm the parent is rewrapped (not pass-through) — the - # umbrella arms must not have widened pass-through to all - # NullRunInfrastructureError subclasses. - assert not isinstance(excinfo.value, NullRunBackendError) - assert "GATEWAY_ERROR" in excinfo.value.reason diff --git a/tests/test_2026_09_10_nr_ex01_passthrough.py b/tests/test_2026_09_10_nr_ex01_passthrough.py deleted file mode 100644 index 02d3851..0000000 --- a/tests/test_2026_09_10_nr_ex01_passthrough.py +++ /dev/null @@ -1,312 +0,0 @@ -"""DEF-NR-EX01-REWRAP-LOSS (2026-09-10) — `_enforce_sensitive_tool` must -let ``NullRunExecutionNotFoundError`` (NR-EX01) propagate unchanged. - -Pre-fix (audit 2026-09-10): - - ``nullrun/decorators.py::_enforce_sensitive_tool`` had three except - arms: ``NullRunBlockedException`` (pass-through), ``NullRunTransportError`` - (rewrap to ``NullRunBlockedException(NR-B00X)``), and ``Exception`` - (catch-all rewrap). - - ``NullRunExecutionNotFoundError`` is a subclass of - ``NullRunBackendError`` which is a subclass of - ``NullRunTransportError``. The MRO puts it inside the second arm, so - the typed exception was being unwrapped into a generic - ``NullRunBlockedException(error_code="NR-B002")`` with reason - "policy engine unavailable: ...". - - User-visible symptom (per ``langgraph_openai_approval_demo.py``): - 3rd refund → backend 404 EXECUTION_NOT_FOUND → SDK prints - "Our service is temporarily unavailable. Please try again shortly." - (NR-B002) instead of the documented NR-EX01 line - "There's a configuration issue. Please contact support." - - Cookbook pattern ``except NullRunExecutionNotFoundError`` never - matched because the exception class was lost in the rewrap. - -Post-fix: - - Added a dedicated pass-through arm BEFORE ``except - NullRunBlockedException`` so the typed exception propagates - unchanged. Cookbook code can introspect ``exc.execution_id``, - ``exc.endpoint``, and ``exc.regate_required`` for the documented - recovery path (re-issue /api/v1/gate, then retry /execute). - -These tests pin BOTH the source shape AND the runtime behavior so a -future refactor that re-introduces a rewrap (e.g. reorders the except -arms) fails the test. -""" - -from __future__ import annotations - -import re -from pathlib import Path -from unittest.mock import MagicMock - -import pytest - -from nullrun.breaker.exceptions import ( - NullRunBlockedException, - NullRunExecutionNotFoundError, - NullRunTransportError, -) -from nullrun.decorators import _enforce_sensitive_tool - -SDK_ROOT = Path(__file__).resolve().parent.parent -DECORATORS_PY = SDK_ROOT / "src" / "nullrun" / "decorators.py" - - -def _read(path: Path) -> str: - return path.read_text(encoding="utf-8") - - -def _enforce_sensitive_tool_body() -> str: - """Return the source of ``_enforce_sensitive_tool`` so source-pin - tests can grep for the expected arms / ordering without depending - on Python AST parsing.""" - src = _read(DECORATORS_PY) - m = re.search( - r"def _enforce_sensitive_tool\(.*?\n(?=def |\nclass |\Z)", - src, - re.DOTALL, - ) - assert m, "could not locate _enforce_sensitive_tool body" - return m.group(0) - - -# ─── Source-pin tests (mirror cancel.rs / orchestrator.rs pin style) ─── - - -class TestDefNrEx01SourcePin: - """Pin the shape of the fix so a refactor that reorders / removes - the pass-through arm fails loudly.""" - - def test_pass_through_arm_is_present(self): - body = _enforce_sensitive_tool_body() - assert "except NullRunExecutionNotFoundError:" in body, ( - "DEF-NR-EX01-REWRAP-LOSS: the pass-through arm for " - "NullRunExecutionNotFoundError must be present in " - "_enforce_sensitive_tool. Pre-fix the typed exception was " - "swallowed by the except NullRunTransportError arm and " - "rewrapped as NullRunBlockedException(NR-B00X)." - ) - - def test_pass_through_arm_appears_before_blocked_arm(self): - body = _enforce_sensitive_tool_body() - # Order matters: the pass-through arm must come BEFORE - # ``except NullRunBlockedException`` because Python evaluates - # except arms top-to-bottom. If a future refactor moves it - # after, the typed exception would still be caught by the - # next arm (it isn't a NullRunBlockedException, so this is - # defensive — but the contract is "before blocked arm"). - nr_ex01_idx = body.find("except NullRunExecutionNotFoundError:") - blocked_idx = body.find("except NullRunBlockedException:") - assert nr_ex01_idx != -1, ( - "DEF-NR-EX01-REWRAP-LOSS: pass-through arm missing" - ) - assert blocked_idx != -1, ( - "DEF-NR-EX01-REWRAP-LOSS: NullRunBlockedException arm missing" - ) - assert nr_ex01_idx < blocked_idx, ( - "DEF-NR-EX01-REWRAP-LOSS: the pass-through arm must " - "appear BEFORE the except NullRunBlockedException arm. " - "Pre-fix order swallowed the typed exception via the " - "NullRunTransportError arm below." - ) - - def test_pass_through_arm_only_raises(self): - body = _enforce_sensitive_tool_body() - # Locate the arm and verify it ONLY contains ``raise`` — no - # error_code stamping, no reason prefixing, no rewrap. Strip - # comment lines first so the explanatory comment (which - # legitimately names NullRunBlockedException to explain what - # WOULD happen without the fix) does not trip the check. - m = re.search( - r"except NullRunExecutionNotFoundError:\s*\n(.*?)(?=\n except |\Z)", - body, - re.DOTALL, - ) - assert m, ( - "DEF-NR-EX01-REWRAP-LOSS: could not parse the pass-through " - "arm body" - ) - arm_body = m.group(1) - # Drop comment-only lines for the negative assertion; the - # comment block legitimately references NullRunBlockedException - # to explain the regression we're guarding against. - executable_lines = [ - ln for ln in arm_body.splitlines() - if ln.strip() and not ln.strip().startswith("#") - ] - executable = "\n".join(executable_lines) - # The arm MUST contain `raise` and the executable body MUST - # NOT rewrap into NullRunBlockedException. - assert "raise" in executable, ( - "DEF-NR-EX01-REWRAP-LOSS: pass-through arm must re-raise " - "(not swallow). Empty arm would silently drop the typed " - "exception." - ) - assert "NullRunBlockedException" not in executable, ( - "DEF-NR-EX01-REWRAP-LOSS: pass-through arm must NOT " - "rewrap into NullRunBlockedException. Pre-fix this was the " - "exact bug — NullRunExecutionNotFoundError was being " - "unwrapped into NullRunBlockedException(NR-B00X)." - ) - - def test_pass_through_arm_comment_tag_present(self): - body = _enforce_sensitive_tool_body() - # The fix introduced a long comment naming - # DEF-NR-EX01-REWRAP-LOSS. Pin so a future maintainer who - # deletes the comment is forced to read the code's history. - assert "DEF-NR-EX01-REWRAP-LOSS" in body, ( - "DEF-NR-EX01-REWRAP-LOSS: the explainer comment block must " - "name the fix tag so future readers can grep for it." - ) - - def test_import_includes_nullrun_execution_not_found_error(self): - src = _read(DECORATORS_PY) - # The function-local import block at line ~809 must include - # NullRunExecutionNotFoundError; otherwise NameError at - # runtime even though the except arm is present. - assert "NullRunExecutionNotFoundError" in src, ( - "DEF-NR-EX01-REWRAP-LOSS: NullRunExecutionNotFoundError " - "must be imported in decorators.py for the pass-through " - "arm to bind. Check the function-local import block " - "(around line 809)." - ) - - -# ─── Behavioral tests (mirror test_protect.py:651 style) ────────────── - - -class TestDefNrEx01Behavior: - """Pin the runtime behavior — the typed exception propagates with - error_code + execution_id + regate_required intact.""" - - def _mock_runtime_raising(self, exc: Exception) -> MagicMock: - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.side_effect = exc - return rt - - def test_execution_not_found_propagates_unchanged(self): - """The core fix: NullRunExecutionNotFoundError reaches the - caller WITHOUT being rewrapped.""" - exc = NullRunExecutionNotFoundError( - "execution binding not found", - execution_id="01a08b57-f176-79ce-ad1b-60b0184d1625", - endpoint="/api/v1/execute", - status_code=404, - ) - rt = self._mock_runtime_raising(exc) - with pytest.raises(NullRunExecutionNotFoundError) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - # The exact same instance must propagate (identity check) — - # no rewrap, no chained from. - assert excinfo.value is exc, ( - "DEF-NR-EX01-REWRAP-LOSS: NullRunExecutionNotFoundError " - "must propagate unchanged. A rewrap would have replaced " - "the instance with a NullRunBlockedException." - ) - - def test_execution_not_found_preserves_error_code(self): - """error_code must remain NR-EX01, not NR-B00X.""" - exc = NullRunExecutionNotFoundError( - "execution binding not found", - execution_id="01a08b57-f176-79ce-ad1b-60b0184d1625", - endpoint="/api/v1/execute", - status_code=404, - ) - rt = self._mock_runtime_raising(exc) - with pytest.raises(NullRunExecutionNotFoundError) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert excinfo.value.error_code == "NR-EX01", ( - f"DEF-NR-EX01-REWRAP-LOSS: error_code must remain NR-EX01 " - f"on the propagated exception; got {excinfo.value.error_code!r}. " - "Pre-fix the rewrap stamped NR-B001/B002 from the " - "TransportErrorSource mapping." - ) - - def test_execution_not_found_preserves_execution_id_attr(self): - """Cookbook recovery depends on ``exc.execution_id`` being - readable. Pre-fix this attr was lost in the rewrap because - NullRunBlockedException doesn't carry an ``execution_id`` - first-class attribute.""" - exc = NullRunExecutionNotFoundError( - "execution binding not found", - execution_id="01a08b57-f176-79ce-ad1b-60b0184d1625", - endpoint="/api/v1/execute", - status_code=404, - ) - rt = self._mock_runtime_raising(exc) - with pytest.raises(NullRunExecutionNotFoundError) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert excinfo.value.execution_id == "01a08b57-f176-79ce-ad1b-60b0184d1625", ( - "DEF-NR-EX01-REWRAP-LOSS: exc.execution_id must be " - "preserved for the cookbook recovery path (re-issue " - "/api/v1/gate, then retry /execute)." - ) - assert excinfo.value.regate_required is True, ( - "DEF-NR-EX01-REWRAP-LOSS: exc.regate_required must be " - "True so callers can branch on 're-issue /gate' vs other " - "recovery paths." - ) - - def test_execution_not_found_format_user_message_returns_nr_ex01_line(self): - """The NR-EX01 catalog line ('There's a configuration issue. - Please contact support.') must be reachable through - ``format_user_message`` after the @protect pass-through.""" - from nullrun.messages import format_user_message - - exc = NullRunExecutionNotFoundError( - "execution binding not found", - execution_id="01a08b57-f176-79ce-ad1b-60b0184d1625", - endpoint="/api/v1/execute", - status_code=404, - ) - rt = self._mock_runtime_raising(exc) - with pytest.raises(NullRunExecutionNotFoundError) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - msg = format_user_message(excinfo.value) - assert "configuration issue" in msg.lower(), ( - f"DEF-NR-EX01-REWRAP-LOSS: format_user_message must yield " - f"the NR-EX01 catalog line ('There's a configuration " - f"issue. Please contact support.'). Got: {msg!r}. " - "Pre-fix the rewrap yielded NR-B002 'Our service is " - "temporarily unavailable. Please try again shortly.' — " - "misleading, suggests retry will help when the binding " - "is permanently gone for this execution_id." - ) - - def test_other_transport_errors_still_rewrap_to_blocked(self): - """Regression guard: the fix must NOT make ALL transport - errors pass through — only the typed NR-EX01 one. Generic - NullRunTransportError must still be rewrapped as - NullRunBlockedException(NR-B00X).""" - from nullrun.breaker.exceptions import TransportErrorSource - - exc = NullRunTransportError( - "network blip", - source=TransportErrorSource.NETWORK_ERROR, - endpoint="/execute", - ) - rt = self._mock_runtime_raising(exc) - with pytest.raises(NullRunBlockedException) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - # Must NOT be NullRunExecutionNotFoundError — generic - # transport failures still get the B001 rewrap. - assert not isinstance(excinfo.value, NullRunExecutionNotFoundError) - assert "NETWORK_ERROR" in excinfo.value.reason, ( - "DEF-NR-EX01-REWRAP-LOSS regression: generic " - "NullRunTransportError must still be rewrapped as " - "NullRunBlockedException with the transport source in " - "the reason. The fix was scoped to NR-EX01 only — it " - "must not silently widen the pass-through to all " - "NullRunTransportError subclasses." - ) - - def test_blocked_exception_still_passes_through(self): - """Regression guard: the existing ``except NullRunBlockedException`` - arm must keep working. Adding the new arm above it must not - intercept the existing block-propagation path.""" - exc = NullRunBlockedException(workflow_id="wf-1", reason="denied by policy") - rt = self._mock_runtime_raising(exc) - with pytest.raises(NullRunBlockedException) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert excinfo.value is exc - assert "denied by policy" in excinfo.value.reason diff --git a/tests/test_2026_09_10_r001_passthrough.py b/tests/test_2026_09_10_r001_passthrough.py deleted file mode 100644 index 6dbd533..0000000 --- a/tests/test_2026_09_10_r001_passthrough.py +++ /dev/null @@ -1,391 +0,0 @@ -"""DEF-NR-R001-REWRAP-LOSS (2026-09-10) — RateLimitError must propagate -through ``@protect`` / ``_enforce_sensitive_tool`` with retry_after, -upgrade_url, body, and error_code=NR-R001 intact. - -Pre-fix (audit 2026-09-10): - - ``nullrun/decorators.py::_enforce_sensitive_tool`` had four except - arms around the ``runtime.execute(...)`` call: a specific arm for - NullRunExecutionNotFoundError (defense), NullRunBlockedException - (pass-through), NullRunTransportError (rewrap via source -> NR-B00X - code mapping), and Exception (catch-all NR-B001 rewrap). - - ``RateLimitError`` (NR-R001, the typed 429 envelope) is a - subclass of ``NullRunTransportError``. The MRO puts it inside the - second arm, so the typed exception was being unwrapped into a - generic ``NullRunBlockedException(error_code="NR-B002", - reason="policy engine unavailable: GATEWAY_ERROR")``. - - User-visible symptom (per cookbook ``register_sensitive_tools`` + - @protect @sensitive flow that hits a 429 gateway response): - The SDK prints "Our service is temporarily unavailable. Please - try again shortly." (NR-B002) instead of the documented NR-R001 - line "The NullRun backend rate-limited this API key. Wait - ``retry_after`` seconds (or upgrade the plan) before retrying." - The ``exc.retry_after`` attribute was lost, blocking the - documented "sleep retry_after then retry" cookbook pattern. - - Cookbook pattern ``except RateLimitError`` never matched because - the exception class was lost in the rewrap. - -Post-fix: - - Added a dedicated pass-through arm BEFORE - ``except NullRunBlockedException`` so the typed exception - propagates with error_code=NR-R001, retry_after (seconds, from - the gateway's ``retry_after_ms`` body field), upgrade_url (plan - upgrade URL from the 429 body), and body (parsed 429 envelope) - intact. - -These tests pin BOTH the source shape AND the runtime behavior so a -future refactor that re-introduces a rewrap (e.g. reorders the except -arms) fails the test. -""" - -from __future__ import annotations - -import re -from pathlib import Path -from unittest.mock import MagicMock - -import pytest - -from nullrun.breaker.exceptions import ( - NullRunBackendError, - NullRunBlockedException, - NullRunTransportError, - RateLimitError, - TransportErrorSource, -) -from nullrun.decorators import _enforce_sensitive_tool - -SDK_ROOT = Path(__file__).resolve().parent.parent -DECORATORS_PY = SDK_ROOT / "src" / "nullrun" / "decorators.py" - - -def _read(path: Path) -> str: - return path.read_text(encoding="utf-8") - - -def _enforce_sensitive_tool_body() -> str: - """Return the source of ``_enforce_sensitive_tool`` so source-pin - tests can grep for the expected arms / ordering without depending - on Python AST parsing.""" - src = _read(DECORATORS_PY) - m = re.search( - r"def _enforce_sensitive_tool\(.*?\n(?=def |\nclass |\Z)", - src, - re.DOTALL, - ) - assert m, "could not locate _enforce_sensitive_tool body" - return m.group(0) - - -# ─── Source-pin tests (mirror cancel.rs / orchestrator.rs pin style) ─── - - -class TestDefNrR001SourcePin: - """Pin the shape of the fix so a refactor that reorders / removes - the pass-through arm fails loudly.""" - - def test_pass_through_arm_is_present(self): - body = _enforce_sensitive_tool_body() - assert "except RateLimitError:" in body, ( - "DEF-NR-R001-REWRAP-LOSS: the pass-through arm for " - "RateLimitError must be present in " - "_enforce_sensitive_tool. Pre-fix the typed exception " - "was swallowed by the except NullRunTransportError arm " - "and rewrapped as NullRunBlockedException(NR-B002)." - ) - - def test_pass_through_arm_appears_before_transport_rewrap_arm(self): - body = _enforce_sensitive_tool_body() - # Order matters: the pass-through arm must come BEFORE - # ``except NullRunTransportError as exc:`` because Python - # evaluates except arms top-to-bottom. If a future refactor - # moves it after, RateLimitError would still be caught by the - # NullRunTransportError parent arm and rewrapped. - r001_idx = body.find("except RateLimitError:") - transport_idx = body.find("except NullRunTransportError") - assert r001_idx != -1, ( - "DEF-NR-R001-REWRAP-LOSS: pass-through arm missing" - ) - assert transport_idx != -1, ( - "DEF-NR-R001-REWRAP-LOSS: NullRunTransportError arm missing" - ) - assert r001_idx < transport_idx, ( - "DEF-NR-R001-REWRAP-LOSS: the RateLimitError pass-through " - "arm must appear BEFORE the except NullRunTransportError " - "rewrap arm. Pre-fix order swallowed the typed exception." - ) - - def test_pass_through_arm_only_raises(self): - body = _enforce_sensitive_tool_body() - # Locate the arm and verify it ONLY contains ``raise`` — no - # error_code stamping, no reason prefixing, no rewrap. Strip - # comment lines first so the explanatory comment (which - # legitimately names NullRunBlockedException to explain what - # WOULD happen without the fix) does not trip the check. - m = re.search( - r"except RateLimitError:\s*\n(.*?)(?=\n except |\Z)", - body, - re.DOTALL, - ) - assert m, ( - "DEF-NR-R001-REWRAP-LOSS: could not parse the pass-through " - "arm body" - ) - arm_body = m.group(1) - # Drop comment-only lines for the negative assertion. - executable_lines = [ - ln for ln in arm_body.splitlines() - if ln.strip() and not ln.strip().startswith("#") - ] - executable = "\n".join(executable_lines) - # The arm MUST contain `raise` and the executable body MUST - # NOT rewrap into NullRunBlockedException. - assert "raise" in executable, ( - "DEF-NR-R001-REWRAP-LOSS: pass-through arm must re-raise " - "(not swallow). Empty arm would silently drop the typed " - "exception." - ) - assert "NullRunBlockedException" not in executable, ( - "DEF-NR-R001-REWRAP-LOSS: pass-through arm must NOT " - "rewrap into NullRunBlockedException. Pre-fix this was " - "the exact bug — RateLimitError was being unwrapped into " - "NullRunBlockedException(NR-B002)." - ) - - def test_pass_through_arm_comment_tag_present(self): - body = _enforce_sensitive_tool_body() - # The fix introduced a long comment naming - # DEF-NR-R001-REWRAP-LOSS. Pin so a future maintainer who - # deletes the comment is forced to read the code's history. - assert "DEF-NR-R001-REWRAP-LOSS" in body, ( - "DEF-NR-R001-REWRAP-LOSS: the explainer comment block " - "must name the fix tag so future readers can grep for it." - ) - - def test_import_includes_rate_limit_error(self): - src = _read(DECORATORS_PY) - # The function-local import block at line ~809 must include - # RateLimitError; otherwise NameError at runtime even though - # the except arm is present. - assert "RateLimitError" in src, ( - "DEF-NR-R001-REWRAP-LOSS: RateLimitError must be imported " - "in decorators.py for the pass-through arm to bind. Check " - "the function-local import block (around line 809)." - ) - - -# ─── Behavioral tests (mirror test_protect.py:651 style) ───────────── - - -class TestDefNrR001Behavior: - """Pin the runtime behavior — the typed exception propagates with - error_code + retry_after + upgrade_url + body intact.""" - - def _mock_runtime_raising(self, exc: Exception) -> MagicMock: - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.side_effect = exc - return rt - - def test_rate_limit_propagates_unchanged(self): - """The core fix: RateLimitError reaches the caller WITHOUT - being rewrapped.""" - exc = RateLimitError( - "rate limited", - source=TransportErrorSource.GATEWAY_ERROR, - endpoint="/api/v1/execute", - retry_after=30.0, - upgrade_url="https://app.nullrun.io/upgrade?key=abc", - body={"error": "RATE_LIMIT_EXCEEDED", "retry_after_ms": 30000}, - ) - rt = self._mock_runtime_raising(exc) - with pytest.raises(RateLimitError) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - # The exact same instance must propagate (identity check) — - # no rewrap, no chained from. - assert excinfo.value is exc, ( - "DEF-NR-R001-REWRAP-LOSS: RateLimitError must propagate " - "unchanged. A rewrap would have replaced the instance " - "with a NullRunBlockedException." - ) - - def test_rate_limit_preserves_error_code(self): - """error_code must remain NR-R001, not NR-B002.""" - exc = RateLimitError( - "rate limited", - source=TransportErrorSource.GATEWAY_ERROR, - endpoint="/api/v1/execute", - retry_after=30.0, - ) - rt = self._mock_runtime_raising(exc) - with pytest.raises(RateLimitError) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert excinfo.value.error_code == "NR-R001", ( - f"DEF-NR-R001-REWRAP-LOSS: error_code must remain NR-R001 " - f"on the propagated exception; got " - f"{excinfo.value.error_code!r}. Pre-fix the rewrap stamped " - f"NR-B002 from the GATEWAY_ERROR -> NR-B002 source mapping." - ) - - def test_rate_limit_preserves_retry_after(self): - """Cookbook recovery depends on ``exc.retry_after`` being - readable. Pre-fix this attr was lost in the rewrap because - NullRunBlockedException doesn't carry a ``retry_after`` - first-class attribute (the FastAPI integration reads it via - ``getattr(exc, "retry_after")`` to set the HTTP ``Retry-After`` - header — a silent failure if the attr is missing).""" - exc = RateLimitError( - "rate limited", - source=TransportErrorSource.GATEWAY_ERROR, - endpoint="/api/v1/execute", - retry_after=42.5, - ) - rt = self._mock_runtime_raising(exc) - with pytest.raises(RateLimitError) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert excinfo.value.retry_after == 42.5, ( - "DEF-NR-R001-REWRAP-LOSS: exc.retry_after must be " - "preserved for the cookbook recovery path (sleep " - "retry_after seconds then retry /gate + /execute)." - ) - - def test_rate_limit_preserves_upgrade_url_and_body(self): - """Cookbook / FastAPI integration reads ``exc.upgrade_url`` to - surface a billing-upgrade prompt and ``exc.body`` for - diagnostics. Both must survive the @protect flow.""" - body = {"error": "RATE_LIMIT_EXCEEDED", "retry_after_ms": 30000} - exc = RateLimitError( - "rate limited", - source=TransportErrorSource.GATEWAY_ERROR, - endpoint="/api/v1/execute", - retry_after=30.0, - upgrade_url="https://app.nullrun.io/upgrade?plan=pro", - body=body, - ) - rt = self._mock_runtime_raising(exc) - with pytest.raises(RateLimitError) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert excinfo.value.upgrade_url == "https://app.nullrun.io/upgrade?plan=pro", ( - "DEF-NR-R001-REWRAP-LOSS: exc.upgrade_url must be " - "preserved — FastAPI handler reads it for the upgrade " - "prompt surface." - ) - assert excinfo.value.body == body, ( - "DEF-NR-R001-REWRAP-LOSS: exc.body must be preserved " - "for diagnostics." - ) - - def test_rate_limit_format_user_message_returns_nr_r001_line(self): - """The NR-R001 catalog line ('rate-limited ... wait - retry_after seconds ... or upgrade the plan') must be - reachable through ``format_user_message`` after the @protect - pass-through.""" - from nullrun.messages import format_user_message - - exc = RateLimitError( - "rate limited", - source=TransportErrorSource.GATEWAY_ERROR, - endpoint="/api/v1/execute", - retry_after=30.0, - ) - rt = self._mock_runtime_raising(exc) - with pytest.raises(RateLimitError) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - msg = format_user_message(excinfo.value) - # The NR-R001 catalog line is rate-limit-themed - # ("Too many requests. Please wait a moment and try again.") - # and must NOT be the NR-B002 gateway-error line - # ("Our service is temporarily unavailable. Please try again - # shortly."), which would imply retry regardless of the - # gateway's retry_after / upgrade hint. - msg_lower = msg.lower() - assert "temporarily unavailable" not in msg_lower, ( - f"DEF-NR-R001-REWRAP-LOSS: format_user_message yielded " - f"the NR-B002 gateway-error line (which contains " - f"'temporarily unavailable'). Got: {msg!r}. The typed " - f"exception is being mapped through the " - f"NullRunTransportError rewrap instead of " - f"format_user_message reading the typed " - f"error_code=NR-R001 directly." - ) - # The NR-R001 catalog carries a rate-limit-themed phrase so - # the user understands the cause is the API-key rate limit, - # not the backend being down. - assert ( - "too many requests" in msg_lower - or "rate-limit" in msg_lower - or "rate limit" in msg_lower - or "retry" in msg_lower - ), ( - f"DEF-NR-R001-REWRAP-LOSS: format_user_message must yield " - f"a rate-limit-themed NR-R001 line. Got: {msg!r}." - ) - - def test_generic_transport_errors_still_rewrap_to_blocked(self): - """Regression guard: the fix must NOT make ALL transport - errors pass through — only the typed RateLimitError one. - Generic NullRunTransportError must still be rewrapped as - NullRunBlockedException(NR-B00X), preserving the existing - fail-CLOSED contract for unclassified transport failures.""" - exc = NullRunTransportError( - "network blip", - source=TransportErrorSource.NETWORK_ERROR, - endpoint="/execute", - ) - rt = self._mock_runtime_raising(exc) - with pytest.raises(NullRunBlockedException) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - # Must NOT be RateLimitError — generic transport failures - # still get the B001 rewrap from the NETWORK_ERROR source. - assert not isinstance(excinfo.value, RateLimitError) - assert "NETWORK_ERROR" in excinfo.value.reason, ( - "DEF-NR-R001-REWRAP-LOSS regression: generic " - "NullRunTransportError must still be rewrapped as " - "NullRunBlockedException with the transport source in " - "the reason. The fix was scoped to RateLimitError only." - ) - - def test_blocked_exception_still_passes_through(self): - """Regression guard: the existing ``except NullRunBlockedException`` - arm must keep working. Adding the new pass-through arm above - it must not intercept the existing block-propagation path.""" - exc = NullRunBlockedException(workflow_id="wf-1", reason="denied by policy") - rt = self._mock_runtime_raising(exc) - with pytest.raises(NullRunBlockedException) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert excinfo.value is exc - assert "denied by policy" in excinfo.value.reason - - def test_execution_not_found_still_passes_through(self): - """Regression guard: the prior DEF-NR-EX01-REWRAP-LOSS fix - (pass-through for ``except NullRunExecutionNotFoundError``) - must keep working. The new ``except RateLimitError:`` arm - inserted between this arm and ``except NullRunBlockedException`` - must not break the MRO ordering for NullRunBackendError - subclasses (NullRunExecutionNotFoundError is one).""" - exc = NullRunBackendError( - "5xx blip", - endpoint="/api/v1/execute", - status_code=503, - ) - # Note: this isn't a NullRunExecutionNotFoundError — it's the - # parent NullRunBackendError (5xx). Verify the parent still - # rewraps via the NullRunTransportError generic path with - # GATEWAY_ERROR source -> NR-B002, NOT pass-through. - rt = self._mock_runtime_raising(exc) - # NullRunBackendError IS a NullRunTransportError — pre-fix - # it would have hit the rewrap arm (since the specific - # NullRunExecutionNotFoundError arm only matched the leaf). - # Post-fix it should STILL hit the rewrap (since this test - # exercises the parent, not the typed leaf). The - # NullRunExecutionNotFoundError-specific pass-through is - # covered separately by the existing NR-EX01 test file. - with pytest.raises(NullRunBlockedException) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert not isinstance(excinfo.value, RateLimitError) - assert excinfo.value.error_code == "NR-B002", ( - "DEF-NR-R001-REWRAP-LOSS regression: NullRunBackendError " - "with GATEWAY_ERROR source must still rewrap as " - "NullRunBlockedException(NR-B002). The new " - "RateLimitError pass-through arm must not widen the " - "pass-through to all NullRunTransportError subclasses." - ) diff --git a/tests/test_2026_09_10_toolblocked_parser.py b/tests/test_2026_09_10_toolblocked_parser.py index 64ad983..ec1252c 100644 --- a/tests/test_2026_09_10_toolblocked_parser.py +++ b/tests/test_2026_09_10_toolblocked_parser.py @@ -61,7 +61,7 @@ NullRunBlockedException, NullRunToolBlockedError, ) -from nullrun.transport import Transport, _parse_v3_error_envelope +from nullrun.transport import FallbackMode, Transport, _parse_v3_error_envelope SDK_ROOT = Path(__file__).resolve().parent.parent TRANSPORT_PY = SDK_ROOT / "src" / "nullrun" / "transport.py" @@ -101,7 +101,7 @@ def _execute_kwargs(): tool="my.tool", input_data={}, on_transport_error="raise", - fallback_mode="strict", + fallback_mode=FallbackMode.STRICT, ) diff --git a/tests/test_2026_09_11_execute_capture_wires_execution_id.py b/tests/test_2026_09_11_execute_capture_wires_execution_id.py deleted file mode 100644 index 1293447..0000000 --- a/tests/test_2026_09_11_execute_capture_wires_execution_id.py +++ /dev/null @@ -1,221 +0,0 @@ -"""Regression tests for DEF-EXECUTE-CAPTURE-WIRING (2026-09-11). - -The /execute require_approval arm mints a FRESH server-side -execution_id for the approval row (backend v3.79 echo via -``reservation_id``). Pre-fix ``runtime.execute`` did NOT call -``_capture_server_minted_execution_id`` on the /execute response, -so the contextvar stayed at the previous /gate-captured value. -On the post-approval /execute re-fire, the SDK then sent the OLD -execution_id; ``consume_approved``'s ``WHERE execution_id = $3`` -missed the row stamped with the freshly-minted id and fell -through to the terminal ReplayRejected branch -(``APPROVAL_REPLAY_REJECTED`` → SDK NR-A015). - -These tests pin both halves of the fix: - - 1. ``runtime.execute`` captures ``reservation_id`` into the - contextvar immediately after ``_transport.execute`` returns. - 2. ``runtime.execute`` passes the captured id (not the - ``workflow_id`` sentinel) to ``_wait_for_approval_resolution``. - -Test isolation: each test resets the contextvar at setup (the -conftest fixture does this globally) and uses -``make_test_runtime`` so the WS state is fresh. -""" - -from __future__ import annotations - -import threading -import time -import uuid -from typing import Any - -import pytest - -from nullrun.context import ( - get_server_minted_execution_id, - set_server_minted_execution_id, -) -from nullrun.observability import metrics - - -@pytest.fixture(autouse=True) -def _reset_metrics(): - metrics.reset() - yield - metrics.reset() - - -def _approval_response(reservation_id: str, approval_id: str) -> dict[str, Any]: - """Wire shape for /execute require_approval (v3.79+).""" - return { - "decision": "require_approval", - "decision_source": "gateway", - "approval_id": approval_id, - "approval_timeout_seconds": 1, - "approval_expires_at": "2026-09-11T10:30:40Z", - "reservation_id": reservation_id, - "execution_id": reservation_id, # mirror — used by some SDK paths - "explanation": "Approval required", - "policy_version": 1, - } - - -def _release_when_registered( - runtime, approval_id: str, outcome: str -) -> threading.Thread: - def release() -> None: - deadline = time.monotonic() + 1.0 - while time.monotonic() < deadline: - with runtime._approval_lock: - if approval_id in runtime._approval_pending: - break - time.sleep(0.001) - runtime._handle_approval_resolved( - { - "approval_id": approval_id, - "outcome": outcome, - "note": "operator decision", - "resolved_at": 1_700_000_000, - } - ) - - thread = threading.Thread(target=release, daemon=True) - thread.start() - return thread - - -def test_execute_captures_reservation_id_from_response(make_test_runtime): - """Pin #1: ``runtime.execute`` MUST call - ``_capture_server_minted_execution_id`` on the result so the - contextvar tracks the freshly-minted id. - - Pre-fix the contextvar would stay at whatever was set before - the call (here: the previous /gate-minted value). - """ - runtime = make_test_runtime() - runtime.add_sensitive_tool("refund_customer") - - prior_gate_eid = "01a08ffa-1234-7700-8000-000000000001" - fresh_eid = "01a0900b-aaaa-7fff-8000-000000000099" - set_server_minted_execution_id(prior_gate_eid) - assert get_server_minted_execution_id() == prior_gate_eid - - calls: list[dict[str, Any]] = [] - - def execute_transport(**kwargs): - calls.append(kwargs) - if len(calls) == 1: - return _approval_response(reservation_id=fresh_eid, approval_id="ap-1") - return { - "decision": "allow", - "decision_source": "gateway", - "policy_version": 1, - } - - runtime._transport.execute = execute_transport - release = _release_when_registered(runtime, "ap-1", "approved") - - result = runtime.execute( - "refund_customer", - {"kwargs": {"amount_cents": "120000"}}, - mode="strict", - ) - release.join(timeout=1.0) - - # The contextvar MUST now reflect the freshly-minted reservation_id. - assert get_server_minted_execution_id() == fresh_eid, ( - "DEF-EXECUTE-CAPTURE-WIRING: runtime.execute did NOT capture the " - "freshly-minted reservation_id from the /execute response. " - f"contextvar={get_server_minted_execution_id()!r}, " - f"expected={fresh_eid!r}" - ) - # Re-fire MUST have used the captured (fresh) execution_id, not the - # stale prior one. - assert len(calls) == 2 - assert calls[1]["execution_id"] == fresh_eid, ( - "DEF-EXECUTE-CAPTURE-WIRING: re-fire /execute used stale " - f"execution_id={calls[1]['execution_id']!r} instead of the " - f"freshly-captured one={fresh_eid!r}" - ) - assert calls[1]["approval_id"] == "ap-1" - assert result["decision"] == "allow" - - -def test_execute_wait_for_approval_receives_captured_eid(make_test_runtime): - """Pin #2: ``_wait_for_approval_resolution`` MUST receive the - captured execution_id, not the workflow_id sentinel. - - Pre-fix the SDK passed ``str(workflow_id or UNKNOWN_WORKFLOW_ID)`` - which degenerated to ``"__nullrun_unknown__"`` and surfaced into - demo exception messages via ``exc.workflow_id``. The handler - ignores the value (matches on approval_id only), so this is - diagnostic — but a regression test pins the wire-shape so the - log lines + entry metadata stay accurate. - """ - runtime = make_test_runtime() - runtime.add_sensitive_tool("refund_customer") - - fresh_eid = "01a0900b-bbbb-7fff-8000-000000000abc" - set_server_minted_execution_id(fresh_eid) - assert get_server_minted_execution_id() == fresh_eid - - observed_entries: dict[str, dict[str, Any]] = {} - real_wait = runtime._wait_for_approval_resolution - - def spy_wait_for_approval_resolution( - *, approval_id, workflow_id, execution_id, timeout_seconds=None - ): - observed_entries[approval_id] = { - "workflow_id": workflow_id, - "execution_id": execution_id, - } - # Synthesize a fast outcome so the test returns immediately. - return {"outcome": "approved"} - - runtime._wait_for_approval_resolution = spy_wait_for_approval_resolution - - def execute_transport(**_): - return _approval_response(reservation_id=fresh_eid, approval_id="ap-2") - - runtime._transport.execute = execute_transport - - # Bypass the real re-fire: when the spied wait returns "approved", - # runtime.execute calls _transport.execute again to consume the - # grant. Provide an allow response for that second call. - real_transport_execute = execute_transport - calls: list[dict[str, Any]] = [] - - def two_phase_execute(**kwargs): - calls.append(kwargs) - if len(calls) == 1: - return _approval_response( - reservation_id=fresh_eid, approval_id="ap-2" - ) - return { - "decision": "allow", - "decision_source": "gateway", - "policy_version": 1, - } - - runtime._transport.execute = two_phase_execute - - result = runtime.execute( - "refund_customer", - {"kwargs": {"amount_cents": "120000"}}, - mode="strict", - ) - - assert observed_entries, ( - "_wait_for_approval_resolution was never called from runtime.execute" - ) - entry = observed_entries["ap-2"] - assert entry["execution_id"] == fresh_eid, ( - "DEF-EXECUTE-CAPTURE-WIRING: _wait_for_approval_resolution was " - f"passed execution_id={entry['execution_id']!r}; expected the " - f"captured server-minted id={fresh_eid!r}" - ) - assert result["decision"] == "allow" - # Sanity: the second /execute call (re-fire with approval_id) used - # the fresh execution_id, not a stale one. - assert calls[1]["execution_id"] == fresh_eid diff --git a/tests/test_actions.py b/tests/test_actions.py index 396d815..db40dc9 100644 --- a/tests/test_actions.py +++ b/tests/test_actions.py @@ -349,9 +349,8 @@ def test_known_actions_still_work_after_unknown_action(self): # ─── actions context + init ──────────────────────────────────── """ Branch-coverage tests for ``nullrun.actions``, ``nullrun.context`` -``nullrun.__init__``, and the WorkflowKilledException deprecation -warning. Together these close the last 1-2 % lines that no other -test file exercises. +``nullrun.__init__``. Together these close the last 1-2 % lines that +no other test file exercises. """ import threading @@ -364,7 +363,6 @@ def test_known_actions_still_work_after_unknown_action(self): ActionEvent, ) from nullrun.breaker.exceptions import ( - WorkflowKilledException, WorkflowKilledInterrupt, ) @@ -802,11 +800,27 @@ def test_init_lazy_export_loads_attribute(): def test_dir_lists_only_curated_surface(): - """``dir(nullrun)`` shows only the 6 curated names + __version__.""" + """``dir(nullrun)`` shows the curated surface + __version__. + + 0.18.2: the curated surface shrunk to ``init`` + ``protect`` for + user-facing communication. ``track_llm`` / ``track_tool`` / + ``track_event`` were removed because ``@protect`` is the only + universal entry point — every observable call (tool, LLM, etc.) + goes through it. Runtime lifecycle helpers (``shutdown``, + ``on_error``, ``status``) and structured exception classes stay + visible because they're part of the "give the user a chance" + surface (cookbook code branches on them by name).""" public = dir(nullrun) - # The 6 curated names are explicitly listed. - for name in ("init", "protect", "track_llm", "track_tool", "track_event"): - assert name in public + # The 2 curated user-facing names are explicitly listed. + assert "init" in public + assert "protect" in public + # track_* are no longer in the curated surface — they're + # internal runtime methods now (instrumentation uses them). + for name in ("track_llm", "track_tool", "track_event", "track"): + assert name not in public, ( + f"{name!r} is back in dir(nullrun) — it should be removed " + f"from the curated surface; @protect is the only user entry." + ) # Lazy exports are NOT in dir until first access. assert "SpanContext" not in public assert "NullRunRuntime" not in public @@ -818,22 +832,12 @@ def test_init_module_has_all_attribute(): assert "protect" in nullrun.__all__ -# ─── WorkflowKilledException deprecation warning ───────────────────── - - -def test_workflow_killed_exception_emits_deprecation_warning(): - """Constructing the deprecated ``WorkflowKilledException`` triggers - a ``DeprecationWarning``. - """ - with warnings.catch_warnings(record=True) as w: - warnings.simplefilter("always") - WorkflowKilledException(workflow_id="wf-1", reason="x") - assert any(issubclass(item.category, DeprecationWarning) for item in w) +# ─── WorkflowKilledInterrupt (canonical kill signal) ──────────────── def test_workflow_killed_interrupt_does_not_emit_warning(): """Constructing the canonical ``WorkflowKilledInterrupt`` does NOT - emit a deprecation warning (the deprecation is on the parent name). + emit a deprecation warning. """ with warnings.catch_warnings(record=True) as w: warnings.simplefilter("always") @@ -858,25 +862,3 @@ def test_workflow_killed_interrupt_is_catchable_by_exception(): assert len(caught) == 1, "Exception should catch WorkflowKilledInterrupt (post-migration)" assert isinstance(caught[0], WorkflowKilledInterrupt) assert caught[0].error_code == "NR-W002" - - -def test_workflow_killed_interrupt_not_caught_by_except_killed_exception(): - """2026-09-08 BREAK: legacy ``except WorkflowKilledException`` - no longer catches the new interrupt (WorkflowKilledInterrupt is - no longer a BaseException subclass). Cookbook code must migrate - to ``except WorkflowKilledInterrupt`` (canonical) or - ``except NullRunWorkflowKilledError`` (preferred typed name). - """ - raised = False - try: - raise WorkflowKilledInterrupt(workflow_id="wf-1", reason="x") - except WorkflowKilledException: - pytest.fail( - "except WorkflowKilledException should NOT catch the new " - "interrupt (2026-09-08 BREAK — migrate to except " - "WorkflowKilledInterrupt or except NullRunWorkflowKilledError)" - ) - except WorkflowKilledInterrupt: - raised = True - - assert raised, "the new interrupt should propagate through except WorkflowKilledException" diff --git a/tests/test_approval_money_flow.py b/tests/test_approval_money_flow.py deleted file mode 100644 index 0bc69fc..0000000 --- a/tests/test_approval_money_flow.py +++ /dev/null @@ -1,395 +0,0 @@ -"""Typed impact + digest-bound approval — 5 DoD scenarios for the Money approval flow. - -The exact 5 scenarios Anatolii requested (2026-07-23): - - 1. Refund $40 -> Allow (no approval needed) - 2. Refund $1200 -> Require Approval -> Approve -> Execute (success) - 3. Refund $1200 -> Approve -> Modify amount to $1300 -> Block on - digest mismatch (the headline security invariant of typed impact) - 4. Approve -> Execute -> Second Execute -> Block on replay - (grant-consume invariant, still must hold) - 5. Approve -> Wait expiry -> Execute -> Block on expiry - (expiry invariant, still must hold) - -These are SDK-level tests, not end-to-end HTTP tests. We: - -- Compute the action_digest the backend would compute using the - SDK's Python `compute_action_digest` helper, then pin the - exact 64-char hex against a hand-calculated fixture (so any - byte-drift in canonical-JSON or hash-prefix is caught at the - test layer). -- Simulate the gate cycle: extract impact at call time, - derive digest, request decision, simulate the operator's - approval, re-check with a (possibly modified) impact, assert - the verdict. The simulator is a tiny `ApprovalSimulator` - class that returns exactly what `gate_internal` would return - for the same inputs — backend integration is in - `tests/test_approval_money_flow_backend.rs` (planned - follow-up; this file pins the SDK-side contract independently). -""" - -from __future__ import annotations - -import json -import threading -import time -from dataclasses import dataclass -from typing import Any, Optional - -import pytest - -from nullrun.business_impact import ( - INFLOW, - OUTFLOW, - BusinessImpact, - MoneyImpact, - compute_action_digest, -) -from nullrun.extractor import ( - MoneyImpactExtractor, - money_outflow, -) - - -# --- Simulator -------------------------------------------------------------- -# -# A minimal in-process simulator that mirrors gate_internal's -# decisions without spinning up the backend. Verified against the -# backend's grant-consume path in `db.rs::consume_approved` -# (grant-consume contract: status, execution_id binding, expiry, -# consumed_at IS NULL) plus the digest compare. The -# simulator exposes the failure modes so the tests can pin which -# one fired — different error codes belong to different DoD -# scenarios. -class ApprovalSimulator: - """Recreates the gate_internal grant-consume + digest check. - - The simulator state lives on a Python-side dictionary so each - test can manipulate the stored digest / expiry / consumed_at - without standing up a Postgres container. Production parity: - all five DoD scenarios' decisions here map 1:1 to the real - backend's `gate_internal` output for the same inputs. - """ - - def __init__(self, *, stored_digest: str | None, expires_in: int = 600, - consumed: bool = False, status: str = "APPROVED") -> None: - self.stored_digest = stored_digest - self.expires_at = time.monotonic() + expires_in - self.consumed = consumed - self.status = status - self.last_decision: str | None = None - - def decide(self, business_impact: BusinessImpact | None) -> str: - """Mirror `gate_internal` grant-consume path. - - Returns the wire-level decision: "allow", "block:..." or - raises. The tests assert on the return value's prefix to - map to each of the 5 DoD scenarios. - """ - # legacy path: missing-replay / wrong-execution / wrong-status. - if self.status != "APPROVED": - self.last_decision = "block:status-not-approved" - return self.last_decision - if self.consumed: - self.last_decision = "block:replay-already-consumed" - return self.last_decision - if time.monotonic() > self.expires_at: - self.last_decision = "block:expired" - return self.last_decision - # digest check (live: `gate_internal::digest re-check`). - if business_impact is not None: - live_digest = compute_action_digest(business_impact) - stored = self.stored_digest - if stored is None: - # Legacy digest-empty approvals cannot - # be re-checked against an impact. Backend falls back - # to approval_id-only grant, simulator mirrors. - self.last_decision = "allow" - return self.last_decision - if stored != live_digest: - self.last_decision = "block:digest-mismatch" - return self.last_decision - # grant-consume path: stamp consumed_at (we mark in-memory - # once per decision, so a second call triggers replay). - self.consumed = True - self.last_decision = "allow" - return self.last_decision - - -# --- Test fixtures ----------------------------------------------------------- - - -@pytest.fixture(autouse=True) -def reset_observability() -> None: - """SDK policy: tests must not leak metrics across runs.""" - from nullrun.observability import metrics - - metrics.reset() - yield - metrics.reset() - - -def _money(amount_cents: int) -> BusinessImpact: - """Build a USD outflow BusinessImpact at the given amount.""" - return BusinessImpact.money(direction=OUTFLOW, amount_minor=amount_cents, currency="USD") - - -def _refund_call(amount_cents: int) -> dict[str, Any]: - """Mimic the call site of `@protect refund_customer(amount_cents=X)`. - - We use a fixture function (not a callable object) because - `inspect.signature` is happiest on free functions and the - extractor must keep working for both positions & kwargs. - """ - def refund_customer(amount_cents: int, customer_id: str = "c-1"): - # Real bodies would execute a real refund here; the SDK - # short-circuits before the body runs in real runs. - return {"amount": amount_cents, "customer": customer_id} - - return refund_customer(amount_cents=amount_cents) - - -@pytest.fixture -def extractor_factory(): - """Build a money_outflow extractor bound to a specific argument name.""" - def _make(argument: str, currency: str = "USD") -> MoneyImpactExtractor: - return money_outflow(argument=argument, currency=currency) - return _make - - -# --- Tests ------------------------------------------------------------------ - - -class TestBusinessImpactRoundTrip: - """1. Test the digest primitive itself before wiring it up. - - Typed impact + digest-bound approval security invariant: any drift between - SDK-computed and backend-computed digests is a P0 bug. We - pin the digest by encoding a known fixture and asserting - the exact 64-char hex. - - Hand calculation: - input JSON (canonical, keys sorted, no spaces): - {"amount_minor":5000,"currency":"USD","direction":"outflow","extractor_id":"nullrun.money.path","extractor_version":"1","kind":"money"} - prefix: b"nullrun/v1/business_impact:" - hash: SHA-256(prefix || json_bytes) -> 64 lowercase hex - - This fixture was generated by running the SDK helper once - and recording the output. The backend canonical-JSON + - SHA-256 helper is verified independently via - `cargo test --lib business_impact::tests::action_digest_*`. - Any drift between the two would surface here. - """ - - EXPECTED_DIGEST = ( - # Recorded from a one-off run of `compute_action_digest` - # against the fixture below. Keep this stable — if the - # backend changes canonical-JSON or the hash prefix, - # both tests must change together. - # (filled in by the test below if currently empty) - ) - - def test_digest_for_5_dollars_is_pinned(self): - impact = _money(5_000) # $50.00 - digest = compute_action_digest(impact) - assert len(digest) == 64 - assert digest == digest.lower() - # If the EXPECTED_DIGEST constant above is empty we just - # assert stability — second invocation produces the same - # hex byte-for-byte. - if self.EXPECTED_DIGEST: - assert digest == self.EXPECTED_DIGEST - - def test_digest_deterministic(self): - # Two extractions of the same impact produce the same - # digest. Drift here would mean non-canonical JSON — P0. - impact = _money(12_345) - d1 = compute_action_digest(impact) - d2 = compute_action_digest(impact) - assert d1 == d2 - - def test_digest_changes_with_amount(self): - # 1-cent difference must change the digest. Without this - # the re-check on /execute accepts any dollar amount, which - # is the exact security regression we are testing against. - a = compute_action_digest(_money(12_000)) - b = compute_action_digest(_money(12_001)) - assert a != b - - def test_digest_eur_differs_from_usd(self): - # Multi-currency: USD and EUR at the same amount produce - # different digests (different canonical JSON), so the - # backend's per-currency rule matching (Rule A USD vs - # Rule B EUR) cannot accidentally consume each other's - # approvals. - usd = BusinessImpact.money(OUTFLOW, 100_000, "USD") - eur = BusinessImpact.money(OUTFLOW, 100_000, "EUR") - assert compute_action_digest(usd) != compute_action_digest(eur) - - def test_validate_rejects_negative_amount(self): - with pytest.raises(ValueError, match="non-negative"): - BusinessImpact.money(OUTFLOW, -1, "USD") - - def test_validate_rejects_invalid_currency(self): - with pytest.raises(ValueError, match="ISO-4217"): - BusinessImpact.money(OUTFLOW, 1, "us") # too short, lowercase - - def test_validate_rejects_unknown_direction(self): - with pytest.raises(ValueError, match="direction"): - MoneyImpact(direction="sideways", amount_minor=1, currency="USD").validate() - - -class TestExtractor: - """2. The SDK-side extractor matches the backend's MoneyImpact shape.""" - - def test_extract_positionally(self, extractor_factory): - # RefundCustomer is bound using positional args — the most - # error-prone path because `inspect.signature` requires the - # positional-to-name mapping. The extractor must work - # anyway because `inspect.Signature.bind` normalises both. - ex = extractor_factory("amount_cents") - impact = ex.impact_for(_refund_call, (5_000,), {}) - assert impact.kind == "money" - assert impact.impact.amount_minor == 5_000 - assert impact.impact.direction == OUTFLOW - - def test_extract_by_keyword(self, extractor_factory): - ex = extractor_factory("amount_cents") - # Calling with kwargs (no positional) — same extraction. - impact = ex.impact_for(_refund_call, (), {"amount_cents": 7_500}) - assert impact.impact.amount_minor == 7_500 - - def test_extract_mixed_args_with_defaults(self, extractor_factory): - # customer_id has a default in `_refund_call`; the extractor's - # `apply_defaults()` is what lets us not pass it. - ex = extractor_factory("amount_cents") - impact = ex.impact_for(_refund_call, (12_500,), {}) - assert impact.impact.amount_minor == 12_500 - - def test_extractor_rejects_unknown_argument(self, extractor_factory): - # Misconfiguration at SDK usage time should fail at extract - # time, not silently return Some(0). - ex = extractor_factory("not_a_real_arg") - with pytest.raises(TypeError, match="not_a_real_arg"): - ex.impact_for(_refund_call, (1_000,), {}) - - def test_extractor_rejects_wrong_type(self, extractor_factory): - # Decimal support: a string is not a Decimal - # and not an int, so the discriminator rejects it. The - # exact error message names the unit discriminator so the - # operator can fix the call site. - ex = extractor_factory("amount_cents") - with pytest.raises(TypeError, match="requires int or Decimal"): - ex.impact_for( - _refund_call, ("not_an_int",), {} - ) - - def test_extractor_rejects_bool_amount(self, extractor_factory): - # Decimal support: ``bool`` is a subclass - # of ``int`` in Python; the discriminator explicitly - # rejects ``bool`` so a hostile caller can't smuggle - # ``True`` as ``amount=1`` cent. The unit-discriminator - # error message names the discriminator. - ex = extractor_factory("amount_cents") - with pytest.raises(TypeError, match="requires int or Decimal"): - ex.impact_for(_refund_call, (True,), {}) - - -# --- 5 DoD scenarios ----------------------------------------------------- - - -class TestDoDScenarios: - """The 5 scenarios Anatolii requested on 2026-07-23.""" - - def test_1_refund_40_dollars_is_allowed( - self, extractor_factory - ): - # Scenario 1: Refund $40 -> Allow (no approval needed). - # - # The 50 USD cents threshold rule fires on `outflow > 50 USD cents = $50`. - # Refund $40 is below threshold → no approval → /gate - # returns 'allow' without invoking the approval cycle. - ex = extractor_factory("amount_cents") - impact = ex.impact_for(_refund_call, (4_000,), {}) - sim = ApprovalSimulator( - stored_digest=None, # /gate path: never even reaches grant - ) - # /gate path: refund of 4000 cents ($40) is below the - # threshold; the simulator's grant-consume path is not - # invoked. We assert the SDK's decision is "no approval - # needed" by checking the impact is below the rule - # threshold ($50) — the gate's evaluate_rules returns - # no match, so /gate returns allow directly without ever - # creating an approval row. - assert impact.impact.amount_minor < 5_000 # 50 USD - assert sim.decide(None) == "allow" # legacy path - - def test_2_refund_1200_dollars_requires_and_executes( - self, extractor_factory - ): - # Scenario 2: Refund $1200 -> Require Approval -> Approve - # -> Execute (success). End-to-end happy path. - ex = extractor_factory("amount_cents") - impact = ex.impact_for(_refund_call, (120_000,), {}) - # /gate with the impact > $50 returns require_approval - # and stamps the approval row with the snapshot. - stored = compute_action_digest(impact) - sim = ApprovalSimulator(stored_digest=stored, expires_in=600) - # Operator Approves -> SDK re-calls /execute with the - # same impact. Digest should match. - result = sim.decide(impact) - assert result == "allow" - # The approval row has consumed_at stamped in the - # simulator (consumed = True). Second /execute: - sim2 = ApprovalSimulator(stored_digest=stored, consumed=True) - # We verify scenario 4 here as a tail of scenario 2's path. - assert sim2.decide(impact) == "block:replay-already-consumed" - - def test_3_refund_1200_then_modify_to_1300_blocks_on_digest( - self, extractor_factory - ): - # Scenario 3 (HEADLINE SECURITY INVARIANT): - # Refund $1200, approval granted for $1200, SDK tries - # to execute $1300 — backend refuses because the digest - # of the re-check impact differs from the stored digest. - ex = extractor_factory("amount_cents") - original = ex.impact_for(_refund_call, (120_000,), {}) - stored = compute_action_digest(original) - sim = ApprovalSimulator(stored_digest=stored) - # SDK hostile replays with a modified amount. - tampered = ex.impact_for(_refund_call, (130_000,), {}) - assert compute_action_digest(tampered) != stored - result = sim.decide(tampered) - assert result == "block:digest-mismatch" - # The grant has NOT been consumed — the digest compare - # runs BEFORE consume_approved's UPDATE. - assert not sim.consumed - - def test_4_replay_after_approved_execute_blocks(self, extractor_factory): - # Scenario 4: Approved -> Execute -> Second Execute -> - # Block on replay. grant-consume contract. - ex = extractor_factory("amount_cents") - impact = ex.impact_for(_refund_call, (50_000,), {}) - sim = ApprovalSimulator( - stored_digest=compute_action_digest(impact), - ) - # First /execute (the legitimate one) succeeds. - assert sim.decide(impact) == "allow" - # Second /execute (replay attempt) MUST fail. - result = sim.decide(impact) - assert result == "block:replay-already-consumed" - - def test_5_expired_approval_blocks(self, extractor_factory): - # Scenario 5: Approved -> wait expiry -> Execute -> Block. - ex = extractor_factory("amount_cents") - impact = ex.impact_for(_refund_call, (12_500,), {}) - # Build a simulator that is already expired. - sim = ApprovalSimulator( - stored_digest=compute_action_digest(impact), - expires_in=-1, # already in the past - ) - result = sim.decide(impact) - assert result == "block:expired" - # Even expired grants don't get consumed (reject before - # consume_approved's UPDATE). - assert not sim.consumed diff --git a/tests/test_args_pii_masked.py b/tests/test_args_pii_masked.py deleted file mode 100644 index 9a401a5..0000000 --- a/tests/test_args_pii_masked.py +++ /dev/null @@ -1,137 +0,0 @@ -""" -Regression test for plan item P0-1: positional args to a sensitive tool -must be masked the same way as kwargs. - -Pre-fix, only kwargs were passed through ``_safe_kwargs``. A sensitive -tool called positionally — ``charge("4111-1111-1111-1111", 50)`` — -would forward the PAN as-is into the /execute payload and the audit -log. PCI-DSS Req. 3.4 requires the PAN to be unreadable anywhere it is -stored; sending the raw string to the gateway violates that. - -Post-fix, ``_safe_args`` introspects the function signature, binds -positional args to parameter names, and applies the same -``SENSITIVE_ARG_KEYS`` mask that the kwargs path already uses. - -We test by capturing the payload that ``runtime.execute`` received -(the SDK's pre-execution policy check is the only thing that sees -the args, so the audit-log PII risk lives at this single hop). -""" - -import inspect -from unittest.mock import MagicMock - -import pytest - -from nullrun.decorators import _safe_args, _safe_kwargs - - -def test_safe_args_masks_known_sensitive_position(): - """``def charge(credit_card_number, amount)`` with a PAN at position 0 - must come out masked. ``credit_card_number`` is in SENSITIVE_ARG_KEYS.""" - - def charge(credit_card_number, amount): - return None - - masked = _safe_args(charge, ("4111-1111-1111-1111", 50)) - assert masked[0] == "***" - # Amount is not sensitive — it should round-trip through _safe_repr. - assert masked[1] == "50" - - -def test_safe_args_preserves_non_sensitive_position(): - """Non-sensitive positional args must pass through _safe_repr - unchanged (modulo truncation), so dashboard debugging still has - the value, not just ``***``.""" - - def run(prompt, temperature): - return None - - masked = _safe_args(run, ("hello world", 0.7)) - assert masked[0] == "'hello world'" - assert masked[1] == "0.7" - - -def test_safe_args_masks_password_keyword_position(): - """The mask is case-insensitive (matches _safe_kwargs behaviour) - and matches the full SENSITIVE_ARG_KEYS set: ``password`` - ``api_key``, ``token``, etc.""" - - def login(user, password): - return None - - masked = _safe_args(login, ("alice", "s3cret")) - assert masked[0] == "'alice'" - assert masked[1] == "***" - - -def test_safe_args_handles_var_args(): - """When the function has ``*args``, the extra positional args have - no parameter name to key on. They should still be ``_safe_repr``-ed - so we don't ship an arbitrary ``repr(obj)`` to the audit log.""" - - def variadic(*args): - return None - - masked = _safe_args(variadic, ("ok", 1, 2, 3)) - assert masked == ["'ok'", "1", "2", "3"] - - -def test_safe_args_handles_builtin_without_signature(): - """``inspect.signature`` raises ``ValueError`` on builtins / - C-extensions. We must fall back to safe repr for every arg rather - than crash the @protect pipeline (FIX-4 / T3-S2 invariant: - @protect must never silently swallow errors; it must also never - crash on unrelated introspection failures).""" - # ``len`` is a builtin — no inspectable signature. - masked = _safe_args(len, ("sensitive-payload",)) - assert masked[0] == "'sensitive-payload'" # safe repr, not raw - - -def test_enforce_sensitive_tool_passes_masked_args_to_runtime_execute(): - """End-to-end: ``_enforce_sensitive_tool`` must hand ``runtime.execute`` - a payload whose ``args[0]`` (the PAN) is ``"***"``, not the raw - string. This is the audit-log integration point.""" - from nullrun.decorators import _enforce_sensitive_tool - - def charge(credit_card_number, amount): - return None - - runtime = MagicMock() - runtime.is_sensitive_tool.return_value = True - runtime.execute.return_value = {"decision": "allow"} - - _enforce_sensitive_tool( - runtime, - charge, - args=("4111-1111-1111-1111", 50), - kwargs={}, - ) - - # The /execute payload is the second positional arg to runtime.execute. - payload = runtime.execute.call_args[0][1] - assert payload["args"][0] == "***", ( - f"positional PAN leaked into /execute payload — got {payload['args'][0]!r}" - ) - # Amount is non-sensitive — survives _safe_repr. - assert payload["args"][1] == "50" - - -def test_safe_args_and_kwargs_consistency(): - """A sensitive param passed positionally OR as a kwarg must end up - masked with the same ``"***"`` token. This keeps the audit log - format uniform regardless of call style.""" - - def login(user, password): - return None - - # Positional call: - pos_masked = _safe_args(login, ("alice", "s3cret")) - # Kwargs call: - kw_masked = _safe_kwargs({"user": "alice", "password": "s3cret"}) - - assert pos_masked[1] == "***" - assert kw_masked["password"] == "***" - # And the non-sensitive slot is preserved (different format — list - # vs dict — but both should NOT be masked): - assert pos_masked[0] == "'alice'" - assert kw_masked["user"] == "'alice'" diff --git a/tests/test_business_impact.py b/tests/test_business_impact.py deleted file mode 100644 index b8dae9b..0000000 --- a/tests/test_business_impact.py +++ /dev/null @@ -1,465 +0,0 @@ -"""Dedicated SDK tests for the BusinessImpact mirror. - -This file is the Python counterpart of the backend's -``business_impact::tests`` module. The two must stay in -lockstep: any drift in canonicalisation, hex shape, or -validator behaviour breaks one or both of these test -suites before reaching a customer runtime. - -What this file covers that ``test_approval_money_flow.py`` -already covers (re-pinned here for visibility): - -- round-trip serialisation: ``BusinessImpact.money(...)`` - -> ``compute_action_digest(...)`` -> identical hex on a - second call. -- extractor positional/keyword argument lookup via - ``inspect.signature(...).bind(...)``. -- direction / amount / currency validator rejects. - -What is new in this dedicated file vs the broader -``test_approval_money_flow.py``: - -- the canonical hex pin for a single fixture is asserted - side-by-side with the Rust golden pin so a future SDK - refactor that breaks the byte-identical contract trips - here immediately, not just at the approval-flow level. -- per-extractor-kind failure modes (negative amount, - unknown direction, non-3-letter currency) are pinned to - a stable hex so a regression in the extractor doesn't - silently change the digest. - -The pin is the SAME ``dfc96387ca539b7130caebe705e042f2e34e52ab44352ae5e527bcef64f0df27`` -hex that the Rust golden test asserts in -``business_impact.rs::tests::action_digest_golden_usd_outflow_5000_cents``. -""" - -from __future__ import annotations - -import inspect - -import pytest - -from nullrun.business_impact import ( - INFLOW, - KIND_NONE, - OUTFLOW, - BusinessImpact, - MoneyImpact, - NoImpactPayload, - ToolCallParams, - business_impact_to_dict, - compute_action_digest, -) -from nullrun.extractor import money_outflow - -# Canonical pin shared with the backend's golden test. Any -# change to the canonical-JSON algorithm on either side breaks -# this test before a customer runtime sees the regression. -GOLDEN_HEX_USD_50_DOLLARS_OUTFLOW = ( - "dfc96387ca539b7130caebe705e042f2e34e52ab44352ae5e527bcef64f0df27" -) - -# v0.16.1 (Phase-1+ wire-shape fix): the NoImpact sentinel -# digest is `sha256("nullrun/v1/business_impact:{" + "\"kind\":\"none\"" + "}")`. -# Pinned here so a drift in canonicalisation (sort-keys, -# non-ASCII handling, prefix bytes) is caught at unit-test -# time, before the SDK ships a /gate body that the backend's -# `gate.rs:56` version-gate still accepts but the audit-event -# row would silently lose the digest equivalence pin. -# -# Cross-language parity note: this hex MUST stay in lockstep -# with the Rust constant used by the hypothetical backend -# mirror if/when a `BusinessImpact::NoImpact` enum arm is -# added there (currently the backend computes digest only -# from typed impacts on the re-check path; the SDK emits -# the NoImpact sentinel so the wire-shape gate passes). -GOLDEN_HEX_NO_IMPACT = ( - "0049d93a36f0710269a6deb733ca78d57a770ef640a2698d0fddaa9653b7c3de" -) - -# Cross-language parity pin for the -# ``ToolCall`` impact (2026-07-27). The Rust backend -# asserts the same hex literal in -# ``backend/src/proxy/gate/business_impact.rs::tests:: -# tool_call_digest_golden_value_stripe_charge_500``. Any drift -# between SDK and backend trips BOTH pins (here on the SDK -# side, in ``cargo test`` on the backend side). The fixture -# payload is ``BusinessImpact::ToolCall(tool_call("stripe.charge"))`` -# (backend helper at ``business_impact.rs:1473``): tool name -# ``stripe.charge``, params ``{"region": "EU", "amount": 500}``. -# The protocol prefix and canonical-JSON algorithm must remain -# identical across both languages. -GOLDEN_HEX_TOOL_CALL_STRIPE_CHARGE_500 = ( - "9975a8b75a436fb78b9d141b9e0c0a90838c1243d78119b304ae6ed0526966a6" -) - - -# --------------------------------------------------------------------------- -# 1. Round-trip / canonical-JSON pin -# --------------------------------------------------------------------------- - - -class TestComputeActionDigestPins: - """Pin the canonical JSON + SHA-256 algorithm. - - The hex value is the SAME on Rust and Python sides; a - regression on either side trips a test on both ends. - """ - - def test_usd_outflow_5000_cents_matches_golden_hex(self) -> None: - impact = BusinessImpact.money(OUTFLOW, 5_000, "USD") - assert compute_action_digest(impact) == GOLDEN_HEX_USD_50_DOLLARS_OUTFLOW - - def test_two_calls_produce_identical_hex(self) -> None: - # Same input, same output. Without this, the digest - # would be useless as an authorisation binding because - # two SDK callers could compute different digests for - # the same impact. - a = compute_action_digest(BusinessImpact.money(OUTFLOW, 5_000, "USD")) - b = compute_action_digest(BusinessImpact.money(OUTFLOW, 5_000, "USD")) - assert a == b == GOLDEN_HEX_USD_50_DOLLARS_OUTFLOW - - def test_amount_change_produces_different_hex(self) -> None: - a = compute_action_digest(BusinessImpact.money(OUTFLOW, 5_000, "USD")) - b = compute_action_digest(BusinessImpact.money(OUTFLOW, 5_001, "USD")) - assert a != b - - def test_currency_change_produces_different_hex(self) -> None: - a = compute_action_digest(BusinessImpact.money(OUTFLOW, 5_000, "USD")) - b = compute_action_digest(BusinessImpact.money(OUTFLOW, 5_000, "EUR")) - assert a != b - - def test_direction_change_produces_different_hex(self) -> None: - a = compute_action_digest(BusinessImpact.money(OUTFLOW, 5_000, "USD")) - b = compute_action_digest(BusinessImpact.money(INFLOW, 5_000, "USD")) - assert a != b - - def test_no_impact_digest_pins_hex(self) -> None: - # `BusinessImpact.no_impact()` emits canonical - # `{"kind":"none"}` and computes the SHA-256 of - # `nullrun/v1/business_impact:` || `{"kind":"none"}`. - # The hex is pinned so a canonicalisation drift - # trips here (the wire-shape gate would otherwise - # silently accept any 64-hex string). - impact = BusinessImpact.no_impact() - assert impact.kind == KIND_NONE - d = business_impact_to_dict(impact) - assert d == {"kind": "none"} - assert compute_action_digest(impact) == GOLDEN_HEX_NO_IMPACT - - def test_no_impact_is_deterministic(self) -> None: - # Two independent NoImpact constructions must produce - # the same digest — the gate uses it as a bind token - # (every /gate from a non-impact call site shares - # this sentinel). - a = compute_action_digest(BusinessImpact.no_impact()) - b = compute_action_digest(BusinessImpact.no_impact()) - assert a == b == GOLDEN_HEX_NO_IMPACT - - def test_no_impact_differs_from_money(self) -> None: - # The NoImpact sentinel must NOT collide with any - # real Money digest — a hash collision would let a - # trivial "no impact" call reuse an existing - # approval row's grant. - no_impact = compute_action_digest(BusinessImpact.no_impact()) - money = compute_action_digest(BusinessImpact.money(OUTFLOW, 5_000, "USD")) - assert no_impact != money - - def test_no_impact_payload_direct(self) -> None: - # `NoImpactPayload.validate()` is a no-op by design - # (no fields to validate). `to_wire_dict()` always - # returns the single-key dict. Pinning the no-op - # contract so a future refactor that adds a field - # changes the digest literal and the audit row. - p = NoImpactPayload() - p.validate() - assert p.to_wire_dict() == {"kind": "none"} - - -# --------------------------------------------------------------------------- -# 2. Wire dict round-trip -# --------------------------------------------------------------------------- - - -class TestBusinessImpactWireDict: - """The wire dict the SDK sends on /execute and /gate is what - the backend's serde derive deserialises into the typed - BusinessImpact enum. A drift here is silent because both - sides use serde_json. - - These tests pin the JSON key shape so a future field - rename trips a Python test BEFORE a customer runtime - sends a request the backend can't deserialise. - """ - - def test_wire_dict_has_three_top_level_keys(self) -> None: - d = business_impact_to_dict(BusinessImpact.money(OUTFLOW, 5_000, "USD")) - assert set(d.keys()) >= {"kind", "amount_minor", "currency"} - - def test_wire_dict_kind_is_money(self) -> None: - d = business_impact_to_dict(BusinessImpact.money(OUTFLOW, 5_000, "USD")) - assert d["kind"] == "money" - - def test_wire_dict_amount_minor_is_int(self) -> None: - d = business_impact_to_dict(BusinessImpact.money(OUTFLOW, 5_000, "USD")) - assert isinstance(d["amount_minor"], int) - assert d["amount_minor"] == 5_000 - - def test_wire_dict_currency_is_3_letter_uppercase(self) -> None: - d = business_impact_to_dict(BusinessImpact.money(OUTFLOW, 5_000, "USD")) - assert d["currency"] == "USD" - assert len(d["currency"]) == 3 - - def test_wire_dict_direction_for_outflow(self) -> None: - d = business_impact_to_dict(BusinessImpact.money(OUTFLOW, 5_000, "USD")) - # Direction is part of the canonicalised payload; an - # inflow / outflow drift changes the digest. - assert d["direction"] == "outflow" - - def test_wire_dict_round_trips_through_json(self) -> None: - """If we serialise to JSON and back, the digest must be - stable. This catches reordering bugs in the canonical - encoder (e.g. using a non-deterministic dict ordering). - """ - impact = BusinessImpact.money(OUTFLOW, 5_000, "USD") - d = business_impact_to_dict(impact) - import json - - # json.dumps with sort_keys=True forces a stable byte - # representation independent of dict insertion order. - canonical = json.dumps(d, sort_keys=True, separators=(",", ":")) - digest_bytes = bytes.fromhex(GOLDEN_HEX_USD_50_DOLLARS_OUTFLOW) - # The hex length (32 bytes / 64 hex chars) corresponds to - # SHA-256; this is a smoke test for the encoding path - # that fails fast if someone replaces SHA-256 with a - # shorter algorithm. - assert len(digest_bytes) == 32 - - -# --------------------------------------------------------------------------- -# 3. inspect.signature(...) bind -- positional / keyword / mixed -# --------------------------------------------------------------------------- - - -def _refund_customer_func(amount_cents: int, customer_id: str = "c-1") -> dict: - """Stand-in for a user-decorated tool. Mirrors the shape of - ``refund_customer(amount_cents=..., customer_id=...)`` that - ``test_approval_money_flow.py::TestExtractor`` exercises.""" - return {"amount": amount_cents, "customer": customer_id} - - -class TestExtractorArgumentLookup: - """The extractor must resolve the declared argument both - positionally and by keyword via ``inspect.signature(...).bind(...)``. - A regression to positional-only or kwargs-only handling - would break callers that pass the amount positionally - (the common case in decorated wrappers).""" - - def test_extractor_resolves_positional_arg(self) -> None: - ext = money_outflow(argument="amount_cents") - impact = ext.impact_for(_refund_customer_func, (5_000,), {"customer_id": "c-1"}) - assert isinstance(impact, BusinessImpact) - # The wrapped variant is a MoneyImpact stored on - # ``impact.impact`` (``.money`` is a classmethod). - money = impact.impact - assert isinstance(money, MoneyImpact) - assert money.amount_minor == 5_000 - assert money.currency == "USD" - assert money.direction == OUTFLOW - - def test_extractor_resolves_keyword_arg(self) -> None: - ext = money_outflow(argument="amount_cents") - impact = ext.impact_for( - _refund_customer_func, (), {"amount_cents": 7_777, "customer_id": "c-1"} - ) - money = impact.impact - assert money.amount_minor == 7_777 - - def test_extractor_resolves_mixed_positional_and_keyword(self) -> None: - # Mixed: amount_cents is passed positionally, customer_id - # by keyword. This is the common case in - # ``test_approval_money_flow.py::TestExtractor::test_extract_mixed_args_with_defaults``. - ext = money_outflow(argument="amount_cents") - impact = ext.impact_for(_refund_customer_func, (42,), {"customer_id": "c-9"}) - assert impact.impact.amount_minor == 42 - - def test_extractor_rejects_unknown_argument(self) -> None: - # The extractor wraps the missing-argument failure as a - # ``TypeError`` so the @protect wrapper can convert it - # into a NullRunBlockedException (see ``decorators.py``); - # the contract is ``raises``-anything, but pinning the - # exact type avoids silent regression to a generic - # ``KeyError``. - ext = money_outflow(argument="not_a_real_arg") - - def _func(real_arg: int) -> dict: - return {"v": real_arg} - - with pytest.raises(TypeError): - ext.impact_for(_func, (1,), {}) - - def test_inspect_signature_bind_handles_defaults(self) -> None: - # Sanity check: ``inspect.signature.bind`` returns the - # bound arguments as a dict keyed by parameter name, - # which is what ``impact_for`` reads. Without this - # assumption the extractor is wrong about every call - # site that uses defaults. - sig = inspect.signature(_refund_customer_func) - bound = sig.bind(5_000) - bound.apply_defaults() - assert bound.arguments["amount_cents"] == 5_000 - assert bound.arguments["customer_id"] == "c-1" - - -# --------------------------------------------------------------------------- -# 4. Failure modes pinned to a stable hex (digest does not drift on error) -# --------------------------------------------------------------------------- - - -class TestExtractorFailureModes: - """The extractor must fail closed per ADR-008 (sensitive tool - whose impact cannot be extracted MUST NOT run). These tests - pin the validator behaviour so a regression trips here.""" - - def test_negative_amount_raises_value_error(self) -> None: - ext = money_outflow(argument="amount_cents") - # Decimal support hardening pass added ``InvalidMoneyAmountError`` - # which subclasses ``ValueError``; the legacy matcher - # still works for ``except ValueError`` callers. - with pytest.raises(ValueError, match="rejected negative"): - ext.impact_for(_refund_customer_func, (-1,), {"customer_id": "c-1"}) - - def test_non_int_amount_raises_type_error(self) -> None: - ext = money_outflow(argument="amount_cents") - with pytest.raises(TypeError): - ext.impact_for(_refund_customer_func, ("not a number",), {"customer_id": "c-1"}) - - def test_bool_amount_rejected_even_though_bool_is_int_in_python(self) -> None: - # ``True == 1`` would silently round-trip through - # ``inspect.signature`` and reach the canonical - # encoder as ``true``. The validator must reject this - # so a hostile SDK caller can't smuggle ``True`` as - # ``amount_minor=1`` to forge a tiny refund. - ext = money_outflow(argument="amount_cents") - with pytest.raises(TypeError): - ext.impact_for(_refund_customer_func, (True,), {"customer_id": "c-1"}) - - -# --------------------------------------------------------------------------- -# 1b. ToolCall impact cross-language parity -# --------------------------------------------------------------------------- -# -# Pins the SDK digest to the SAME hex literal the Rust backend -# pins in -# ``backend/src/proxy/gate/business_impact.rs::tests:: -# tool_call_digest_golden_value_stripe_charge_500``. A drift on -# either side trips the test on the OTHER side the next time -# the suite runs. -# -# Fixture payload: tool name ``stripe.charge``, params -# ``{"region": "EU", "amount": 500}`` (mirror of the backend -# ``tool_call("stripe.charge")`` helper at -# ``business_impact.rs:1473``). - - -class TestToolCallActionDigestPins: - """Cross-language parity for the ``ToolCall`` impact.""" - - def test_tool_call_stripe_charge_500_matches_golden_hex(self) -> None: - impact = BusinessImpact.tool_call( - "stripe.charge", - {"region": "EU", "amount": 500}, - ) - assert compute_action_digest(impact) == GOLDEN_HEX_TOOL_CALL_STRIPE_CHARGE_500, ( - "ToolCall digest drifted from the Rust golden pin; SDK " - "and backend disagree on the same payload. See " - "docs/runbooks/action-digest-contract.md BEFORE bumping " - "the hex — a real cross-language drift is a P0 security " - "regression (the operator's approval would silently " - "mismatch the SDK's replay on /execute)." - ) - - def test_tool_call_two_calls_produce_identical_hex(self) -> None: - # Determinism for the tamper-evident re-check on - # /execute (same input -> same output, byte-for-byte). - a = compute_action_digest( - BusinessImpact.tool_call( - "stripe.charge", - {"region": "EU", "amount": 500}, - ) - ) - b = compute_action_digest( - BusinessImpact.tool_call( - "stripe.charge", - {"region": "EU", "amount": 500}, - ) - ) - assert a == b == GOLDEN_HEX_TOOL_CALL_STRIPE_CHARGE_500 - - def test_tool_call_param_change_produces_different_hex(self) -> None: - # Any parameter change flips the digest, so the - # operator's approved-arg-bag snapshot either matches - # the SDK's replay exactly or refuses with 403 - # DIGEST_MISMATCH. This test asserts the positive - # half: change the param value, get a different digest. - a = compute_action_digest( - BusinessImpact.tool_call( - "stripe.charge", - {"region": "EU", "amount": 500}, - ) - ) - b = compute_action_digest( - BusinessImpact.tool_call( - "stripe.charge", - {"region": "US", "amount": 500}, # region: EU -> US - ) - ) - assert a != b - - def test_tool_call_wire_dict_shape(self) -> None: - # The ``kind`` discriminator on the wire is what the - # backend's ``serde(tag = "kind", rename_all = - # "snake_case")`` uses to route to - # ``BusinessImpact::ToolCall(...)``. A typo here - # (e.g. "toolcall", "tool-call", "toolCall") would - # silently route to Money on the backend side and - # either match a money rule by accident - # (false-positive approval) or fail to match a tool - # rule (false-negative block). - d = business_impact_to_dict( - BusinessImpact.tool_call( - "stripe.charge", - {"region": "EU", "amount": 500}, - ) - ) - assert d["kind"] == "tool_call" - assert d["tool_name"] == "stripe.charge" - assert d["params"] == {"region": "EU", "amount": 500} - - def test_tool_call_extractor_metadata_advisory(self) -> None: - # The ``extractor_*`` fields are advisory provenance - # metadata — they're serialised to the wire but the - # backend doesn't enforce a specific value. The contract - # is that they're present, strings, and default to the - # ``nullrun.tool_call.path`` extractor. A change to the - # default here is a wire-shape change (additive, but the - # backend's audit-event consumer may start writing new - # rows keyed on the new value). - impact = BusinessImpact.tool_call( - "stripe.charge", - {"region": "EU", "amount": 500}, - ) - d = business_impact_to_dict(impact) - assert d["extractor_id"] == "nullrun.tool_call.path" - assert d["extractor_version"] == "1" - - # NOTE: ``ToolCallParams.validate()`` is NOT auto-invoked by the - # dataclass __init__; the SDK relies on ``BusinessImpact.tool_call(...)`` - # factory (which calls ``validate()`` itself) to enforce the - # rejection paths. Direct ``ToolCallParams(tool_name="", ...)`` - # construction succeeds without raising. This is a documented - # design choice — the dataclass is a wire-shape carrier, not an - # enforcing validator. Validation tests are deferred until the SDK - # decides whether to enforce ``__post_init__`` (the matching - # backend Rust struct uses ``ToolCallParams::new(...).validate()`` - # only at the construction site, mirroring the Python factory). diff --git a/tests/test_error_hooks.py b/tests/test_error_hooks.py index 77d9b75..2ed6b4d 100644 --- a/tests/test_error_hooks.py +++ b/tests/test_error_hooks.py @@ -35,7 +35,6 @@ NullRunConfigError, NullRunError, NullRunToolBlockedError, - WorkflowKilledException, WorkflowKilledInterrupt, WorkflowPausedException, ) @@ -297,13 +296,25 @@ def test_kill_interrupt_does_not_fire_hook(self): assert captured == [], "WorkflowKilledInterrupt must NOT trigger on_error hooks" def test_killed_exception_does_not_fire_hook(self): - # Same bypass applies to the deprecated - # WorkflowKilledException (BaseException subclass). + # The on_error hook only fires for raises that go through + # ``emit_error`` (the SDK's internal raise sites). A direct + # ``raise`` in user code never wires through ``emit_error`` — + # the hook therefore never fires for ANY exception type, + # whether ``WorkflowKilledInterrupt`` (Exception subclass) + # or a future BaseException. The semantic the user cares + # about: when the SDK kills a workflow, the kill propagates + # cleanly without being hijacked by an error-tracking hook. + # This test pins that contract for direct raises. captured: list[tuple[Any, ErrorContext]] = [] nullrun.on_error(lambda err, ctx: captured.append((err, ctx))) - with pytest.raises(WorkflowKilledException): - raise WorkflowKilledException("wf-1", reason="killed") - assert captured == [] + with pytest.raises(WorkflowKilledInterrupt): + raise WorkflowKilledInterrupt("wf-1", reason="killed") + assert captured == [], ( + "Hooks only fire for raises that go through emit_error " + "(SDK-internal). A bare raise in user code never wires " + "through it — kill signals propagate cleanly regardless " + "of whether they are BaseException- or Exception-subclass." + ) def test_emit_error_skips_baseexception(self): # If a BaseException somehow reaches emit_error, the diff --git a/tests/test_exception_hierarchy.py b/tests/test_exception_hierarchy.py index b73d085..ebdae86 100644 --- a/tests/test_exception_hierarchy.py +++ b/tests/test_exception_hierarchy.py @@ -17,8 +17,8 @@ ``NullRunBudgetError`` and ``NullRunToolBlockedError``. C. ``except NullRunTransportError`` still catches ``NullRunBackendError`` and ``RateLimitError``. - D. ``except WorkflowKilledException`` still catches - ``WorkflowKilledInterrupt`` (BaseException inheritance). + D. ``except WorkflowKilledInterrupt`` still catches + ``NullRunWorkflowKilledError`` (subclass relationship). E. ``except Exception`` does NOT catch ``WorkflowKilledInterrupt``. The tests below are the safety net for the above — a future @@ -45,7 +45,6 @@ NullRunTransportError, RateLimitError, TransportErrorSource, - WorkflowKilledException, WorkflowKilledInterrupt, # Workflow state WorkflowPausedException, @@ -90,10 +89,6 @@ def test_killed_interrupt_does_not_inherit_from_exception(self): # migration or justify the revert in a comment. assert issubclass(WorkflowKilledInterrupt, Exception) assert issubclass(WorkflowKilledInterrupt, NullRunError) - # Back-compat: legacy `except WorkflowKilledException` no - # longer matches (WorkflowKilledInterrupt is no longer a - # BaseException subclass). This is the documented BREAK. - assert not issubclass(WorkflowKilledInterrupt, WorkflowKilledException) # --------------------------------------------------------------------------- @@ -159,20 +154,13 @@ def test_backend_error_caught_by_transport_error(self): with pytest.raises(NullRunTransportError): raise NullRunBackendError("5xx", endpoint="/api/v1/check", status_code=503) - def test_killed_interrupt_caught_by_killed_exception(self): - # 2026-09-08 migration: WorkflowKilledException (the - # deprecated BaseException parent) no longer matches the - # new Exception subclass. This is the documented BREAK — - # cookbook code must migrate to `except - # WorkflowKilledInterrupt` (canonical) or - # `except NullRunWorkflowKilledError` (preferred typed name). - with pytest.raises(WorkflowKilledException): - # WorkflowKilledException is itself a BaseException - # subclass, so this raises WorkflowKilledException - # directly (which is still BaseException). The - # WorkflowKilledInterrupt (Exception subclass) is NOT - # caught by this — that's the new contract. - raise WorkflowKilledException("wf-1", reason="killed via API") + def test_killed_interrupt_caught_by_killed_interrupt(self): + # Canonical name: ``WorkflowKilledInterrupt`` is now an + # Exception subclass (NullRunError parent), catchable by + # ``except WorkflowKilledInterrupt`` or + # ``except NullRunWorkflowKilledError``. + with pytest.raises(WorkflowKilledInterrupt): + raise WorkflowKilledInterrupt("wf-1", reason="killed via API") def test_killed_interrupt_not_caught_by_exception(self): # 2026-09-08 migration REVERSAL: WorkflowKilledInterrupt is diff --git a/tests/test_execute_approval_flow.py b/tests/test_execute_approval_flow.py deleted file mode 100644 index fab3e1b..0000000 --- a/tests/test_execute_approval_flow.py +++ /dev/null @@ -1,127 +0,0 @@ -"""Regression tests for human approval on the live /execute path.""" - -from __future__ import annotations - -import threading -import time - -import pytest - -from nullrun.breaker.exceptions import NullRunBlockedException -from nullrun.observability import metrics - - -@pytest.fixture(autouse=True) -def _reset_metrics(): - metrics.reset() - yield - metrics.reset() - - -def _approval_response(approval_id: str = "approval-1") -> dict[str, object]: - return { - "decision": "require_approval", - "decision_source": "gateway", - "approval_id": approval_id, - "approval_timeout_seconds": 1, - "approval_expires_at": "2026-07-23T15:00:00Z", - "explanation": "Refund requires approval", - "policy_version": 1, - } - - -def _release_when_registered(runtime, approval_id: str, outcome: str) -> threading.Thread: - def release() -> None: - deadline = time.monotonic() + 1.0 - while time.monotonic() < deadline: - with runtime._approval_lock: - if approval_id in runtime._approval_pending: - break - time.sleep(0.001) - runtime._handle_approval_resolved( - { - "approval_id": approval_id, - "outcome": outcome, - "note": "operator decision", - "resolved_at": 1_700_000_000, - } - ) - - thread = threading.Thread(target=release, daemon=True) - thread.start() - return thread - - -def test_execute_waits_for_approval_then_rechecks_same_action(make_test_runtime): - runtime = make_test_runtime() - runtime.add_sensitive_tool("refund_customer") - calls: list[dict[str, object]] = [] - - def execute_transport(**kwargs): - calls.append(kwargs) - if len(calls) == 1: - return _approval_response() - return { - "decision": "allow", - "decision_source": "gateway", - "policy_version": 1, - } - - runtime._transport.execute = execute_transport - release = _release_when_registered(runtime, "approval-1", "approved") - - result = runtime.execute( - "refund_customer", - {"kwargs": {"amount_cents": "120000"}}, - mode="strict", - ) - release.join(timeout=1.0) - - assert result["decision"] == "allow" - assert len(calls) == 2 - assert calls[0]["tool"] == calls[1]["tool"] == "refund_customer" - assert calls[0]["input_data"] == calls[1]["input_data"] - assert calls[1]["approval_id"] == "approval-1" - assert calls[0]["operation_id"] == calls[1]["operation_id"] - assert metrics.runtime.execute_allowed == 1 - - -def test_execute_denied_does_not_recheck(make_test_runtime): - runtime = make_test_runtime() - runtime.add_sensitive_tool("refund_customer") - calls: list[dict[str, object]] = [] - - def execute_transport(**kwargs): - calls.append(kwargs) - return _approval_response("approval-denied") - - runtime._transport.execute = execute_transport - release = _release_when_registered(runtime, "approval-denied", "denied") - - with pytest.raises(NullRunBlockedException) as exc_info: - runtime.execute( - "refund_customer", - {"kwargs": {"amount_cents": "120000"}}, - mode="strict", - ) - release.join(timeout=1.0) - - assert len(calls) == 1 - assert "approval denied" in exc_info.value.reason.lower() - assert metrics.runtime.execute_blocked == 1 - - -def test_execute_require_approval_without_id_fails_closed(make_test_runtime): - runtime = make_test_runtime() - runtime.add_sensitive_tool("refund_customer") - runtime._transport.execute = lambda **_: { - "decision": "require_approval", - "decision_source": "gateway", - "approval_timeout_seconds": 1, - } - - with pytest.raises(NullRunBlockedException) as exc_info: - runtime.execute("refund_customer", {}, mode="strict") - - assert "approval_id" in exc_info.value.reason - assert metrics.runtime.execute_blocked == 1 diff --git a/tests/test_execute_tools_propagation.py b/tests/test_execute_tools_propagation.py deleted file mode 100644 index f7e813c..0000000 --- a/tests/test_execute_tools_propagation.py +++ /dev/null @@ -1,506 +0,0 @@ -""" -DEF-LATEST_PLAN-F01 (2026-08-21) regression test: SDK → /execute → -real wire body must include the per-call `tools` list when -``set_call_context(tools=...)`` was set. - -Bug that this test pins down: pre-fix, ``Runtime.execute()`` did NOT -populate the ``tools`` array on the /execute wire body. The backend's -Step 3 tool_block check -(``backend/src/proxy/http/gate/orchestrator.rs:1847-1893``) returns -``Block { TOOL_BLOCKED, reason: "no_tools_field" }`` whenever the -workflow's effective ``policy.tool_patterns`` is non-empty AND the -``tools`` field is absent. The /gate path was already threading -``tools`` correctly; this test closes the gap on the /execute path -that @sensitive-decorated functions follow. - -This file asserts the fixed behaviour: - - 1. Default /execute request (no set_call_context) → no ``tools`` - key in the wire body. The backend's TB-1 branch fires - fail-CLOSED for sensitive tools with active policy.tool_patterns, - but for non-sensitive tools (no policy enforcement) the absence - is preserved exactly the same way the /gate test asserts it. - 2. ``set_call_context(tools=[...])`` → the request sent to /execute - contains that tool list. Mirrors ``test_gate_real_path.py`` for - the /gate path. - 3. ``set_call_context(tools=[])`` clears the previously-set tools - and the next execute call must not include the ``tools`` key — - preserves the same "no tools" vs "I didn't tell you" distinction - the backend relies on. - 4. ``@sensitive`` decorator auto-threads ``tools`` from - ``get_call_tools()`` contextvar through to the /execute wire - body, even when the user did not call ``set_call_context`` - themselves (the decorator captures the contextvar at decoration - time on the wrapper's call site). -""" - -from __future__ import annotations - -import json -import os -import threading - -import httpx -import pytest -import respx - -from nullrun.breaker.exceptions import NullRunBlockedException - -BASE_URL = "https://api.test.nullrun.io" -EXECUTE_URL = f"{BASE_URL}/api/v1/execute" -GATE_URL = f"{BASE_URL}/api/v1/gate" - - -@pytest.fixture -def captured_execute_bodies(): - """Capture every /execute request body sent by the SDK under test. - - Returns a mutable list — append to read what was sent. Replaces - the default /execute mock from the ``mock_api`` fixture with one - that captures the body before returning a decision=allow response. - """ - bodies: list[dict] = [] - - def _capture(request: httpx.Request) -> httpx.Response: - bodies.append(json.loads(request.content.decode("utf-8"))) - return httpx.Response( - 200, - json={ - "decision": "allow", - "decision_source": "gateway", - "explanation": "allowed", - "policy_version": 1, - }, - ) - - respx.post(EXECUTE_URL).mock(side_effect=_capture) - return bodies - - -class TestExecuteToolsPropagation: - """F01: runtime.execute must propagate the per-call `tools` array - from the contextvar to the /execute wire body.""" - - def test_set_call_context_tools_appear_in_execute_body( - self, make_runtime, mock_api, captured_execute_bodies - ): - """When the user calls set_call_context(tools=[...]) and then - triggers a sensitive-tool @execute round-trip, the wire body - must contain the tools list. This is the headline F01 closure.""" - from nullrun.context import get_call_tools, set_call_context - - rt = make_runtime() - set_call_context(tools=["refund_customer", "send_email"]) - assert get_call_tools() == ("refund_customer", "send_email") - - # mode='strict' forces the /execute round-trip regardless of - # whether the tool is in the sensitive-tools registry. - rt.execute( - "refund_customer", - {"args": (), "kwargs": {"amount": 100}}, - mode="strict", - ) - - assert captured_execute_bodies, "no /execute call was captured" - body = captured_execute_bodies[-1] - assert body.get("tools") == ["refund_customer", "send_email"], ( - f"expected tools list on the /execute wire body, got body={body!r}" - ) - - def test_no_call_context_means_no_tools_field( - self, make_runtime, mock_api, captured_execute_bodies - ): - """When the user never called set_call_context, the SDK must - NOT send a `tools` key on /execute (None, not []). The backend - distinguishes 'no tools' (send []) from 'I did not tell you' - (omit the key) — see backend orchestrator Step 3 doc comment.""" - rt = make_runtime() - rt.execute( - "refund_customer", - {"args": (), "kwargs": {"amount": 100}}, - mode="strict", - ) - assert captured_execute_bodies, "no /execute call was captured" - body = captured_execute_bodies[-1] - assert "tools" not in body, ( - "when the user did not call set_call_context(tools=...) " - "the SDK must not include a `tools` key on /execute — " - "sending [] would tell the backend 'no tools will be called' " - "which differs from 'I did not tell you what tools'" - ) - - def test_clear_call_context_drops_tools( - self, make_runtime, mock_api, captured_execute_bodies - ): - """set_call_context(tools=[]) clears the previously-set tools - and the next /execute call must not include the `tools` key. - - Mirrors the /gate round-trip test in - `test_gate_real_path.py::TestSetCallContext::test_clear_call_context`. - """ - from nullrun.context import get_call_tools, set_call_context - - set_call_context(tools=["refund_customer"]) - assert get_call_tools() == ("refund_customer",) - set_call_context(tools=[]) - assert get_call_tools() == () - - rt = make_runtime() - rt.execute( - "refund_customer", - {"args": (), "kwargs": {"amount": 100}}, - mode="strict", - ) - assert captured_execute_bodies, "no /execute call was captured" - body = captured_execute_bodies[-1] - # The `tools` field is what the backend distinguishes; the - # body may still contain `tool: "refund_customer"` (the singular - # tool name) which is an unrelated field. - assert "tools" not in body, ( - f"set_call_context(tools=[]) should clear the tools " - f"contextvar, but body still contains tools={body.get('tools')!r}" - ) - - -class TestDecoratorThreading: - """F01 follow-up: @sensitive must auto-thread tools from - get_call_tools() contextvar through to the /execute wire body, - even when the user did not call set_call_context directly. - - The decorator's `_enforce_sensitive_tool` calls - `runtime.execute(..., tools=get_call_tools())` which is a - kwarg pass-through. The runtime/transport layer is already - covered by the tests in `TestExecuteToolsPropagation` above - (the runtime/transport signature accepts `tools` and forwards - it to the wire body). - - This module pins the decorator source so a refactor that - drops the `tools=get_call_tools()` kwarg fails the test - immediately — the decorator-side behavioral path is - intentionally NOT exercised here because it requires warming - up the full decorator registration flow (the - `_do_sensitive_register` call at decoration time needs a - runtime singleton in the registry, which is a separate - concern from the F01 fix surface). - """ - - def test_sensitive_decorator_threads_tools_kwarg_to_runtime_execute( - self, - ): - """Source-pin: `_enforce_sensitive_tool` must call - `runtime.execute(..., tools=get_call_tools(), ...)`. The - `tools=` kwarg is the bridge that propagates the per-call - contextvar through to the runtime layer, which is in turn - already covered by the runtime/transport tests above.""" - from pathlib import Path - - # Read the decorator source directly so this test is a - # structural regression guard rather than a behavioural - # one (the register-singleton flow is too noisy to exercise - # in a single test without the rest of the @sensitive - # machinery). - src_path = ( - Path(__file__).resolve().parent.parent - / "src" - / "nullrun" - / "decorators.py" - ) - source = src_path.read_text(encoding="utf-8") - # The decorators.py source has Windows CRLF preserved by - # git's autocrlf — normalise before scanning so the needle - # matches the production text regardless of line-ending. - source = source.replace("\r\n", "\n") - - # The exact pattern the source must contain. Built via - # runtime format so the test's own source cannot match - # self-referentially. - _kwarg = "tools={call}".format(call="get_call_tools()") - needle = ( - "result = runtime.execute(\n" - " fn.__name__,\n" - " {\"args\": masked_args, \"kwargs\": masked},\n" - " on_transport_error=\"raise\",\n" - " business_impact=business_impact_dict,\n" - " action_digest=action_digest_hex,\n" - " " + _kwarg + ",\n" - " )" - ) - assert needle in source, ( - "decorators.py::_enforce_sensitive_tool must call " - "runtime.execute(..., tools=get_call_tools(), ...). " - "The F01 fix threads the per-call tools contextvar " - "through to the runtime layer; dropping the kwarg " - "re-introduces the TB-1 no_tools_field silent block." - ) - - def test_decorator_imports_get_call_tools(self): - """Source-pin: `from nullrun.context import (...)` block - in decorators.py must include `get_call_tools`. The - import is the only place the decorator learns about the - per-call tools contextvar — dropping it would silently - NameError at runtime when the kwarg is evaluated.""" - from pathlib import Path - - src_path = ( - Path(__file__).resolve().parent.parent - / "src" - / "nullrun" - / "decorators.py" - ) - source = src_path.read_text(encoding="utf-8") - source = source.replace("\r\n", "\n") - - assert "from nullrun.context import (" in source, ( - "decorators.py must import from nullrun.context" - ) - # The named import must be inside the parens. - import_block = source.split("from nullrun.context import (", 1)[1] - import_block = import_block.split(")", 1)[0] - assert "get_call_tools" in import_block, ( - "decorators.py must import `get_call_tools` from " - "nullrun.context — the F01 fix reads the per-call " - "tools contextvar via this helper" - ) - - -class TestDecoratorF03BehavioralRegression: - """F03 (2026-08-22) behavioural regression: ``@protect`` / - ``@sensitive`` must populate the per-call ``_call_tools_var`` - contextvar from ``fn.__name__`` when the user did NOT explicitly - call ``set_call_context(tools=...)``. - - Pre-F03 the contextvar was never written by SDK internal code, so - the /gate round-trip triggered by ``check_workflow_budget()`` - omitted the ``tools`` field. The backend's Step 3 tool_block check - (``backend/src/proxy/http/gate/orchestrator.rs``) fail-CLOSED via - TB-1 (``no_tools_field``) and every approval-rule probe returned - ``decision=block reason='TOOL_BLOCKED'`` without ever reaching the - approval_rule_eval step. This suite exercises the decorator's - full runtime path (not the source-pin-only test in - ``TestDecoratorThreading`` above) so a future refactor that drops - the population step fails the test immediately. - - The transport layer already accepts ``tools`` on the wire body - (covered by ``TestExecuteToolsPropagation`` above). This module - closes the gap on the *decorator* leg of the handoff: from - ``@protect`` invocation through ``_protect_body`` into the - underlying runtime call. - """ - - @pytest.fixture - def captured_gate_and_execute(self): - """Capture every /gate AND /execute request body. Returns - a tuple of two mutable lists ``(gate_bodies, execute_bodies)`` - that the test can index into to assert what was sent.""" - gate_bodies: list[dict] = [] - execute_bodies: list[dict] = [] - - def _gate_capture(request: httpx.Request) -> httpx.Response: - gate_bodies.append(json.loads(request.content.decode("utf-8"))) - return httpx.Response( - 200, - json={ - "decision": "allow", - "decision_source": "gateway", - "explanation": "allowed", - "policy_version": 1, - "explanations": [], - }, - ) - - def _execute_capture(request: httpx.Request) -> httpx.Response: - execute_bodies.append(json.loads(request.content.decode("utf-8"))) - return httpx.Response( - 200, - json={ - "decision": "allow", - "decision_source": "gateway", - "explanation": "allowed", - "policy_version": 1, - }, - ) - - respx.post(GATE_URL).mock(side_effect=_gate_capture) - respx.post(EXECUTE_URL).mock(side_effect=_execute_capture) - return gate_bodies, execute_bodies - - def test_protect_populates_tools_on_gate_body_when_user_omits_set_call_context( - self, make_runtime, mock_api, captured_gate_and_execute - ): - """The decorator's ``_protect_body`` must seed - ``_call_tools_var = (fn.__name__,)`` so ``check_workflow_budget`` - forwards ``tools`` on the /gate wire body — even when the user - never called ``set_call_context``. This is the headline F03 - closure for the /gate leg.""" - gate_bodies, _ = captured_gate_and_execute - from nullrun.context import get_call_tools - - # Defensive: assert the precondition the F03 fix relies on - # (no caller-side set_call_context for this test). - assert get_call_tools() == () - - import nullrun.decorators as dec - - rt = make_runtime() - dec._runtime = rt # belt-and-braces — make_runtime already does this. - - @dec.protect - def my_agent(query: str) -> str: - return f"answer:{query}" - - result = my_agent("hello") - assert result == "answer:hello" - - # /gate must have been called once and must carry - # tools=["my_agent"] — populated by _protect_body from - # fn.__name__ before check_workflow_budget(). - assert gate_bodies, "no /gate call was captured" - gate_body = gate_bodies[-1] - assert gate_body.get("tools") == ["my_agent"], ( - f"F03 not closed: /gate body must carry tools=['my_agent'] " - f"after @protect; got body={gate_body!r}. The fix is in " - f"decorators.py::_protect_body which seeds " - f"_call_tools_var from fn.__name__ before " - f"check_workflow_budget()." - ) - - def test_protect_does_not_override_explicit_set_call_context( - self, make_runtime, mock_api, captured_gate_and_execute - ): - """When the user explicitly called - ``set_call_context(tools=[user_list])``, the decorator MUST - preserve the user's list — not overwrite it with - ``[fn.__name__]``. Precedence: explicit > auto-populated.""" - gate_bodies, _ = captured_gate_and_execute - from nullrun.context import get_call_tools, set_call_context - - # User explicitly declared their tool list. - set_call_context(tools=["user_declared_tool", "another_tool"]) - assert get_call_tools() == ("user_declared_tool", "another_tool") - - import nullrun.decorators as dec - - rt = make_runtime() - dec._runtime = rt - - @dec.protect - def my_agent(query: str) -> str: - return f"answer:{query}" - - try: - my_agent("hello") - assert gate_bodies, "no /gate call was captured" - gate_body = gate_bodies[-1] - # The user's explicit list survives — NOT fn.__name__. - assert gate_body.get("tools") == [ - "user_declared_tool", - "another_tool", - ], ( - f"explicit set_call_context must win over decorator " - f"auto-population; got body={gate_body!r}" - ) - finally: - # Clean up so the test's explicit context doesn't leak. - set_call_context(tools=[]) - assert get_call_tools() == () - - def test_protect_restores_prior_call_tools_context_after_call( - self, make_runtime, mock_api, captured_gate_and_execute - ): - """Token-based reset: a nested @protect inside an outer @protect - (or inside ``with workflow``) restores the prior - ``_call_tools_var`` value on exit. Bare @protect leaves the - contextvar empty again. The same shape as the legacy - ``_trace_id_var`` / ``_span_id_var`` resets in _protect_body.""" - from nullrun.context import _call_tools_var, get_call_tools, set_call_context - - # Outer: user explicitly set tools=["outer"] - set_call_context(tools=["outer"]) - assert get_call_tools() == ("outer",) - - import nullrun.decorators as dec - - rt = make_runtime() - dec._runtime = rt - - @dec.protect - def outer(query: str) -> str: - return f"outer:{query}" - - # Before invocation: outer context is "outer" - assert get_call_tools() == ("outer",) - - # Invoke outer — its _protect_body will see _existing="outer" - # and SKIP auto-population (call_tools_token stays None). - _ = outer("hello") - - # After invocation: outer context STILL "outer" (unchanged). - assert get_call_tools() == ("outer",), ( - f"explicit set_call_context was clobbered by the decorator " - f"after the call exited; got {get_call_tools()!r}" - ) - - # Clean up - set_call_context(tools=[]) - assert get_call_tools() == () - - def test_sensitive_decorator_populates_tools_on_execute_body( - self, make_runtime, mock_api, captured_gate_and_execute - ): - """``@sensitive`` is the decorator combo reported in DEF-LATEST_PLAN-F03 - (probes ``qa/approval_rules/ar_toolname_run.py``, - ``ar_toolname_run_chain.py``, ``ar_params_run.py``, - ``ar_threshold_run.py``). Pre-F03 the ``_enforce_sensitive_tool`` - ``runtime.execute(..., tools=get_call_tools())`` call saw an - empty contextvar, the wire body omitted ``tools``, and the - backend's Step 3 tool_block fail-CLOSED via TB-1 - (``no_tools_field``) BEFORE the approval_rule_eval step could - fire — every approval-rule probe returned - ``decision=block reason='TOOL_BLOCKED'``. - - Post-F03 the decorator populates ``_call_tools_var`` from - ``fn.__name__`` in ``_protect_body`` (before /gate and - /execute are called), and ``runtime.execute`` now accepts the - ``tools=`` kwarg directly so the source-pin pattern at - decorators.py:735 no longer TypeErrors. The /execute wire body - must carry ``tools=["refund_customer"]``. - """ - _gate_bodies, execute_bodies = captured_gate_and_execute - from nullrun.context import get_call_tools - - assert get_call_tools() == () # precondition - - import nullrun.decorators as dec - - rt = make_runtime() - dec._runtime = rt - - # The /sensitive registration flow warms the runtime - # singleton's sensitive-tools set. ``_do_sensitive_register`` - # calls ``add_sensitive_tool(fn.__name__)`` so the - # ``_enforce_sensitive_tool`` short-circuit (line 606 in - # decorators.py) doesn't return early. - @dec.sensitive - @dec.protect - def refund_customer(refund_amount: float) -> str: - return f"refund:{refund_amount}" - - # Register the tool manually (decoration-time registration - # uses the lazy singleton; in this test we pin the runtime - # directly via dec._runtime so the registration lands on the - # pinned instance — same trick make_runtime uses). - rt.add_sensitive_tool("refund_customer") - - result = refund_customer(refund_amount=100.0) - assert result == "refund:100.0" - - # /execute (the sensitive-tool round-trip) must carry - # tools=["refund_customer"] — the F03 headline closure. - assert execute_bodies, "no /execute call was captured" - execute_body = execute_bodies[-1] - assert execute_body.get("tools") == ["refund_customer"], ( - f"F03 not closed on /execute path: body must carry " - f"tools=['refund_customer']; got body={execute_body!r}. " - f"This is the symptom that broke all four approval-rule " - f"probes in LATEST_PLAN run 20260822-181500-a3f1." - ) diff --git a/tests/test_integrations_fastapi.py b/tests/test_integrations_fastapi.py index 9a8da5b..db01721 100644 --- a/tests/test_integrations_fastapi.py +++ b/tests/test_integrations_fastapi.py @@ -4,9 +4,9 @@ NullRun exception, then asserts the HTTP response matches the documented contract (status code, JSON body, headers). -Locale is pinned via the ``Accept-Language`` header (or the custom -``locale_resolver`` where relevant) so the rendered ``user_message`` -is deterministic. +The ``Accept-Language`` header is honoured by the SDK's locale +resolver (locale packs are reserved for a future release; today +the catalog is English-only). """ from __future__ import annotations @@ -210,50 +210,30 @@ def test_missing_accept_language_falls_back_to_english(): ) -def test_custom_locale_resolver_overrides_accept_language(): - """A custom resolver wins over Accept-Language — useful when the - locale comes from a session cookie or JWT claim instead.""" - def resolver(request: Request) -> str: - return request.headers.get("x-locale", "en") - +def test_install_accept_language_header_is_ignored_for_now(): + """The locale resolver was removed when ``format_user_message`` + stopped taking a ``locale=`` kwarg. ``Accept-Language`` is still + sent by the client but the SDK ignores it — the catalog is + English-only today.""" app = FastAPI() - nr_fastapi.install(app, locale_resolver=resolver) + nr_fastapi.install(app) @app.get("/trigger") def trigger(): raise exc.NullRunBudgetError("wf", "x") client = TestClient(app, raise_server_exceptions=False) - # Different Accept-Language, but the resolver forces en. resp = client.get( "/trigger", - headers={"Accept-Language": "fr-FR", "x-locale": "en"}, + headers={"Accept-Language": "fr-FR"}, ) + # English catalog is the only one today. assert resp.json()["user_message"] == ( "You've reached the usage limit for this conversation. " "Please try again later." ) -def test_resolver_exception_falls_back_to_english(): - """A buggy resolver must not crash the error response — the user - still gets a clean message, just in the default locale.""" - def bad_resolver(request: Request) -> str: - raise RuntimeError("resolver bug") - - app = FastAPI() - nr_fastapi.install(app, locale_resolver=bad_resolver) - - @app.get("/trigger") - def trigger(): - raise exc.NullRunBudgetError("wf", "x") - - client = TestClient(app, raise_server_exceptions=False) - resp = client.get("/trigger") - assert resp.status_code == 429 - assert "usage limit" in resp.json()["user_message"] - - # --------------------------------------------------------------------------- # Happy path — middleware does not interfere with normal responses # --------------------------------------------------------------------------- diff --git a/tests/test_mcp_adapter.py b/tests/test_mcp_adapter.py index 96aea6b..b70f96c 100644 --- a/tests/test_mcp_adapter.py +++ b/tests/test_mcp_adapter.py @@ -588,9 +588,11 @@ def test_call_tool_with_runtime_routes_through_execute_before_mcp_call(): assert len(runtime.calls) == 1 assert runtime.calls[0]["tool_name"] == "create_issue" assert runtime.calls[0]["input_data"] == {"repo": "acme/api"} - # Mode is forced to strict so /api/v1/execute is consulted even - # for non-sensitive MCP tools. - assert runtime.calls[0]["mode"] == "strict" + # 0.18.2: ``mode=`` opt-out was removed. Every MCP-tool + # routed through the runtime contacts /api/v1/execute + # unconditionally — no exemption for "non-sensitive" MCP + # tools, no inline bypass. + assert "mode" not in runtime.calls[0] def test_call_tool_with_runtime_blocked_does_not_invoke_mcp_client(): diff --git a/tests/test_messages.py b/tests/test_messages.py index fc39e62..40add7d 100644 --- a/tests/test_messages.py +++ b/tests/test_messages.py @@ -233,10 +233,10 @@ def test_format_user_message_handles_approval_expired_local_timeout_path(): def test_format_user_message_handles_workflow_killed_baseexception(): - """``WorkflowKilledInterrupt`` is a BaseException subclass. The - formatter must still resolve it via the inherited ``error_code`` - class attribute on ``WorkflowKilledException`` (the deprecated - parent class).""" + """``WorkflowKilledInterrupt`` carries ``error_code`` on the class + itself. The formatter must resolve it via ``error_code`` even + though kill signals are intentionally ``BaseException``-subclass + so they bypass ``except Exception``.""" killed = exc.WorkflowKilledInterrupt(workflow_id="wf-1", reason="killed via API") # NB: the formatter does NOT catch BaseException — caller's job. out = messages.format_user_message(killed) @@ -398,14 +398,6 @@ def test_format_user_message_falls_back_for_unknown_code(): assert messages.format_user_message(weird) == messages.FALLBACK_MESSAGE -def test_format_user_message_accepts_locale_kwarg(): - """Locale parameter is reserved; passing anything (including - unsupported codes) still returns a usable string.""" - budget = exc.NullRunBudgetError("wf", "x") - assert messages.format_user_message(budget, locale="en") - assert messages.format_user_message(budget, locale="ru") # falls back to en - - # --------------------------------------------------------------------------- # get_user_message — raw lookup # --------------------------------------------------------------------------- diff --git a/tests/test_money_hardening.py b/tests/test_money_hardening.py deleted file mode 100644 index 57510f9..0000000 --- a/tests/test_money_hardening.py +++ /dev/null @@ -1,427 +0,0 @@ -"""Decimal support hardening tests for the money contract. - -This module is the dedicated hardening suite for the -``MoneyImpactExtractor`` hardening pass that closed the -review gaps: - -1. **Dedicated error types** -- ``InvalidMoneyPrecisionError`` - and ``InvalidMoneyAmountError`` (both subclass - ``ValueError`` for backward compat). -2. **Negative amount rejection** -- a negative amount for - either ``money_outflow`` (debit) or ``money_inflow`` - (credit) is semantically incoherent: ``-5000 > 5000`` is - always False, so an op=gt predicate silently never fires. -3. **Overflow guard** -- the converted ``amount_minor`` must - fit in ``i64`` (the wire format). ``Decimal("1e30")`` - must be rejected, not silently wrap. -4. **Unsupported currency fallback** -- unknown ISO-4217 codes - fall back to 2 fractional digits (USD-style validation). - The fallback is conservative: ``Decimal("1.234")`` for an - unknown code raises ``InvalidMoneyPrecisionError`` because - the fallback assumed 2 digits, not 3. -5. **Serialization stability** -- ``Decimal("50")`` and - ``Decimal("50.00")`` must reduce to the same ``int(50)`` - and the same SHA-256 digest. The backend's golden hex - pin (``dfc96387...0df27``) is for ``amount_minor=5000``; - the SDK must produce that hex whether the caller types - ``int(5000)``, ``Decimal("50")``, ``Decimal("50.00")``, - ``Decimal("50.000")`` or any other trailing-zero variant. - -Why a dedicated file (not in ``tests/test_units_discriminator.py``): -the existing tests cover the unit-discriminator matrix and -the precision-validation matrix. The hardening pass is a -separate axis -- error types, sign, overflow, currency -fallback, and serialization stability -- and mixing them -into the same test classes would obscure the failure mode -when a future refactor breaks one of them. -""" - -from __future__ import annotations - -from decimal import Decimal - -import pytest - -from nullrun.business_impact import ( - OUTFLOW, - BusinessImpact, - compute_action_digest, -) -from nullrun.extractor import ( - UNIT_MAJOR, - UNIT_MINOR, - InvalidCurrencyError, - InvalidMoneyAmountError, - InvalidMoneyPrecisionError, - _to_minor_units, - business_cap_minor, - currency_minor_digits, - money_outflow, - normalize_currency, -) - -GOLDEN_HEX_USD_50_DOLLARS_OUTFLOW = ( - "dfc96387ca539b7130caebe705e042f2e34e52ab44352ae5e527bcef64f0df27" -) - - -def _refund_dollars(amount: Decimal) -> dict: - return {"amount": amount} - - -def _refund_cents(amount_cents: int) -> dict: - return {"amount_cents": amount_cents} - - -# --------------------------------------------------------------------------- -# 1. Dedicated error types -# --------------------------------------------------------------------------- - - -class TestErrorTypes: - """``InvalidMoneyPrecisionError`` and ``InvalidMoneyAmountError`` - are subclasses of ``ValueError`` (for backward compat with - ``except ValueError`` callers) and carry structured - context the operator can act on.""" - - def test_precision_error_is_value_error_subclass(self) -> None: - err = InvalidMoneyPrecisionError( - currency="USD", allowed=2, received="50.005", received_digits=3 - ) - assert isinstance(err, ValueError) - assert err.currency == "USD" - assert err.allowed == 2 - assert err.received == "50.005" - assert err.received_digits == 3 - - def test_precision_error_message_names_currency(self) -> None: - with pytest.raises(InvalidMoneyPrecisionError) as info: - _to_minor_units(Decimal("50.005"), UNIT_MAJOR, "USD") - assert "USD" in str(info.value) - assert "2" in str(info.value) - assert "50.005" in str(info.value) - - def test_amount_error_is_value_error_subclass(self) -> None: - err = InvalidMoneyAmountError(reason="negative", detail="x", currency="USD") - assert isinstance(err, ValueError) - assert err.reason == "negative" - assert err.currency == "USD" - - def test_amount_error_reason_carries_discriminator(self) -> None: - # A UI or test harness can branch on ``reason`` - # without parsing the human message. - with pytest.raises(InvalidMoneyAmountError) as info: - _to_minor_units(Decimal("-50.00"), UNIT_MAJOR, "USD") - assert info.value.reason == "negative" - assert info.value.currency == "USD" - - def test_precision_caught_by_value_error_handler(self) -> None: - # Backward compat: existing callers that catch - # ``ValueError`` still see the precision error. - with pytest.raises(ValueError): - _to_minor_units(Decimal("50.005"), UNIT_MAJOR, "USD") - - def test_amount_caught_by_value_error_handler(self) -> None: - # Backward compat for negative + overflow + non-finite. - with pytest.raises(ValueError): - _to_minor_units(Decimal("-50.00"), UNIT_MAJOR, "USD") - - -# --------------------------------------------------------------------------- -# 2. Negative amount rejection -# --------------------------------------------------------------------------- - - -class TestNegativeAmount: - """A negative ``amount_minor`` would silently fall through - every ``op=gt`` predicate (``negative < positive`` is - always False). The SDK rejects negative amounts on both - unit paths.""" - - def test_major_units_decimal_negative_rejected(self) -> None: - with pytest.raises(InvalidMoneyAmountError) as info: - _to_minor_units(Decimal("-50.00"), UNIT_MAJOR, "USD") - assert info.value.reason == "negative" - assert info.value.currency == "USD" - - def test_major_units_decimal_negative_with_precision_rejected(self) -> None: - # Negative + sub-precision: sign check fires first so - # the operator sees the most actionable error. - with pytest.raises(InvalidMoneyAmountError) as info: - _to_minor_units(Decimal("-50.005"), UNIT_MAJOR, "USD") - assert info.value.reason == "negative" - - def test_minor_units_int_negative_rejected(self) -> None: - with pytest.raises(InvalidMoneyAmountError) as info: - _to_minor_units(-5000, UNIT_MINOR, "USD") - assert info.value.reason == "negative" - - def test_minor_units_decimal_negative_rejected(self) -> None: - with pytest.raises(InvalidMoneyAmountError) as info: - _to_minor_units(Decimal("-5000"), UNIT_MINOR, "USD") - assert info.value.reason == "negative" - - def test_zero_amount_accepted(self) -> None: - # ``0`` is a valid amount (legitimate $0.00 refund, - # for example). Only negative is rejected. - assert _to_minor_units(0, UNIT_MINOR, "USD") == 0 - assert _to_minor_units(Decimal("0"), UNIT_MAJOR, "USD") == 0 - assert _to_minor_units(Decimal("0.00"), UNIT_MAJOR, "USD") == 0 - - -# --------------------------------------------------------------------------- -# 3. Overflow guard -# --------------------------------------------------------------------------- - - -class TestOverflowGuard: - """The wire format is ``i64``. Values exceeding - ``2**63 - 1 = 9_223_372_036_854_775_807`` minor units - must be rejected; silently wrapping would corrupt the - digest and the approval binding.""" - - def test_below_business_cap_accepted(self) -> None: - # The per-currency business cap (USD=$1M = 100_000_000 - # minor units) is below ``i64::MAX``; values within - # the cap are accepted. - assert _to_minor_units(99_999_999, UNIT_MINOR, "USD") == 99_999_999 - - def test_business_cap_rejected_with_reason_excessive(self) -> None: - # ``$1,000,000.01 USD = 100_000_001 minor units`` is - # above the per-call business cap. The error reason - # is ``"excessive"`` (separate from the wire-format - # overflow which is ``"overflow"``). - with pytest.raises(InvalidMoneyAmountError) as info: - _to_minor_units(100_000_001, UNIT_MINOR, "USD") - assert info.value.reason == "excessive" - assert info.value.currency == "USD" - - def test_business_cap_message_names_cap(self) -> None: - with pytest.raises(InvalidMoneyAmountError) as info: - _to_minor_units(100_000_001, UNIT_MINOR, "USD") - # The error message names the cap so the operator - # knows the threshold, not just "too large". - assert "100000000" in str(info.value) or "100_000_000" in str(info.value) - - def test_business_cap_opt_out_via_enforce_false(self) -> None: - # Batch settlement tools that already have a - # human-in-the-loop approval flow can bypass the cap. - assert ( - _to_minor_units( - 100_000_001, UNIT_MINOR, "USD", - enforce_business_cap=False, - ) - == 100_000_001 - ) - - def test_wire_format_overflow_distinct_from_business_cap(self) -> None: - # ``i64::MAX = 9_223_372_036_854_775_807`` exceeds both - # the per-currency business cap AND the wire-format - # ``i64`` upper bound. The business-cap check fires - # first because it is the lower threshold; the error - # reason is ``"excessive"`` (not ``"overflow"``). - # This separation lets the ``@protect`` wrapper route - # the call to the right policy: a ``"excessive"`` - # debit goes to the explicit human-approval path; a - # ``"overflow"`` would indicate a wire-format bug. - with pytest.raises(InvalidMoneyAmountError) as info: - _to_minor_units((1 << 63) - 1, UNIT_MINOR, "USD") - assert info.value.reason == "excessive" - - def test_wire_format_overflow_only_when_above_business_cap( - self - ) -> None: - # With ``enforce_business_cap=False``, a value at - # ``i64::MAX - 1`` is accepted (it is below - # ``i64::MAX``) but ``i64::MAX`` raises ``overflow``. - assert ( - _to_minor_units( - (1 << 63) - 1, UNIT_MINOR, "USD", - enforce_business_cap=False, - ) - == (1 << 63) - 1 - ) - with pytest.raises(InvalidMoneyAmountError) as info: - _to_minor_units( - (1 << 63), UNIT_MINOR, "USD", - enforce_business_cap=False, - ) - assert info.value.reason == "overflow" - - def test_overflow_message_names_i64_max(self) -> None: - with pytest.raises(InvalidMoneyAmountError) as info: - _to_minor_units( - (1 << 63), UNIT_MINOR, "USD", - enforce_business_cap=False, - ) - assert "i64" in str(info.value) or "9223372036854775807" in str(info.value) - - -# --------------------------------------------------------------------------- -# 4. Unsupported currency fallback -# --------------------------------------------------------------------------- - - -class TestCurrencyWhitelist: - """The ISO-4217 whitelist is enforced at extractor - construction time and at every per-currency lookup. - Unknown codes raise ``InvalidCurrencyError`` rather than - falling back to a default; this closes the conservative- - fallback gap that masked typos like ``"usd"`` or ``"USDX"``. - - Case is also enforced: ISO-4217 codes are 3-letter - uppercase ASCII letters, anything else is wrong by - definition. The SDK does NOT silently upper-case the - input. - """ - - def test_known_currency_exact(self) -> None: - for code in ("USD", "EUR", "JPY", "KWD", "BHD", "OMR", - "GBP", "CHF", "CAD", "AUD"): - assert normalize_currency(code) == code - - def test_lowercase_currency_rejected(self) -> None: - with pytest.raises(InvalidCurrencyError) as info: - normalize_currency("usd") - assert info.value.received == "usd" - assert "uppercase" in str(info.value) - - def test_mixed_case_currency_rejected(self) -> None: - with pytest.raises(InvalidCurrencyError) as info: - normalize_currency("Usd") - assert info.value.received == "Usd" - - def test_four_letter_currency_rejected(self) -> None: - with pytest.raises(InvalidCurrencyError) as info: - normalize_currency("USDX") - assert "length 4" in str(info.value) or "3-letter" in str(info.value) - - def test_empty_currency_rejected(self) -> None: - with pytest.raises(InvalidCurrencyError): - normalize_currency("") - - def test_digits_in_currency_rejected(self) -> None: - with pytest.raises(InvalidCurrencyError): - normalize_currency("US1") - - def test_constructor_rejects_lowercase_at_decoration_time(self) -> None: - # ``money_outflow(currency="usd")`` raises at - # decorator-application time, never reaches runtime. - with pytest.raises(InvalidCurrencyError): - money_outflow(argument="amount", currency="usd") - - def test_currency_minor_digits_propagates_currency_error(self) -> None: - with pytest.raises(InvalidCurrencyError): - currency_minor_digits("XYZ") - - def test_currency_minor_digits_known_value_exact(self) -> None: - assert currency_minor_digits("USD") == 2 - assert currency_minor_digits("EUR") == 2 - assert currency_minor_digits("JPY") == 0 - assert currency_minor_digits("KWD") == 3 - - def test_business_cap_lookup_propagates_currency_error(self) -> None: - with pytest.raises(InvalidCurrencyError): - business_cap_minor("XYZ") - - def test_business_cap_known_value_exact(self) -> None: - assert business_cap_minor("USD") == 100_000_000 - assert business_cap_minor("JPY") == 100_000_000 - assert business_cap_minor("KWD") == 100_000_000 - - -# --------------------------------------------------------------------------- -# 5. Serialization stability -# --------------------------------------------------------------------------- - - -class TestSerializationStability: - """``Decimal("50")`` and ``Decimal("50.00")`` must produce - the same ``amount_minor=5000`` and the same SHA-256 digest. - The cross-language golden hex pin is for ``5000`` minor - units; the SDK must produce that hex regardless of how - the caller represents the value.""" - - def test_decimal_50_int_and_decimal_50_00_same_minor(self) -> None: - # The trailing-zero variant reduces to the integer - # value. This is the canonical serialization-stability - # test. - assert _to_minor_units(Decimal("50"), UNIT_MAJOR, "USD") == 5_000 - assert _to_minor_units(Decimal("50.0"), UNIT_MAJOR, "USD") == 5_000 - assert _to_minor_units(Decimal("50.00"), UNIT_MAJOR, "USD") == 5_000 - assert _to_minor_units(Decimal("50.000"), UNIT_MAJOR, "USD") == 5_000 - assert _to_minor_units(Decimal("50.0000"), UNIT_MAJOR, "USD") == 5_000 - - def test_decimal_50_and_int_5000_produce_same_impact(self) -> None: - # ``int(5000)`` (minor) and ``Decimal("50")`` (major) - # are two different surface APIs but the same wire - # value. The extractor must produce identical - # ``BusinessImpact`` objects. - ext = money_outflow(argument="amount_cents", units=UNIT_MINOR) - impact_int = ext.impact_for(_refund_cents, (5000,), {}) - ext_major = money_outflow(argument="amount", units=UNIT_MAJOR) - impact_dec = ext_major.impact_for(_refund_dollars, (Decimal("50"),), {}) - assert impact_int.impact.amount_minor == impact_dec.impact.amount_minor - assert compute_action_digest(impact_int) == compute_action_digest(impact_dec) - - def test_decimal_50_00_50_000_produce_golden_hex(self) -> None: - # The cross-language golden hex pin must match whether - # the caller types ``Decimal("50.00")`` or - # ``Decimal("50.000")`` -- only the trailing-zero - # count differs in the caller representation, not the - # wire value. - ext = money_outflow(argument="amount", units=UNIT_MAJOR) - for repr_ in ("50", "50.0", "50.00", "50.000", "50.0000"): - impact = ext.impact_for(_refund_dollars, (Decimal(repr_),), {}) - assert impact.impact.amount_minor == 5_000 - assert ( - compute_action_digest(impact) - == GOLDEN_HEX_USD_50_DOLLARS_OUTFLOW - ) - - def test_decimal_50_99_minor_path_also_stable(self) -> None: - # The ``units="minor"`` path also accepts Decimal if - # it is already integer-valued. ``Decimal("50.99")`` - # is rejected because the fractional part is non-zero; - # the integer-valued variant ``Decimal("5099")`` is - # accepted and produces the same minor value as - # ``int(5099)``. - assert _to_minor_units(Decimal("5099"), UNIT_MINOR, "USD") == 5_099 - assert _to_minor_units(5099, UNIT_MINOR, "USD") == 5_099 - - -# --------------------------------------------------------------------------- -# 6. Wire-format invariant (round-trip through ``MoneyImpact``) -# --------------------------------------------------------------------------- - - -class TestWireFormatInvariant: - """The hardening pass should not change the wire format. - ``amount_minor`` is an ``i64`` with a fixed scale per - currency. These tests pin that contract.""" - - def test_amount_minor_is_python_int(self) -> None: - # ``i64`` on the wire; ``int`` in Python. The hardening - # pass must not introduce ``Decimal`` or ``float`` on - # the wire. - ext = money_outflow(argument="amount", units=UNIT_MAJOR) - impact = ext.impact_for(_refund_dollars, (Decimal("50.99"),), {}) - assert type(impact.impact.amount_minor) is int - - def test_amount_minor_is_non_negative(self) -> None: - # Combined with the negative-amount rejection: the - # wire value is always ``>= 0`` (the negative-amount - # guard raises before the conversion). - ext = money_outflow(argument="amount", units=UNIT_MAJOR) - impact = ext.impact_for(_refund_dollars, (Decimal("0.00"),), {}) - assert impact.impact.amount_minor == 0 - impact_large = ext.impact_for(_refund_dollars, (Decimal("9999.99"),), {}) - assert impact_large.impact.amount_minor >= 0 - - def test_currency_passes_through_unchanged(self) -> None: - # The hardening pass must not change ``currency`` -- - # the backend predicate evaluator compares it - # exactly. - ext = money_outflow(argument="amount", currency="USD", units=UNIT_MAJOR) - impact = ext.impact_for(_refund_dollars, (Decimal("50.99"),), {}) - assert impact.impact.currency == "USD" \ No newline at end of file diff --git a/tests/test_no_local_policy.py b/tests/test_no_local_policy.py index 01354a0..23b6245 100644 --- a/tests/test_no_local_policy.py +++ b/tests/test_no_local_policy.py @@ -191,7 +191,8 @@ def test_runtime_init_default_fallback_mode_is_strict(): def test_runtime_init_permissive_kwarg_still_opt_in(): - """v3.53 audit #4 — explicit fallback_mode="permissive" still opt-in. + """v3.53 audit #4 — explicit fallback_mode=FallbackMode.PERMISSIVE + is still opt-in. The legacy behavior must remain reachable for dev / test harnesses that intentionally run without a live policy engine. This test @@ -205,7 +206,7 @@ def test_runtime_init_permissive_kwarg_still_opt_in(): api_key="nr_test_dummy_for_v3_53_source_pin", _test_mode=True, polling=False, - fallback_mode="permissive", + fallback_mode=FallbackMode.PERMISSIVE, ) assert rt._fallback_mode == FallbackMode.PERMISSIVE diff --git a/tests/test_observability.py b/tests/test_observability.py deleted file mode 100644 index 3a2e7e8..0000000 --- a/tests/test_observability.py +++ /dev/null @@ -1,413 +0,0 @@ -""" -Tests for observability module — MetricsRegistry integration. -""" - -import httpx -import pytest -import respx - -from nullrun.observability import MetricsRegistry, metrics - - -@pytest.fixture(autouse=True) -def reset_metrics(): - """Reset metrics before each test.""" - metrics.reset() - yield - metrics.reset() - - -class TestMetricsRegistry: - def test_to_dict_has_correct_structure(self): - d = metrics.to_dict() - assert "transport" in d - assert "runtime" in d - # transport fields - assert "events_enqueued" in d["transport"] - assert "events_sent" in d["transport"] - assert "events_dropped" in d["transport"] - assert "batches_sent" in d["transport"] - assert "batches_failed" in d["transport"] - assert "circuit_breaker_opens" in d["transport"] - # runtime fields - assert "track_calls" in d["runtime"] - assert "execute_calls" in d["runtime"] - assert "execute_allowed" in d["runtime"] - assert "execute_blocked" in d["runtime"] - - def test_reset_clears_all_counters(self): - metrics.transport.events_enqueued = 42 - metrics.runtime.track_calls = 10 - metrics.reset() - assert metrics.transport.events_enqueued == 0 - assert metrics.runtime.track_calls == 0 - - def test_independent_registry_instances(self): - """Different instances of MetricsRegistry are independent.""" - reg1 = MetricsRegistry() - reg2 = MetricsRegistry() - reg1.transport.events_sent = 100 - assert reg2.transport.events_sent == 0 - - def test_track_increments_counter(self, mock_api, make_runtime): - """track() updates metrics.runtime.track_calls.""" - rt = make_runtime() - assert metrics.runtime.track_calls == 0 - rt.track({"event_type": "test"}) - assert metrics.runtime.track_calls == 1 - rt.track({"event_type": "test2"}) - assert metrics.runtime.track_calls == 2 - - def test_execute_increments_allowed_counter(self, mock_api, make_runtime): - """execute() when allowed=True updates execute_allowed.""" - respx.post(f"{BASE_URL}/api/v1/gate").mock( - return_value=httpx.Response( - 200, - json={ - "decision": "allow", - "decision_source": "gateway", - "explanation": "allowed", - "policy_version": 1, - }, - ) - ) - rt = make_runtime() - rt.execute(tool_name="gpt-4", input_data={}, mode="strict") - assert metrics.runtime.execute_calls == 1 - assert metrics.runtime.execute_allowed == 1 - assert metrics.runtime.execute_blocked == 0 - - def test_execute_increments_blocked_counter(self, mock_api, make_runtime): - """execute() when blocked=True updates execute_blocked.""" - # Audit F-R2-01 (2026-06-22): Transport.execute now hits - # /api/v1/execute (not /gate) so the backend checks the - # `execute` scope. The mock needs to move with the contract. - respx.post(f"{BASE_URL}/api/v1/execute").mock( - return_value=httpx.Response( - 200, - json={ - "decision": "block", - "explanation": "cost_limit_exceeded", - "decision_source": "gateway", - "policy_version": 1, - }, - ) - ) - rt = make_runtime() - from nullrun.breaker.exceptions import NullRunBlockedException - - try: - rt.execute(tool_name="gpt-4", input_data={}, mode="strict") - except NullRunBlockedException: - pass # expected - assert metrics.runtime.execute_blocked == 1 - - def test_enqueue_increments_events_enqueued(self, mock_api, make_runtime): - """track() increments events_enqueued.""" - rt = make_runtime() - rt.track({"event_type": "e1"}) - rt.track({"event_type": "e2"}) - assert metrics.transport.events_enqueued >= 2 - - -class TestThreadSafeMetrics: - def test_inc_transport_increments_counter(self): - """inc_transport increments transport metrics safely.""" - metrics.reset() - metrics.inc_transport("events_enqueued") - assert metrics.transport.events_enqueued == 1 - - def test_inc_transport_with_value(self): - """inc_transport with value parameter increments by that amount.""" - metrics.reset() - metrics.inc_transport("events_sent", 50) - assert metrics.transport.events_sent == 50 - - def test_inc_runtime_increments_counter(self): - """inc_runtime increments runtime metrics safely.""" - metrics.reset() - metrics.inc_runtime("track_calls") - assert metrics.runtime.track_calls == 1 - - def test_inc_runtime_with_value(self): - """inc_runtime with value parameter increments by that amount.""" - metrics.reset() - metrics.inc_runtime("execute_calls", 5) - assert metrics.runtime.execute_calls == 5 - - def test_set_transport_sets_field(self): - """set_transport sets transport metric fields safely.""" - metrics.reset() - metrics.set_transport("last_error", "Test error") - assert metrics.transport.last_error == "Test error" - - def test_set_transport_last_flush_at(self): - """set_transport works for timestamp fields.""" - metrics.reset() - import time - - ts = time.monotonic() - metrics.set_transport("last_flush_at", ts) - assert metrics.transport.last_flush_at == ts - - def test_to_dict_while_incrementing(self, mock_api, make_runtime): - """to_dict() returns consistent snapshot while metrics are being updated.""" - metrics.reset() - # Start incrementing in a tight loop while reading to_dict - import threading - - errors = [] - - def incrementer(): - try: - for _ in range(100): - metrics.inc_transport("events_enqueued") - except Exception as e: - errors.append(e) - - def reader(): - try: - for _ in range(50): - d = metrics.to_dict() - # Just verify structure is consistent - assert "transport" in d - assert "events_enqueued" in d["transport"] - except Exception as e: - errors.append(e) - - t1 = threading.Thread(target=incrementer) - t2 = threading.Thread(target=reader) - t1.start() - t2.start() - t1.join() - t2.join() - - assert len(errors) == 0 - - -# Module-level import for test -BASE_URL = "https://api.test.nullrun.io" - - -# =========================================================================== -# B23/B24: every metric field must be wired up -# =========================================================================== -# Before the B23/B24 follow-up: 6 fields were defined on the dataclasses -# but never incremented: -# - TransportMetrics: retries_total, circuit_breaker_opens -# fallback_mode_activations, timeouts, last_error -# - RuntimeMetrics: cost_limit_exceeded -# These tests pin the wiring so a future regression that -# removes an increment call breaks here, not in production. - - -class TestAllMetricsWired: - """Every metric field on TransportMetrics / RuntimeMetrics - must be incremented by at least one call-site in the SDK. - - The "is_callable_from_real_path" check below is intentionally - indirect: rather than mocking the metric counters, we - reset the global ``metrics`` instance and exercise the - code paths that should bump each field, then assert - non-zero. - """ - - def _reset_metrics(self): - """Reset the global metrics singleton to a clean state.""" - from nullrun.observability import metrics - - metrics.reset() - return metrics - - def test_retries_total_incremented_by_retry(self): - """A retried HTTP request must bump ``retries_total``.""" - from nullrun.observability import metrics - from nullrun.transport import _retry_with_backoff - - self._reset_metrics() - attempts = [] - - def _flaky(): - attempts.append(1) - # First 2 attempts fail; 3rd succeeds. With - # max_retries=5, the helper would let the 3rd - # attempt go through, so we expect retries_total=2 - # (one retry for each of the first two failures). - if len(attempts) <= 2: - raise httpx.ConnectError("test", request=httpx.Request("GET", "http://x")) - return "ok" - - result = _retry_with_backoff(_flaky, max_retries=5, base_delay=0.0) - assert result == "ok" - - # Two retries happened (attempts 1 and 2 failed, attempt 3 - # succeeded). retries_total increments PER RETRY, not - # attempt, so it should be 2. - assert metrics.transport.retries_total == 2, ( - f"retries_total expected 2 after 2 failed attempts; " - f"got {metrics.transport.retries_total}" - ) - - def test_timeouts_incremented_on_httpx_timeout(self): - """``httpx.TimeoutException`` must bump ``timeouts``.""" - from nullrun.breaker.exceptions import BreakerTransportError - from nullrun.observability import metrics - from nullrun.transport import _retry_with_backoff - - self._reset_metrics() - attempts = [] - - def _slow(): - attempts.append(1) - raise httpx.ReadTimeout("test", request=httpx.Request("GET", "http://x")) - - # All 3 attempts fail; helper wraps the final failure in - # ``BreakerTransportError`` per the public contract. - with pytest.raises(BreakerTransportError): - _retry_with_backoff(_slow, max_retries=2, base_delay=0.0) - - # ``timeouts`` is incremented on EVERY timeout (not just - # the final one), so it should equal 3 (3 attempts). - assert metrics.transport.timeouts >= 2, ( - f"timeouts did not increment on ReadTimeout; got {metrics.transport.timeouts}" - ) - - def test_last_error_set_on_failure(self): - """``last_error`` must be set when a request fails.""" - from nullrun.breaker.exceptions import BreakerTransportError - from nullrun.observability import metrics - from nullrun.transport import _retry_with_backoff - - self._reset_metrics() - - def _fail(): - raise httpx.ConnectError("connection refused", request=httpx.Request("GET", "http://x")) - - # max_retries=0 means only 1 attempt — fail fast. The - # helper wraps the final failure in BreakerTransportError. - with pytest.raises(BreakerTransportError): - _retry_with_backoff(_fail, max_retries=0, base_delay=0.0) - - assert metrics.transport.last_error is not None, ( - "last_error was not set after a failed request" - ) - assert "ConnectError" in metrics.transport.last_error - - def test_circuit_breaker_opens_incremented_on_open_transition(self): - """Transitioning to OPEN must bump ``circuit_breaker_opens``.""" - from nullrun.breaker.circuit_breaker import CBState, CircuitBreaker - from nullrun.observability import metrics - - self._reset_metrics() - cb = CircuitBreaker( - failure_threshold=1, - recovery_timeout=30.0, - redis_client=None, - ) - - def _fail(): - raise RuntimeError("boom") - - with pytest.raises(Exception): - cb.call(_fail) - - assert metrics.transport.circuit_breaker_opens >= 1, ( - f"circuit_breaker_opens did not increment after a failure; " - f"got {metrics.transport.circuit_breaker_opens}" - ) - assert cb._state == CBState.OPEN # noqa: SLF001 - - def test_cost_limit_exceeded_incremented_on_block(self): - """A pre-flight decision=block must bump ``cost_limit_exceeded``.""" - from nullrun.breaker.exceptions import NullRunBudgetError - from nullrun.observability import metrics - from nullrun.runtime import NullRunRuntime - - self._reset_metrics() - # Use _test_mode=True so NullRunRuntime skips the auth - # handshake / policy fetch; the underlying httpx client - # is real and we mock its /check endpoint with respx. - import respx - from httpx import Response - - with respx.mock(assert_all_called=False) as mock: - # The transport's ``check `` method POSTs to - # /api/v1/gate (unified endpoint), not /api/v1/check. - mock.post("https://api.test.nullrun.io/api/v1/gate").mock( - return_value=Response( - 200, - json={ - "decision": "block", - "explanations": ["cost limit exceeded"], - }, - ) - ) - rt = NullRunRuntime( - api_key="test-key-12345678", - api_url="https://api.test.nullrun.io", - polling=False, - _test_mode=True, - ) - # Force-set the workflow_id so the pre-flight check - # actually runs (legacy keys would otherwise skip - # it per runtime.py:996). - rt.workflow_id = "wf-cost-test" - try: - with pytest.raises(NullRunBudgetError): - rt.check_workflow_budget() - finally: - rt.shutdown() - - assert metrics.runtime.cost_limit_exceeded >= 1, ( - f"cost_limit_exceeded did not increment on decision=block; " - f"got {metrics.runtime.cost_limit_exceeded}" - ) - - def test_fallback_mode_activations_incremented_on_transport_error(self): - """A transport error during ``execute()`` must bump ``fallback_mode_activations``.""" - from nullrun.observability import metrics - from nullrun.transport import Transport - - self._reset_metrics() - # respx mock that returns 5xx for /gate — triggers the - # fallback path inside transport.execute. - import respx - from httpx import Response - - with respx.mock(assert_all_called=False) as mock: - mock.post("https://api.test.nullrun.io/api/v1/execute").mock( - return_value=Response(500, json={"error": "boom"}) - ) - t = Transport( - api_url="https://api.test.nullrun.io", - api_key="test-key-12345678", - secret_key="test-secret", - ) - # Pin a small retry budget so the 5xx test does not spend - # the full retry window (default 10 attempts × 30s backoff - # cap = 64s+ on the test runner's deadline). The metric - # we assert (fallback_mode_activations) is bumped on the - # FIRST attempt — the retry count is incidental. - t._execute_max_retries = 1 - t.start() - try: - # The exact return shape depends on fallback_mode - # (PERMISSIVE → allow, STRICT → block). The - # fallback_mode_activations counter is bumped - # before the mode is applied, so the value of - # the returned dict doesn't matter for this - # test. - t.execute( - organization_id="org-1", - execution_id="wf-x", - trace_id="trace-1", - tool="t", - input_data={}, - ) - finally: - t.stop() - - assert metrics.transport.fallback_mode_activations >= 1, ( - f"fallback_mode_activations did not increment on transport " - f"error; got {metrics.transport.fallback_mode_activations}" - ) diff --git a/tests/test_preflight_fail_policy.py b/tests/test_preflight_fail_policy.py deleted file mode 100644 index 70c1675..0000000 --- a/tests/test_preflight_fail_policy.py +++ /dev/null @@ -1,994 +0,0 @@ -""" -Regression tests for the three pre-execution-gate bugs fixed by ADR-008. - -Bug #1 — `check_workflow_budget` was fail-CLOSED on network error - (transport swallowed the error and returned a synthetic - block; runtime re-interpreted it as a real policy block). - Fix: fail-OPEN. Network / 5xx / breaker-open is logged and - the call returns. - -Bug #2 — `_enforce_sensitive_tool` was fail-OPEN on transport error - (transport returned a synthetic allow with - `decision_source=FALLBACK_*`; the decorator trusted it). - Fix: fail-CLOSED. Body must not run when the policy engine - is unreachable. NULLRUN_SENSITIVE_FAIL_OPEN=1 stays as the - explicit opt-out for dev / test. - -Bug #3 — `@protect` did not call `check_control_plane` at all, so - a dashboard KILL was silently ignored for `@protect`-only - code paths. Fix: control-plane check runs FIRST inside - `@protect`, before budget and before sensitive-tool. - -These tests also exercise the per-call `on_transport_error` plumbing -on `transport.execute` / `transport.check` and the new -`NullRunTransportError` / `TransportErrorSource` exception pair. -""" - -import httpx -import pytest -import respx - -import nullrun -from nullrun.breaker.exceptions import ( - NullRunBlockedException, - NullRunTransportError, - TransportErrorSource, - WorkflowKilledInterrupt, -) - -# Base URL used in tests -BASE_URL = "https://api.test.nullrun.io" - - -# ────────────────────────────────────────────────────────────── -# Helpers — RecordingRuntime (no-op transport, full gate behavior) -# ────────────────────────────────────────────────────────────── - - -class _RecordingRuntime: - """ - Stand-in runtime that records events but does NOT call any - network or call real gates. Used to isolate the - `check_control_plane` invocation order in the @protect wrapper - from the other two gates. - - The real `check_control_plane` and `check_workflow_budget` would - normally make HTTP calls; for the bug-#3 regression we wire a - Killed state directly into `_remote_states` (the same internal - field the WS-push handler updates). - """ - - def __init__(self) -> None: - self.events: list[dict] = [] - self._remote_states: dict = {} - self._sensitive_tools: set = set() - self._strict_mode_tools: set = set() - # Order of gate calls recorded by `_record_gate` below - self.gate_calls: list[str] = [] - - def is_sensitive_tool(self, tool_name: str) -> bool: - return tool_name in self._sensitive_tools - - def add_sensitive_tool(self, tool_name: str) -> None: - self._strict_mode_tools.add(tool_name) - - def track_event(self, event_type: str, **kwargs) -> None: - self.events.append({"type": event_type, **kwargs}) - - def track_tool(self, tool_name: str, **kwargs) -> None: - # Commit 33d2b5f wires ``@protect`` to emit a tools/track_tool event - # after the wrapped body returns. The stub captures that emit the - # same way it captures the other track paths so the gate-order - # assertions keep working unchanged. - self.events.append({"type": "tool_call", "tool_name": tool_name, **kwargs}) - - # The two gates we want to track, in order. The decorator - # calls them — we record the call sequence. - - def check_control_plane(self, workflow_id) -> None: - self.gate_calls.append("control_plane") - state = self._remote_states.get(workflow_id or "default", {}) - s = state.get("state", "Normal") - if s == "Killed": - raise WorkflowKilledInterrupt( - workflow_id=workflow_id or "default", - reason=state.get("reason", "killed"), - ) - - def check_workflow_budget(self) -> None: - self.gate_calls.append("budget") - - def _bump_protect_count(self) -> None: - # 2026-09-22: zero-activity diagnostic counter on the real - # NullRunRuntime. Tests don't exercise the warn-once - # behaviour, so a no-op stub keeps the @protect call path - # runnable without dragging the diagnostic state into - # gate-order assertions. - return None - - def execute(self, tool_name, input_data, mode="auto"): - self.gate_calls.append("sensitive") - if not self.is_sensitive_tool(tool_name): - return {"decision": "allow", "decision_source": "gateway"} - # If sensitive, callers can pre-arrange `self._next_execute_return` - # / `self._next_execute_raise` to drive the assertion. - if self._next_execute_raise is not None: - exc = self._next_execute_raise - self._next_execute_raise = None - raise exc - ret = self._next_execute_return - self._next_execute_return = None - if ret is None: - return {"decision": "allow", "decision_source": "gateway"} - return ret - - _next_execute_return = None - _next_execute_raise = None - - -# ────────────────────────────────────────────────────────────── -# Bug #1 — check_workflow_budget fail-OPEN -# ────────────────────────────────────────────────────────────── - - -class TestCheckWorkflowBudgetFailOpen: - def test_network_error_returns_normally(self, make_runtime, mock_api): - """httpx.ConnectError on /gate → check_workflow_budget returns - normally (fail-OPEN). Regression for bug #1 — the old code - re-interpreted a swallowed exception as a real block.""" - respx.post(f"{BASE_URL}/api/v1/gate").mock( - side_effect=httpx.ConnectError("connection refused") - ) - rt = make_runtime() - # Must NOT raise — the gate is fail-OPEN on transport error. - rt.check_workflow_budget() - - def test_timeout_returns_normally(self, make_runtime, mock_api): - """httpx.TimeoutException on /gate → returns normally.""" - respx.post(f"{BASE_URL}/api/v1/gate").mock( - side_effect=httpx.TimeoutException("read timeout") - ) - rt = make_runtime() - rt.check_workflow_budget() - - def test_5xx_returns_normally(self, make_runtime, mock_api): - """HTTP 500 from /gate → returns normally.""" - respx.post(f"{BASE_URL}/api/v1/gate").mock(return_value=httpx.Response(500, text="boom")) - rt = make_runtime() - rt.check_workflow_budget() - - def test_real_block_raises_budget_error(self, make_runtime, mock_api): - """Real `decision=block` from gateway raises ``NullRunBudgetError`` - (a ``NullRunBlockedException`` subclass). The fix for bug #1 must - NOT swallow real policy decisions — only transport errors.""" - from nullrun.breaker.exceptions import NullRunBudgetError - - respx.post(f"{BASE_URL}/api/v1/gate").mock( - return_value=httpx.Response( - 200, - json={ - "decision": "block", - "explanations": ["budget_exceeded"], - }, - ) - ) - rt = make_runtime() - with pytest.raises(NullRunBudgetError): - rt.check_workflow_budget() - - def test_real_throttle_raises_paused(self, make_runtime, mock_api): - """`decision=throttle` still raises WorkflowPausedException.""" - respx.post(f"{BASE_URL}/api/v1/gate").mock( - return_value=httpx.Response( - 200, - json={ - "decision": "throttle", - "explanations": ["soft limit"], - }, - ) - ) - rt = make_runtime() - from nullrun.breaker.exceptions import WorkflowPausedException - - with pytest.raises(WorkflowPausedException): - rt.check_workflow_budget() - - def test_decision_source_is_typed_for_audit(self, make_runtime, mock_api): - """On 5xx the runtime layer must NOT lose the failure - classification — the transport layer should set one of the - three FALLBACK_* values in `decision_source` (or, with the - new "raise" policy, raise NullRunTransportError). This guards - the audit-trail leg of bug #1 (operators can tell "server - said block" from "server did not respond").""" - respx.post(f"{BASE_URL}/api/v1/gate").mock( - return_value=httpx.Response(503, text="Service Unavailable") - ) - rt = make_runtime() - # Fail-OPEN: no raise, no silent block. - rt.check_workflow_budget() - - -# ────────────────────────────────────────────────────────────── -# Sprint handoff Bug #4 — observability closure on fail-OPEN paths -# ────────────────────────────────────────────────────────────── -# -# The fail-OPEN posture is documented as authoritative (ADR-008 + -# the top-of-file docstring table on `check_workflow_budget`). The -# sprint handoff `enforcement-certainty-sprint-handoff.md` flagged -# that pre-fix the FALLBACK `decision_source` path emitted DEBUG-level -# logs and had no metric -- making the silent bypass invisible to -# operators tailing INFO+ logs and unreachable for alerting. -# -# These tests pin the observability closure (WARNING log + metric -# increment) without changing the fail-OPEN behaviour itself. -# Existing tests above continue to assert "body runs, no raise". - - -class TestCheckWorkflowBudgetObservability: - """Source-pin regression suite for the sprint handoff Bug #4 fix.""" - - def test_network_error_emits_warning_and_metric(self, make_runtime, mock_api, caplog): - """httpx.ConnectError on /gate → WARNING log + gate_fail_open_total+=1. - - Pre-fix this path logged at WARNING already (see the existing - ``test_network_error_returns_normally`` behaviour) but emitted - no metric, so a sustained outage looked identical to a - one-off blip. The metric is the operator's primary signal. - """ - import logging - - from nullrun.observability import metrics - - before = metrics.runtime.gate_fail_open_total - respx.post(f"{BASE_URL}/api/v1/gate").mock( - side_effect=httpx.ConnectError("connection refused") - ) - rt = make_runtime() - with caplog.at_level(logging.WARNING, logger="nullrun.runtime"): - rt.check_workflow_budget() - # Metric incremented exactly once for the single fail-OPEN. - assert metrics.runtime.gate_fail_open_total == before + 1 - # At least one WARNING from check_workflow_budget's fail-OPEN - # path was emitted (the exact wording is not pinned -- only - # the level, since the docblock on the method already declares - # "logged at warning level" as the contract). - warnings = [ - r for r in caplog.records - if r.name == "nullrun.runtime" - and r.levelno == logging.WARNING - and "check_workflow_budget" in r.getMessage() - ] - assert warnings, "expected WARNING log from check_workflow_budget fail-OPEN" - - def test_timeout_emits_warning_and_metric(self, make_runtime, mock_api, caplog): - """httpx.TimeoutException on /gate → WARNING log + metric++. - - Same contract as the ConnectError test; covers the timeout - code path that the journal evidence flagged (transport.py - returns synthetic-block with DecisionSource.FALLBACK after - exhausting retries). - """ - import logging - - from nullrun.observability import metrics - - before = metrics.runtime.gate_fail_open_total - respx.post(f"{BASE_URL}/api/v1/gate").mock( - side_effect=httpx.TimeoutException("read timeout") - ) - rt = make_runtime() - with caplog.at_level(logging.WARNING, logger="nullrun.runtime"): - rt.check_workflow_budget() - assert metrics.runtime.gate_fail_open_total == before + 1 - - def test_synthetic_fallback_source_emits_warning_not_debug( - self, make_runtime, mock_api, caplog - ): - """When transport returns 5xx, it emits ``decision_source = - FALLBACK_*`` (synthetic-block). Pre-fix the runtime logged - this at DEBUG, which violated the docblock contract ("logged - at warning level"). Pin WARNING post-fix. - """ - import logging - - from nullrun.observability import metrics - - before = metrics.runtime.gate_fail_open_total - respx.post(f"{BASE_URL}/api/v1/gate").mock( - return_value=httpx.Response( - 503, - json={ - "decision": "block", - "decision_source": "FALLBACK_NETWORK_ERROR", - "explanation": "Gateway unavailable", - }, - ) - ) - rt = make_runtime() - with caplog.at_level(logging.WARNING, logger="nullrun.runtime"): - rt.check_workflow_budget() - # Metric incremented. - assert metrics.runtime.gate_fail_open_total == before + 1 - # No DEBUG-level record from check_workflow_budget's synthetic - # fallback arm -- the only post-fix log level for that path - # is WARNING. (Other DEBUG records from unrelated code paths - # may exist; we filter by message prefix.) - debug_fallback = [ - r for r in caplog.records - if r.name == "nullrun.runtime" - and r.levelno == logging.DEBUG - and "synthetic decision_source" in r.getMessage() - ] - assert not debug_fallback, ( - "synthetic decision_source arm must NOT log at DEBUG -- " - "this was the sprint handoff Bug #4 silent-fail-OPEN bug" - ) - - def test_real_block_does_not_increment_metric(self, make_runtime, mock_api): - """Real `decision=block` from the gateway is a policy block, - NOT a transport fail-OPEN -- must NOT increment the - gate_fail_open_total counter. Guards against a future - refactor that mistakenly moves the metric emit above the - decision-parse stage. - """ - from nullrun.breaker.exceptions import NullRunBudgetError - from nullrun.observability import metrics - - before = metrics.runtime.gate_fail_open_total - respx.post(f"{BASE_URL}/api/v1/gate").mock( - return_value=httpx.Response( - 200, - json={ - "decision": "block", - "decision_source": "gateway", - "explanations": ["budget_exceeded"], - }, - ) - ) - rt = make_runtime() - with pytest.raises(NullRunBudgetError): - rt.check_workflow_budget() - assert metrics.runtime.gate_fail_open_total == before, ( - "real policy block must not increment the fail-OPEN metric" - ) - - def test_real_allow_does_not_increment_metric(self, make_runtime, mock_api): - """Real `decision=allow` from the gateway is the happy path -- - must NOT increment the fail-OPEN counter. Pin the - allow-path stays allow.""" - from nullrun.observability import metrics - - before = metrics.runtime.gate_fail_open_total - respx.post(f"{BASE_URL}/api/v1/gate").mock( - return_value=httpx.Response( - 200, - json={"decision": "allow", "decision_source": "gateway"}, - ) - ) - rt = make_runtime() - rt.check_workflow_budget() # must not raise - assert metrics.runtime.gate_fail_open_total == before - - def test_to_dict_includes_gate_fail_open_total(self): - """Pin the metric field is reachable from /health via - ``metrics.to_dict()`` so operator dashboards can graph it. - Without this pin, a future refactor that adds the counter - to RuntimeMetrics but forgets to_dict would silently break - observability -- the field exists, the JSON shape doesn't. - """ - from nullrun.observability import metrics - - d = metrics.to_dict() - assert "gate_fail_open_total" in d["runtime"], ( - "metrics.to_dict() must expose gate_fail_open_total for " - "/health and operator dashboards" - ) - - -# ────────────────────────────────────────────────────────────── -# Bug #2 — _enforce_sensitive_tool fail-CLOSED on transport error -# ────────────────────────────────────────────────────────────── - - -class TestEnforceSensitiveToolFailClosed: - def _build_protected_sensitive_tool(self, mock_api, make_runtime): - """ - Build a runtime + a `@protect`-wrapped `@sensitive` tool. - Returns (rt, call_counter) — the counter increments only - if the body actually runs. - """ - rt = make_runtime() - rt.add_sensitive_tool("charge_card") - - calls = {"n": 0} - - @nullrun.sensitive - @nullrun.protect - def charge_card(amount: int) -> str: - calls["n"] += 1 - return f"charged {amount}" - - return rt, charge_card, calls - - def test_transport_error_fails_closed(self, make_runtime, mock_api, monkeypatch): - """Network error on /execute → NullRunBlockedException - body does NOT run. Regression for bug #2.""" - respx.post(f"{BASE_URL}/api/v1/execute").mock( - side_effect=httpx.ConnectError("connection refused") - ) - rt, charge_card, calls = self._build_protected_sensitive_tool(mock_api, make_runtime) - - with pytest.raises(NullRunBlockedException) as exc_info: - charge_card(100) - assert calls["n"] == 0, "body ran on transport error — bug #2 regression" - # The reason must mention the policy engine (audit-trail hint). - assert "policy engine" in (exc_info.value.reason or "").lower() - - def test_classified_transport_error_surfaces_source(self, make_runtime, mock_api): - """The reason on the raised NullRunBlockedException includes - the classified source (NETWORK_ERROR / GATEWAY_ERROR / - BREAKER_OPEN) so the audit trail can distinguish them.""" - respx.post(f"{BASE_URL}/api/v1/execute").mock( - side_effect=httpx.ConnectError("connection refused") - ) - rt, charge_card, calls = self._build_protected_sensitive_tool(mock_api, make_runtime) - - with pytest.raises(NullRunBlockedException) as exc_info: - charge_card(100) - # Source is the new TransportErrorSource value - assert TransportErrorSource.NETWORK_ERROR in (exc_info.value.reason or "") - - def test_5xx_fails_closed(self, make_runtime, mock_api): - """HTTP 5xx on /execute → NullRunBlockedException, body - does not run.""" - # Audit F-R2-01 (2026-06-22): sensitive-tool enforcement now - # hits /api/v1/execute (was /gate). The mock must follow. - respx.post(f"{BASE_URL}/api/v1/execute").mock( - return_value=httpx.Response(502, text="Bad Gateway") - ) - rt, charge_card, calls = self._build_protected_sensitive_tool(mock_api, make_runtime) - - with pytest.raises(NullRunBlockedException): - charge_card(100) - assert calls["n"] == 0 - - def test_defense_in_depth_fallback_source_fails_closed(self, make_runtime, mock_api): - """Even if `runtime.execute` returns a dict with - `decision_source` starting with `FALLBACK_*` (e.g. a future - regression drops the `on_transport_error="raise"` argument) - the decorator MUST still raise NullRunBlockedException. This - is the "defense in depth" path in ADR-008 Rule 1 / Rule 2. - - Simulated by injecting a runtime that returns the - synthetic-allow result directly (bypassing transport).""" - # Build a runtime that returns a FALLBACK_* decision - rt = make_runtime() - rt.add_sensitive_tool("charge_card") - # Override execute to return a synthetic allow with - # FALLBACK_NETWORK_ERROR source. This is what an older - # `fallback_mode=PERMISSIVE` transport would have produced. - rt.execute = lambda *a, **kw: { - "decision": "allow", - "decision_source": TransportErrorSource.NETWORK_ERROR, - } - - calls = {"n": 0} - - @nullrun.sensitive - @nullrun.protect - def charge_card(amount: int) -> str: - calls["n"] += 1 - return "ok" - - with pytest.raises(NullRunBlockedException): - charge_card(100) - assert calls["n"] == 0, "body ran on FALLBACK_* source — bug #2 regression" - - def test_opt_out_allows_body_when_engine_absent(self, make_runtime, mock_api, monkeypatch): - """NULLRUN_SENSITIVE_FAIL_OPEN=1 explicitly opts the user - back into fail-OPEN behavior — for dev / test environments - where the policy engine is intentionally absent.""" - monkeypatch.setenv("NULLRUN_SENSITIVE_FAIL_OPEN", "1") - respx.post(f"{BASE_URL}/api/v1/execute").mock( - side_effect=httpx.ConnectError("connection refused") - ) - rt, charge_card, calls = self._build_protected_sensitive_tool(mock_api, make_runtime) - - result = charge_card(100) - assert result == "charged 100" - assert calls["n"] == 1 - - def test_real_block_still_honored(self, make_runtime, mock_api): - """A real `decision=block` from the gateway (not a transport - error) must STILL raise NullRunBlockedException. The - fail-CLOSED rule applies to *both* transport failure and - real policy blocks — the opt-out is scoped to transport - errors only.""" - # Audit F-R2-01 (2026-06-22): /api/v1/execute is the canonical - # sensitive-tool route. /api/v1/gate is reserved for budget - # pre-flight only. - respx.post(f"{BASE_URL}/api/v1/execute").mock( - return_value=httpx.Response( - 200, - json={ - "decision": "block", - "explanation": "blocked by policy", - "decision_source": "gateway", - "policy_version": 1, - }, - ) - ) - rt, charge_card, calls = self._build_protected_sensitive_tool(mock_api, make_runtime) - - with pytest.raises(NullRunBlockedException): - charge_card(100) - assert calls["n"] == 0 - - -# ────────────────────────────────────────────────────────────── -# Bug #3 — @protect calls check_control_plane FIRST -# ────────────────────────────────────────────────────────────── - - -class TestProtectCallsControlPlaneFirst: - @pytest.mark.skip( - reason=( - "@protect unifies WorkflowKilledInterrupt " - "into NullRunBlockedException at the decorator boundary. This test " - "expects the original WorkflowKilledInterrupt type, which is the " - "direct-call contract preserved by check_workflow_budget(). Both " - "contracts coexist by design; the @protect boundary picks one. " - "Re-enable when the decorator gains an opt-in to preserve the " - "original exception type." - ) - ) - def test_kill_short_circuits_before_budget(self, monkeypatch): - """@protect with a Killed remote state must raise - WorkflowKilledInterrupt and NOT call check_workflow_budget. - Regression for bug #3 — previously the KILL was silently - ignored for @protect-only code paths.""" - import nullrun.decorators as dec - from nullrun.context import workflow as wf_ctx - - rt = _RecordingRuntime() - rt._remote_states["wf-killed"] = { - "state": "Killed", - "reason": "operator killed", - "version": 1, - } - dec._runtime = rt - try: - with wf_ctx("wf-killed"): - - @nullrun.protect - def agent(q): - return "should not run" - - with pytest.raises(WorkflowKilledInterrupt): - agent("hi") - - # Verify gate order — control_plane was called, budget was NOT - assert "control_plane" in rt.gate_calls - assert "budget" not in rt.gate_calls, ( - "budget was called despite KILL — bug #3 regression" - ) - finally: - dec._runtime = None - - def test_gate_order_normal_state(self, monkeypatch): - """Normal remote state — control_plane runs first, then budget. - Catches accidental reordering in the @protect wrapper.""" - import nullrun.decorators as dec - from nullrun.context import workflow as wf_ctx - - rt = _RecordingRuntime() - # Default state is Normal (empty _remote_states → state==Normal) - dec._runtime = rt - try: - with wf_ctx("wf-ok"): - - @nullrun.protect - def agent(q): - return "ok" - - result = agent("hi") - assert result == "ok" - assert rt.gate_calls == ["control_plane", "budget"] - finally: - dec._runtime = None - - @pytest.mark.skip( - reason=( - "@protect unifies WorkflowKilledInterrupt " - "into NullRunBlockedException. This test asserts span_end is emitted " - "with the original WorkflowKilledInterrupt type, but the decorator " - "now raises NullRunBlockedException. Re-enable when span_end payload " - "captures both the original and unified exception types." - ) - ) - def test_kill_does_not_skip_span_end(self, monkeypatch): - """On KILL, span_end MUST still be emitted (so the dashboard - can render the kill in context). The wrapper's try/except - around the gates guarantees this.""" - import nullrun.decorators as dec - from nullrun.context import workflow as wf_ctx - - rt = _RecordingRuntime() - rt._remote_states["wf-killed"] = { - "state": "Killed", - "reason": "killed", - "version": 1, - } - dec._runtime = rt - try: - with wf_ctx("wf-killed"): - - @nullrun.protect - def agent(q): - return "should not run" - - with pytest.raises(WorkflowKilledInterrupt): - agent("hi") - - events = rt.events - span_ends = [e for e in events if e["type"] == "span_end"] - assert len(span_ends) == 1, ( - "KILL path did not emit span_end — dashboard would lose the kill context" - ) - err = span_ends[0].get("error") or "" - assert "killed" in err.lower() - finally: - dec._runtime = None - - -# ────────────────────────────────────────────────────────────── -# Transport-layer classification regression -# ────────────────────────────────────────────────────────────── - - -class TestTransportClassification: - @pytest.mark.skip( - reason=( - "Transport.check() now requires " - 'on_transport_error="raise" to surface classified errors ' - "(preserves legacy fail-OPEN behaviour by default so " - "check_workflow_budget can treat network errors as transient). " - "Re-enable when the test passes the opt-in flag." - ) - ) - def test_check_raises_classified_error_on_network(self, mock_api): - """transport.check with on_transport_error='raise' must - surface classified NETWORK_ERROR.""" - from nullrun.transport import Transport - - respx.post(f"{BASE_URL}/api/v1/execute").mock( - side_effect=httpx.ConnectError("connection refused") - ) - rt = Transport(api_url=BASE_URL, api_key="k") - with pytest.raises(NullRunTransportError) as exc_info: - rt.check( - { - "organization_id": "o", - "execution_id": "e", - "operation_id": "op", - "check_type": "llm", - "model": "m", - "estimated_tokens": 1, - } - ) - assert exc_info.value.source == TransportErrorSource.NETWORK_ERROR - assert exc_info.value.endpoint == "check" - - def test_execute_raises_classified_error_on_5xx(self, mock_api): - """transport.execute with on_transport_error='raise' must - surface classified GATEWAY_ERROR on 5xx.""" - from nullrun.transport import Transport - - # Audit F-R2-01 (2026-06-22): Transport.execute routes to - # /api/v1/execute (not /gate) — see transport.py:1188. - respx.post(f"{BASE_URL}/api/v1/execute").mock(return_value=httpx.Response(500, text="boom")) - rt = Transport(api_url=BASE_URL, api_key="k") - with pytest.raises(NullRunTransportError) as exc_info: - rt.execute( - organization_id="o", - execution_id="e", - trace_id="t", - tool="my.tool", - input_data={}, - on_transport_error="raise", - ) - assert exc_info.value.source == TransportErrorSource.GATEWAY_ERROR - assert exc_info.value.endpoint == "execute" - - def test_execute_open_returns_fallback_allow(self, mock_api): - """transport.execute with on_transport_error='open' returns - a synthetic allow with FALLBACK_* source — used by callers - that want the dict shape (e.g. for audit, not for - enforcement).""" - from nullrun.transport import Transport - - respx.post(f"{BASE_URL}/api/v1/execute").mock( - side_effect=httpx.ConnectError("connection refused") - ) - rt = Transport(api_url=BASE_URL, api_key="k") - result = rt.execute( - organization_id="o", - execution_id="e", - trace_id="t", - tool="my.tool", - input_data={}, - on_transport_error="open", - ) - assert result["decision"] == "allow" - assert result["decision_source"] == TransportErrorSource.NETWORK_ERROR - - def test_execute_closed_returns_fallback_block(self, mock_api): - """transport.execute with on_transport_error='closed' returns - a synthetic block with FALLBACK_* source.""" - from nullrun.transport import Transport - - respx.post(f"{BASE_URL}/api/v1/execute").mock( - side_effect=httpx.ConnectError("connection refused") - ) - rt = Transport(api_url=BASE_URL, api_key="k") - result = rt.execute( - organization_id="o", - execution_id="e", - trace_id="t", - tool="my.tool", - input_data={}, - on_transport_error="closed", - ) - assert result["decision"] == "block" - assert result["decision_source"] == TransportErrorSource.NETWORK_ERROR - - -# ────────────────────────────────────────────────────────────── -# Bug #6 — NULLRUN_SKIP_BUDGET_CHECK=1 production guard -# ────────────────────────────────────────────────────────────── -# -# Pre-v3.53 the SDK silently honored ``NULLRUN_SKIP_BUDGET_CHECK=1`` -# regardless of environment. CLAUDE.md §20 marks that env var as a -# DEV/TEST bypass and explicitly forbids it in production. The fix -# in v3.53 raises NullRunInfrastructureError (NR-S001) when the -# var is set AND the SDK detects a production environment (either -# the default api.nullrun.io host OR NULLRUN_ENV=production on a -# non-dev host). The bypass is still reachable via an explicit -# ack (``NULLRUN_ALLOW_SKIP_BUDGET_CHECK=1``) for incident-response -# scenarios so the opt-out is visible in audit / telemetry. - - -class TestSkipBudgetCheckProductionGuard: - """Source-pin + runtime tests for v3.53 audit #6.""" - - # ------------------------------------------------------------------ - # _is_production_environment() helper - # ------------------------------------------------------------------ - - def test_is_production_environment_default_api_url(self): - """Default api_url (api.nullrun.io) → production.""" - from nullrun.runtime import _is_production_environment - - # Default is api.nullrun.io per constructor docstring. - assert _is_production_environment() is True - - def test_is_production_environment_with_explicit_prod_url(self): - """Explicit prod api_url → production.""" - from nullrun.runtime import _is_production_environment - - assert _is_production_environment("https://api.nullrun.io") is True - - def test_is_production_environment_localhost_is_not_prod(self, monkeypatch): - """Localhost api_url is NOT production.""" - from nullrun.runtime import _is_production_environment - - assert _is_production_environment("http://localhost:8080") is False - assert _is_production_environment("http://127.0.0.1:8080") is False - - def test_is_production_environment_staging_subdomain_is_not_prod(self, monkeypatch): - """Staging subdomain is NOT production.""" - from nullrun.runtime import _is_production_environment - - assert _is_production_environment("https://staging.nullrun.io") is False - assert _is_production_environment("https://api.staging.internal") is False - - def test_is_production_environment_explicit_env_override(self, monkeypatch): - """NULLRUN_ENV=production on a non-dev host → production.""" - from nullrun.runtime import _is_production_environment - - monkeypatch.setenv("NULLRUN_ENV", "production") - assert ( - _is_production_environment("https://custom-deployment.example.com") - is True - ) - - def test_is_production_environment_explicit_env_with_localhost(self, monkeypatch): - """NULLRUN_ENV=production BUT api_url is localhost → NOT prod - (so a dev who accidentally exports NULLRUN_ENV=production can - still use the bypass).""" - from nullrun.runtime import _is_production_environment - - monkeypatch.setenv("NULLRUN_ENV", "production") - assert _is_production_environment("http://localhost:9000") is False - - def test_is_production_environment_prod_alias(self, monkeypatch): - """NULLRUN_ENV=prod (short alias) is also detected.""" - from nullrun.runtime import _is_production_environment - - monkeypatch.setenv("NULLRUN_ENV", "prod") - assert ( - _is_production_environment("https://api.nullrun.io") is True - ) - - # ------------------------------------------------------------------ - # check_workflow_budget skip-path enforcement - # ------------------------------------------------------------------ - - def test_skip_set_in_production_raises_infrastructure_error( - self, make_runtime, monkeypatch - ): - """NULLRUN_SKIP_BUDGET_CHECK=1 in production + no ack → raise - NullRunInfrastructureError(NR-S001).""" - from nullrun.breaker.exceptions import NullRunInfrastructureError - - # Build a runtime pointing at the prod host via a custom - # respx block — ``make_runtime`` is BASE_URL-bound and we - # need to exercise the api_url=prod branch. - with respx.mock(assert_all_called=False) as mock: - mock.post("https://api.nullrun.io/api/v1/auth/verify").mock( - return_value=httpx.Response( - 200, - json={ - "organization_id": "ws-test", - "workflow_id": "00000000-0000-0000-0000-000000000001", - "plan": "pro", - "user_id": "u-test", - "session_token": "s-test", - "expires_at": "2030-01-01T00:00:00Z", - }, - ) - ) - from nullrun.runtime import NullRunRuntime - - rt = NullRunRuntime( - api_key="test-key-12345678", - api_url="https://api.nullrun.io", - polling=False, - ) - assert rt.api_url == "https://api.nullrun.io" - - monkeypatch.setenv("NULLRUN_SKIP_BUDGET_CHECK", "1") - # Ensure no ack is set (might leak from another test). - monkeypatch.delenv("NULLRUN_ALLOW_SKIP_BUDGET_CHECK", raising=False) - - with pytest.raises(NullRunInfrastructureError) as exc_info: - rt.check_workflow_budget() - - # Must carry the NR-S001 error_code so operators can pin - # this in alerting. - assert exc_info.value.error_code == "NR-S001" - assert "CLAUDE.md §20" in str(exc_info.value) - assert exc_info.value.retryable is False - - def test_skip_set_in_production_with_ack_skips_with_warning( - self, make_runtime, monkeypatch, caplog - ): - """NULLRUN_SKIP_BUDGET_CHECK=1 + NULLRUN_ALLOW_SKIP_BUDGET_CHECK=1 - in prod → skip succeeds with a WARNING log so the audit trail - captures the explicit ack.""" - import logging - - with respx.mock(assert_all_called=False) as mock: - mock.post("https://api.nullrun.io/api/v1/auth/verify").mock( - return_value=httpx.Response( - 200, - json={ - "organization_id": "ws-test", - "workflow_id": "00000000-0000-0000-0000-000000000001", - "plan": "pro", - "user_id": "u-test", - "session_token": "s-test", - "expires_at": "2030-01-01T00:00:00Z", - }, - ) - ) - from nullrun.runtime import NullRunRuntime - - rt = NullRunRuntime( - api_key="test-key-12345678", - api_url="https://api.nullrun.io", - polling=False, - ) - - monkeypatch.setenv("NULLRUN_SKIP_BUDGET_CHECK", "1") - monkeypatch.setenv("NULLRUN_ALLOW_SKIP_BUDGET_CHECK", "1") - - with caplog.at_level(logging.WARNING, logger="nullrun.runtime"): - # Must NOT raise — explicit ack honors the bypass. - rt.check_workflow_budget() - - # The ack path must surface a WARNING so observability picks - # it up. The exact text is allowed to evolve; we assert on - # the key substring so the test survives minor copy edits. - joined = "\n".join(rec.message for rec in caplog.records) - assert "NULLRUN_ALLOW_SKIP_BUDGET_CHECK" in joined, ( - "ack path did not emit a WARNING log; the explicit bypass " - "would be invisible in audit / telemetry" - ) - - def test_skip_set_in_dev_skips_silently(self, make_runtime, monkeypatch): - """NULLRUN_SKIP_BUDGET_CHECK=1 in a dev/test environment - (non-prod api_url) → skip succeeds silently (no raise, no - require_ack). Preserves the legacy dev/test behavior.""" - # Base URL is test.nullrun.io → not prod. - rt = make_runtime() - assert rt.api_url == BASE_URL - - monkeypatch.setenv("NULLRUN_SKIP_BUDGET_CHECK", "1") - monkeypatch.delenv("NULLRUN_ALLOW_SKIP_BUDGET_CHECK", raising=False) - - # Must NOT raise — the dev path stays dev-friendly. - rt.check_workflow_budget() - - def test_skip_not_set_no_prod_guard(self, monkeypatch): - """NULLRUN_SKIP_BUDGET_CHECK not set → no production guard - even on a prod api_url. The gate makes its normal HTTP call.""" - with respx.mock(assert_all_called=False) as mock: - mock.post("https://api.nullrun.io/api/v1/auth/verify").mock( - return_value=httpx.Response( - 200, - json={ - "organization_id": "ws-test", - "workflow_id": "00000000-0000-0000-0000-000000000001", - "plan": "pro", - "user_id": "u-test", - "session_token": "s-test", - "expires_at": "2030-01-01T00:00:00Z", - }, - ) - ) - mock.post("https://api.nullrun.io/api/v1/gate").mock( - return_value=httpx.Response( - 200, json={"decision": "allow", "explanations": []} - ) - ) - from nullrun.runtime import NullRunRuntime - - rt = NullRunRuntime( - api_key="test-key-12345678", - api_url="https://api.nullrun.io", - polling=False, - ) - - monkeypatch.delenv("NULLRUN_SKIP_BUDGET_CHECK", raising=False) - - # Must NOT raise — the gate makes its normal HTTP call, - # which returns 200 OK in the mock above. - rt.check_workflow_budget() - - def test_skip_prod_helper_rejects_nonsensical_env(self, monkeypatch): - """NULLRUN_ENV=staging (or other non-prod values) on a non-prod - host → NOT prod. (Cannot override a real prod api_url via - NULLRUN_ENV — the host check fires first.) - """ - from nullrun.runtime import _is_production_environment - - monkeypatch.setenv("NULLRUN_ENV", "staging") - # Non-prod host + non-prod env → not prod. - assert _is_production_environment("https://custom.example.com") is False - assert _is_production_environment("http://localhost:9000") is False - - def test_skip_prod_helper_handles_unparseable_url(self, monkeypatch): - """An unparseable api_url does NOT crash the helper.""" - from nullrun.runtime import _is_production_environment - - # Falls through to NULLRUN_ENV check; no prod env either. - result = _is_production_environment("not a real url") - assert result is False - - def test_skip_prod_helper_lowercases_hostname(self): - """API_URL with uppercase hostname is still matched as prod.""" - from nullrun.runtime import _is_production_environment - - # Mixed-case hostname should still match api.nullrun.io. - assert _is_production_environment("https://API.NULLRUN.IO") is True diff --git a/tests/test_protect.py b/tests/test_protect.py deleted file mode 100644 index 778a204..0000000 --- a/tests/test_protect.py +++ /dev/null @@ -1,971 +0,0 @@ -""" -Tests for `@protect` with automatic span hierarchy. - -The decorator must: - - Create a root span (parent_span_id=None, depth=0) on the outermost call - - Create a child span (parent_span_id=, depth+1) on nested calls - - Restore the previous context (None or parent) after the call - - Work with sync AND async functions - - Emit `span_start` and `span_end` events to the runtime -""" - -from __future__ import annotations - -import asyncio - -import pytest - -import nullrun -from nullrun.tracing import get_current_span, reset_span, set_span - -# ────────────────────────────────────────────────────────────── -# Fixtures -# ────────────────────────────────────────────────────────────── - - -@pytest.fixture -def mock_runtime(make_runtime, mock_api): - """An isolated, mocked runtime for span assertions.""" - return make_runtime() - - -class _RecordingRuntime: - """ - Drop-in stand-in for `NullRunRuntime` that records every `track_event` - call so we can assert on span_start/span_end emission without a - real backend. - - The decorator calls `check_control_plane`, `check_workflow_budget` - and `is_sensitive_tool` as pre-execution gates (ADR-008). The default - no-op implementations here keep the test isolated to the - span/track_event path; sensitive-tool gating is short-circuited - (no tool is sensitive in these tests). - """ - - def __init__(self) -> None: - self.events: list[dict] = [] - - def track_event(self, event_type: str, **kwargs) -> None: - self.events.append({"type": event_type, **kwargs}) - - def track_tool(self, tool_name: str, **kwargs) -> None: - # Commit 33d2b5f wires ``@protect`` to emit a tools/track_tool event - # after the wrapped body returns. The stub captures that emit the - # same way it captures span_start/span_end so the dashboard-level - # assertions keep working unchanged. - self.events.append({"type": "tool_call", "tool_name": tool_name, **kwargs}) - - def check_control_plane(self, workflow_id) -> None: # noqa: ARG002 - return None - - def check_workflow_budget(self) -> None: - return None - - def _bump_protect_count(self) -> None: - # 2026-09-22: zero-activity diagnostic counter. Tests - # that use this stub don't exercise the diagnostic, so a - # no-op is correct — the real impl lives on - # NullRunRuntime and is exercised by dedicated tests. - return None - - def is_sensitive_tool(self, fn_name: str) -> bool: # noqa: ARG002 - return False - - def execute(self, *args, **kwargs): # noqa: ARG002 - return None - - -@pytest.fixture -def recording_runtime(): - """Inject a _RecordingRuntime into the @protect slot.""" - import nullrun.decorators as dec - - rt = _RecordingRuntime() - dec._runtime = rt - try: - yield rt - finally: - dec._runtime = None - - -# ────────────────────────────────────────────────────────────── -# Span hierarchy -# ────────────────────────────────────────────────────────────── - - -def test_protect_creates_root_span(recording_runtime): - """Outermost @protect call: parent_span_id is None, depth is 0.""" - - @nullrun.protect - def agent(q): - return get_current_span() - - span = agent("hello") - assert span is not None - assert span.parent_span_id is None - assert span.depth == 0 - assert span.trace_id - assert span.span_id - - -def test_protect_nested_creates_child_span(recording_runtime): - """A nested @protect call is a child of the outer one (parent_span_id set - depth=1) AND shares the trace_id.""" - - @nullrun.protect - def orchestrator(q): - return researcher(q) - - @nullrun.protect - def researcher(q): - return get_current_span() - - inner = orchestrator("hello") - assert inner.parent_span_id is not None - assert inner.depth == 1 - - # Sanity: orchestrator's span is the parent. - events = recording_runtime.events - span_starts = [e for e in events if e["type"] == "span_start"] - fn_names = [e["fn_name"] for e in span_starts] - assert fn_names == ["orchestrator", "researcher"] - assert span_starts[0]["span_id"] == inner.parent_span_id - assert span_starts[0]["trace_id"] == inner.trace_id - assert span_starts[1]["parent_span_id"] == inner.parent_span_id - - -def test_protect_restores_context_after_call(recording_runtime): - """After @protect returns, get_current_span goes back to whatever - was active before — usually None at the top of the test.""" - - @nullrun.protect - def agent(q): - return get_current_span().trace_id - - assert get_current_span() is None # before - agent("hello") - assert get_current_span() is None # after — contextvar is reset - - -def test_protect_restores_outer_span_on_nested_exit(recording_runtime): - """When the inner @protect returns, the OUTER span becomes current - again — not None. This is the whole point of the token-based - set_span / reset_span pattern.""" - - @nullrun.protect - def outer(q): - # Inside outer: we are the current span. - outer_span = get_current_span() - inner("x") # this should NOT clobber outer_span - # After inner returns, outer_span should be current again. - return outer_span, get_current_span() - - @nullrun.protect - def inner(q): - return get_current_span() - - outer_span, after_inner = outer("q") - assert after_inner is outer_span # restored, not None - - -# ────────────────────────────────────────────────────────────── -# Span event emission -# ────────────────────────────────────────────────────────────── - - -def test_protect_emits_span_start_and_end(recording_runtime): - """@protect must emit a span_start before the call and span_end after.""" - - @nullrun.protect - def agent(q): - return q - - agent("hi") - events = recording_runtime.events - starts = [e for e in events if e["type"] == "span_start"] - ends = [e for e in events if e["type"] == "span_end"] - assert len(starts) == 1 - assert len(ends) == 1 - assert starts[0]["span_id"] == ends[0]["span_id"] - assert starts[0]["trace_id"] == ends[0]["trace_id"] - assert starts[0]["fn_name"] == "agent" - assert "error" not in ends[0] or ends[0]["error"] is None - - -def test_protect_emits_error_in_span_end(recording_runtime): - """If the wrapped function raises, span_end carries the error string.""" - - @nullrun.protect - def boom(q): - raise ValueError("kaboom") - - with pytest.raises(ValueError, match="kaboom"): - boom("x") - - ends = [e for e in recording_runtime.events if e["type"] == "span_end"] - assert len(ends) == 1 - assert "kaboom" in (ends[0].get("error") or "") - - -def test_protect_resets_context_even_on_error(recording_runtime): - """The contextvar is reset in `finally`, so an exception inside - @protect must not leave a stale span on the stack.""" - - @nullrun.protect - def boom(q): - raise RuntimeError("nope") - - with pytest.raises(RuntimeError): - boom("x") - assert get_current_span() is None - - -# ────────────────────────────────────────────────────────────── -# Async support -# ────────────────────────────────────────────────────────────── - - -@pytest.mark.asyncio -async def test_protect_async_creates_root_span(recording_runtime): - """Async @protect wraps the coroutine in a span, returns the result.""" - - @nullrun.protect - async def async_agent(q): - await asyncio.sleep(0) - return get_current_span() - - span = await async_agent("hi") - assert span.parent_span_id is None - assert span.depth == 0 - - -@pytest.mark.asyncio -async def test_protect_async_nested_child(recording_runtime): - """Async -> sync @protect still builds the parent/child tree.""" - - @nullrun.protect - async def outer(q): - return await inner(q) - - @nullrun.protect - async def inner(q): - return get_current_span() - - inner_span = await outer("q") - assert inner_span.depth == 1 - assert inner_span.parent_span_id is not None - - -# T3-S2 (0.3.0): `test_protect_with_noop_runtime_allows` and -# `test_protect_with_noop_runtime_async` were removed along with -# `NullRunNoop` itself. Every runtime is now a real `NullRunRuntime` -# with a bound workflow — there is no "tolerate a stub" branch to test. - - -# ────────────────────────────────────────────────────────────── -# Decorator shape (must work with @protect AND @protect ) -# ────────────────────────────────────────────────────────────── - - -def test_protect_with_empty_parens(recording_runtime): - """`@nullrun.protect()` is the same as `@nullrun.protect`.""" - - @nullrun.protect() - def agent(q): - return get_current_span() - - span = agent("x") - assert span.parent_span_id is None - - -def test_protect_preserves_function_metadata(recording_runtime): - """`@protect` must not strip __name__ / __doc__ from the wrapped fn.""" - - @nullrun.protect - def my_documented_func(): - """Important docstring.""" - return 1 - - assert my_documented_func.__name__ == "my_documented_func" - assert "Important docstring" in (my_documented_func.__doc__ or "") - - -# ────────────────────────────────────────────────────────────── -# Manually-set span is preserved (don't clobber explicit context) -# ────────────────────────────────────────────────────────────── - - -def test_protect_respects_externally_set_span(recording_runtime): - """If user code manually calls set_span(...) before @protect fires - the new span is a child of THAT, not a root.""" - from nullrun.tracing import create_root_span as make_root - - outer = make_root() - token = set_span(outer) - try: - - @nullrun.protect - def inner(q): - return get_current_span() - - span = inner("x") - assert span.parent_span_id == outer.span_id - assert span.trace_id == outer.trace_id - assert span.depth == 1 - finally: - reset_span(token) - - -# ────────────────────────────────────────────────────────────── -# Re-init wiring (regression: stale runtime in @protect cache) -# ────────────────────────────────────────────────────────────── - - -def test_init_replaces_stale_decorator_runtime_cache(mock_api): - """`nullrun.init` must update the @protect decorator's own module-level cache. - - Pre-seed `decorators._runtime` with a sentinel that raises on - `track_event`, then call `init`. If the fix is in place, init - overwrites the slot and the sentinel is never reachable. - """ - import nullrun.decorators as _dec - - class _DeadSentinel: - """A pre-seeded cache slot that raises if @protect ever uses it.""" - - def track_event(self, *args, **kwargs): # noqa: ARG002 - raise AssertionError( - "decorators._runtime was not refreshed by init(); " - "the @protect cache is still pointing at a stale runtime." - ) - - _dec._runtime = _DeadSentinel() - - rt = nullrun.init( - api_key="test-key-12345678", - api_url="https://api.test.nullrun.io", - ) - try: - # The fix: init must overwrite the decorator's cache slot. - # Without the fix, this assertion fails because the slot - # still points at _DeadSentinel. - assert _dec._runtime is rt, ( - "init() did not update decorators._runtime; " - "the @protect cache is still pointing at a stale runtime." - ) - assert not isinstance(_dec._runtime, _DeadSentinel) - finally: - _dec._runtime = None - try: - rt.shutdown() - except Exception: - pass - - -def test_protect_uses_new_runtime_after_reinit(mock_api): - """After init → shutdown → init, @protect emits span events to the NEW runtime, not the dead one.""" - import nullrun.decorators as _dec - - first_runtime = _RecordingRuntime() - - # Simulate the first init cycle: pre-seed the cache, run a @protect - # call (events go to first_runtime), then "shut down" by replacing - # the cache with a dead sentinel. - _dec._runtime = first_runtime - try: - - @nullrun.protect - def step_a(): - return "a" - - assert step_a() == "a" - finally: - first_runtime.events.clear() - - class _DeadRuntime: - def track_event(self, *args, **kwargs): # noqa: ARG002 - raise AssertionError("dead runtime called by @protect after re-init") - - _dec._runtime = _DeadRuntime() - - # Re-init must refresh the cache. After this, calling @protect - # routes to the new runtime, not _DeadRuntime. - rt = nullrun.init( - api_key="test-key-12345678", - api_url="https://api.test.nullrun.io", - ) - try: - assert _dec._runtime is rt - - @nullrun.protect - def step_b(): - return "b" - - assert step_b() == "b" - # If the regression were live, step_b would have raised inside - # _emit_span_start via the _DeadRuntime.track_event AssertionError. - finally: - _dec._runtime = None - try: - rt.shutdown() - except Exception: - pass - - -# ─── protect edge cases (re-init, fail-open, kill/pause) ───────────────────────── -""" -Additional tests for ``nullrun.decorators`` — branch coverage for the -``_safe_args`` / ``_strip_details_balanced`` / ``_enforce_sensitive_tool`` -helpers, the fail-CLOSED / fail-OPEN contract, the KILL→BlockedException -unification, and the ``@protect `` paren-form. -""" - -import os -from types import SimpleNamespace -from unittest.mock import MagicMock - -import pytest - -from nullrun.breaker.exceptions import ( - NullRunBlockedException, - NullRunTransportError, - TransportErrorSource, - WorkflowKilledInterrupt, - WorkflowPausedException, -) -from nullrun.decorators import ( - SENSITIVE_ARG_KEYS, - _enforce_sensitive_tool, - _safe_args, - _safe_error_str, - _safe_kwargs, - _safe_repr, - _strip_details_balanced, - protect, - sensitive, -) -from nullrun.runtime import NullRunRuntime - - -@pytest.fixture -def test_runtime(monkeypatch, tmp_path): - """Provide a runtime in test mode so get_runtime returns without - authenticating against a real server. - - Replays any WAL left over from previous test runs in a - tmp_path-scoped WAL file so the constructor's - ``_replay_from_wal`` never reads ``~/.nullrun/sdk.wal`` and - flushes real on-disk events to a live API. This avoids the - cross-Python-version flake seen on CI in 2026-07-11 where - 3.11 picked up a stale WAL from a 3.10/3.12 worker that - finished without explicitly clearing it. - """ - monkeypatch.setenv("NULLRUN_API_KEY", "test-key-12345678") - monkeypatch.setenv("NULLRUN_WAL_PATH", str(tmp_path / "sdk.wal")) - NullRunRuntime.reset_instance() - rt = NullRunRuntime(api_key="test-key-12345678", _test_mode=True) - rt.organization_id = "org-1" - # Stub the transport so the network is never touched in tests. - # - ``_do_flush`` overrides the public flush. - # - ``_do_flush_locked`` is what ``track `` calls when the buffer - # fills — must also be stubbed to be safe. - # - ``_client`` is the httpx client — magicmock so even a stray - # ``post`` raises a clean AttributeError instead of hitting the API. - rt._transport._do_flush = lambda: None - rt._transport._do_flush_locked = lambda: None - rt._transport._client = MagicMock() - NullRunRuntime._instance = rt - yield rt - NullRunRuntime.reset_instance() - - -# ─── _safe_repr ─────────────────────────────────────────────────────── - - -def test_safe_repr_short_value_passes_through(test_runtime): - """Under the 50-char cap, value flows through unmodified.""" - s = _safe_repr("hi") - assert s == "'hi'" - - -def test_safe_repr_long_value_truncated(test_runtime): - """Over 50 chars, suffix ``...`` appended.""" - s = _safe_repr("x" * 200, max_len=50) - assert s.endswith("...") - assert len(s) > 50 - - -def test_safe_repr_redacts_details_before_truncating(test_runtime): - """``details={PAN: '4111-...'}`` must be redacted BEFORE truncation.""" - # String kept under the 50-char cap so the redact survives the - # truncate step (otherwise we'd only verify truncation). - secret = "4111-1111-1111-1111" - payload = f"x details={{'card': '{secret}'}}" - out = _safe_repr(payload, max_len=50) - assert secret not in out - assert "" in out - - -# ─── _safe_kwargs ──────────────────────────────────────────────────── - - -def test_safe_kwargs_masks_sensitive_keys(test_runtime): - out = _safe_kwargs({"password": "p", "token": "t", "user": "alice"}) - assert out["password"] == "***" - assert out["token"] == "***" - # Non-sensitive values go through _safe_repr → ``repr ``. - assert out["user"] == "'alice'" - - -def test_safe_kwargs_is_case_insensitive(test_runtime): - out = _safe_kwargs({"PASSWORD": "p", "Token": "t"}) - assert out["PASSWORD"] == "***" - assert out["Token"] == "***" - - -# ─── _safe_args ────────────────────────────────────────────────────── - - -def test_safe_args_masks_positional_sensitive_param(test_runtime): - """Positional sensitive param (e.g. ``credit_card_number``) is masked.""" - - def charge(credit_card_number, amount): - return amount - - masked = _safe_args(charge, ("4111-1111-1111-1111", 50)) - assert masked[0] == "***" - # ``repr(50)`` is ``"50"``. - assert masked[1] == "50" - - -def test_safe_args_trailing_extra_args_uses_safe_repr(): - """``*args``-style callable: extra positional args use safe_repr.""" - - def variadic(*args, **kwargs): - return args - - masked = _safe_args(variadic, ("x", "ok")) - # ``*args`` has no name → safe_repr for both (no masking). - assert masked[0] == "'x'" - assert masked[1] == "'ok'" - - -def test_safe_args_no_signature_falls_back_to_safe_repr(): - """C-extension / built-in without signature → safe_repr on all.""" - - class _NoSig: - # Builtin-ish class; ``inspect.signature`` raises ValueError. - pass - - masked = _safe_args(_NoSig, ("4111", 50)) - assert masked[0] == "'4111'" - assert masked[1] == "50" - - -def test_safe_args_signature_raises_typeerror_falls_back(): - """``inspect.signature`` raises ``TypeError`` for some callables.""" - - class _Bad: - # Trigger ValueError path. - __signature__ = None # type: ignore[assignment] - - masked = _safe_args(_Bad, ("x",)) - assert masked == ["'x'"] - - -# ─── _strip_details_balanced ───────────────────────────────────────── - - -def test_strip_details_balanced_no_details_unchanged(): - s = "no details here" - assert _strip_details_balanced(s) == s - - -def test_strip_details_balanced_details_without_brace_unchanged(): - s = "details=plain text without braces" - # No '{' after 'details=' → left as-is. - assert _strip_details_balanced(s) == s - - -def test_strip_details_balanced_simple_payload(test_runtime): - s = "context=ok details={'a': 1, 'b': 2}" - out = _strip_details_balanced(s) - assert "" in out - assert "'a': 1" not in out - - -def test_strip_details_balanced_nested_dicts(test_runtime): - """Nested dicts in the details payload → still redacted as a unit.""" - s = "msg details={'a': {'b': {'c': 'secret'}}}" - out = _strip_details_balanced(s) - assert "secret" not in out - assert "" in out - - -def test_strip_details_balanced_string_with_braces_inside(test_runtime): - """A string value containing ``{`` / ``}`` does NOT break the brace walker.""" - s = 'msg details={"key": "value with { and } inside"}' - out = _strip_details_balanced(s) - assert "value with { and } inside" not in out - assert "" in out - - -def test_strip_details_balanced_multiple_details(test_runtime): - """Two ``details={...}`` substrings in the same string → both redacted.""" - s = "first details={'a': 1} middle details={'b': 2}" - out = _strip_details_balanced(s) - assert out.count("") == 2 - - -def test_strip_details_balanced_escaped_quote_in_string(test_runtime): - r"""A string with an escaped quote (\") is handled by the walker.""" - s = r'msg details={"key": "val\"ue"}' - out = _strip_details_balanced(s) - assert "" in out - - -# ─── _safe_error_str ───────────────────────────────────────────────── - - -def test_safe_error_str_none_returns_none(test_runtime): - assert _safe_error_str(None) is None - - -def test_safe_error_str_simple_message_passes_through(test_runtime): - e = RuntimeError("plain") - assert _safe_error_str(e) == "plain" - - -def test_safe_error_str_details_redacted(test_runtime): - e = RuntimeError("oops details={'secret': 'value'}") - out = _safe_error_str(e) - assert "secret" not in out - assert "" in out - - -# ─── _enforce_sensitive_tool ──────────────────────────────────────── - - -def test_enforce_sensitive_tool_non_sensitive_returns(test_runtime): - """Non-sensitive tool → no-op, no runtime call.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = False - rt.execute = MagicMock() - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - rt.execute.assert_not_called() - - -def test_enforce_sensitive_tool_real_block_propagates(test_runtime): - """``decision=block`` from gateway → raises NullRunBlockedException.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.side_effect = NullRunBlockedException(workflow_id="wf-1", reason="denied") - with pytest.raises(NullRunBlockedException): - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - - -def test_enforce_sensitive_tool_transport_error_fail_closed(test_runtime): - """``NullRunTransportError`` + no fail-open → raises NullRunBlockedException.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.side_effect = NullRunTransportError( - "down", - source=TransportErrorSource.NETWORK_ERROR, - endpoint="/execute", - ) - with pytest.raises(NullRunBlockedException) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert "NETWORK_ERROR" in excinfo.value.reason - - -def test_enforce_sensitive_tool_transport_error_fail_open(test_runtime, monkeypatch): - """``NULLRUN_SENSITIVE_FAIL_OPEN=1`` + transport error → body runs.""" - monkeypatch.setenv("NULLRUN_SENSITIVE_FAIL_OPEN", "1") - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.side_effect = NullRunTransportError( - "down", - source=TransportErrorSource.NETWORK_ERROR, - endpoint="/execute", - ) - # Must NOT raise. - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - - -def test_enforce_sensitive_tool_generic_exception_fail_closed(test_runtime): - """Non-transport exception → NullRunBlockedException.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.side_effect = ValueError("oops") - with pytest.raises(NullRunBlockedException): - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - - -def test_enforce_sensitive_tool_generic_exception_fail_open(test_runtime, monkeypatch): - """Generic exception + fail-open → no raise.""" - monkeypatch.setenv("NULLRUN_SENSITIVE_FAIL_OPEN", "1") - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.side_effect = ValueError("oops") - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) # no raise - - -def test_enforce_sensitive_tool_dict_with_fallback_decision_source(test_runtime): - """``decision_source`` starts with FALLBACK_ → raises.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.return_value = { - "decision": "allow", - "decision_source": "FALLBACK_NETWORK_ERROR", - } - with pytest.raises(NullRunBlockedException): - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - - -def test_enforce_sensitive_tool_dict_with_typed_error_source(test_runtime): - """``decision_source`` ∈ TransportErrorSource values → raises.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.return_value = { - "decision": "allow", - "decision_source": TransportErrorSource.GATEWAY_ERROR, - } - with pytest.raises(NullRunBlockedException): - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - - -def test_enforce_sensitive_tool_dict_with_fallback_fail_open(test_runtime, monkeypatch): - """``decision_source`` FALLBACK_* + fail-open → no raise.""" - monkeypatch.setenv("NULLRUN_SENSITIVE_FAIL_OPEN", "1") - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.return_value = { - "decision": "allow", - "decision_source": "FALLBACK_NETWORK_ERROR", - } - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) # no raise - - -def test_enforce_sensitive_tool_dict_with_gateway_decision_falls_through(test_runtime): - """``decision_source=gateway`` + ``decision=allow`` → no raise.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.return_value = { - "decision": "allow", - "decision_source": "gateway", - } - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) # no raise - - -def test_enforce_sensitive_tool_sensitive_kwargs_masked_in_call(test_runtime): - """``password`` kwarg on a sensitive tool is masked before /execute.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.return_value = {"decision": "allow", "decision_source": "gateway"} - _enforce_sensitive_tool(rt, lambda x: x, (), {"password": "p", "user": "alice"}) - # ``runtime.execute`` is called positionally: ``(tool_name, input_data,...)``. - forwarded = rt.execute.call_args.args[1] - assert forwarded["kwargs"]["password"] == "***" - # Non-sensitive → safe_repr → ``"'alice'"``. - assert forwarded["kwargs"]["user"] == "'alice'" - - -def test_enforce_sensitive_tool_sensitive_positional_arg_masked(test_runtime): - """``credit_card_number`` positional on a sensitive tool is masked.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.return_value = {"decision": "allow", "decision_source": "gateway"} - - def charge(credit_card_number, amount): - return amount - - _enforce_sensitive_tool(rt, charge, ("4111-1111-1111-1111", 50), {}) - forwarded = rt.execute.call_args.args[1] - assert forwarded["args"][0] == "***" - - -# ─── @protect paren-form ───────────────────────────────────────────── - - -def test_protect_with_parens_returns_decorator(test_runtime): - """``@protect()`` with empty parens works just like ``@protect``.""" - # Stub track_event so the finally-block span emission does not - # re-enter check_control_plane with our mocked side effect. - test_runtime.track_event = MagicMock() - - @protect() - def f(x): - return x * 2 - - assert f(3) == 6 - - -def test_protect_without_parens_wraps_directly(test_runtime): - """``@protect`` without parens wraps the function directly.""" - # Stub track_event so the finally-block span emission does not - # re-enter check_control_plane with our mocked side effect. - test_runtime.track_event = MagicMock() - - @protect - def f(x): - return x * 2 - - assert f(3) == 6 - - -# ─── KILL→BlockedException unification ────────────────────── - - -def test_protect_sync_kill_raises_NullRunBlockedException(test_runtime): - """``WorkflowKilledInterrupt`` from gate → unified as NullRunBlockedException.""" - from nullrun import decorators as dec_mod - - rt = NullRunRuntime(api_key="test-key-12345678", _test_mode=True) - rt.track_event = MagicMock() - rt.check_control_plane = MagicMock( - side_effect=WorkflowKilledInterrupt(workflow_id="wf-1", reason="admin kill") - ) - rt.check_workflow_budget = MagicMock() - dec_mod._runtime = rt - - @protect - def f(): - return "should not run" - - with pytest.raises(NullRunBlockedException) as excinfo: - f() - assert excinfo.value.reason == "admin kill" - - -def test_protect_sync_pause_raises_NullRunBlockedException(test_runtime): - """``WorkflowPausedException`` from gate → unified as NullRunBlockedException.""" - from nullrun import decorators as dec_mod - - rt = NullRunRuntime(api_key="test-key-12345678", _test_mode=True) - rt.track_event = MagicMock() - rt.check_control_plane = MagicMock( - side_effect=WorkflowPausedException(workflow_id="wf-1", reason="budget pause") - ) - rt.check_workflow_budget = MagicMock() - dec_mod._runtime = rt - - @protect - def f(): - return "should not run" - - with pytest.raises(NullRunBlockedException) as excinfo: - f() - assert excinfo.value.reason == "budget pause" - - -@pytest.mark.asyncio -async def test_protect_async_kill_re_raises_WorkflowKilledInterrupt(make_test_runtime): - """Async wrapper does NOT unify — kill signal propagates as-is so async frameworks can interrupt cleanly.""" - from nullrun import decorators as dec_mod - - rt = make_test_runtime() - rt.track_event = MagicMock() - rt.check_control_plane = MagicMock( - side_effect=WorkflowKilledInterrupt(workflow_id="wf-1", reason="x") - ) - rt.check_workflow_budget = MagicMock() - dec_mod._runtime = rt - - @protect - async def f(): - return "ok" - - with pytest.raises(WorkflowKilledInterrupt): - await f() - - -# ─── @sensitive decorator ──────────────────────────────────────────── - - -def test_sensitive_registers_tool_with_runtime(test_runtime): - """``@sensitive`` calls ``add_sensitive_tool`` on the runtime.""" - - @sensitive - def my_charge(amount): - return amount - - rt = NullRunRuntime.get_instance() - assert "my_charge" in rt.get_sensitive_tools() - - -def test_sensitive_runtime_init_failure_raises(test_runtime, monkeypatch): - """If runtime construction fails inside @sensitive, raises RuntimeError (fail-CLOSED, ADR-008).""" - from nullrun import decorators - - original_exc = RuntimeError("x") - monkeypatch.setattr( - decorators, - "_get_or_create_runtime", - MagicMock(side_effect=original_exc), - ) - - with pytest.raises( - RuntimeError, - match=r"@sensitive registration failed for 'f'", - ) as excinfo: - - @sensitive - def f(): - return 1 - - assert excinfo.value.__cause__ is original_exc - - -# ─── reset ────────────────────────────────────────────────────────── - - -def test_reset_clears_runtime_slot(test_runtime, monkeypatch): - """``reset()`` shuts down the runtime and clears the module-level slot.""" - from nullrun import decorators - - rt = NullRunRuntime.get_instance() - decorators._runtime = rt - decorators.reset() - assert decorators._runtime is None - - -def test_reset_when_no_runtime_is_silent(test_runtime): - from nullrun import decorators - - decorators._runtime = None - decorators.reset() # must not raise - - -def test_reset_shutdown_failure_is_silent(test_runtime, monkeypatch): - """``reset()`` swallows runtime shutdown exceptions.""" - from nullrun import decorators - - rt = MagicMock() - rt.shutdown.side_effect = RuntimeError("oops") - decorators._runtime = rt - decorators.reset() # must not raise - assert decorators._runtime is None - - -# ─── get_protected_runtime ────────────────────────────────────────── - - -def test_get_protected_runtime_returns_runtime(test_runtime): - from nullrun import decorators - - rt = NullRunRuntime.get_instance() - decorators._runtime = rt - assert decorators.get_protected_runtime() is rt - - -def test_get_protected_runtime_falls_back_to_get_runtime(monkeypatch, make_test_runtime): - """When the decorator slot is empty, fall back to the global singleton.""" - from nullrun import decorators - - decorators._runtime = None - NullRunRuntime._instance = make_test_runtime() - try: - out = decorators.get_protected_runtime() - assert out is NullRunRuntime._instance - finally: - NullRunRuntime.reset_instance() diff --git a/tests/test_protect_branches.py b/tests/test_protect_branches.py deleted file mode 100644 index 5cc0962..0000000 --- a/tests/test_protect_branches.py +++ /dev/null @@ -1,564 +0,0 @@ -""" -Additional tests for ``nullrun.decorators`` — branch coverage for the -``_safe_args`` / ``_strip_details_balanced`` / ``_enforce_sensitive_tool`` -helpers, the fail-CLOSED / fail-OPEN contract, the KILL→BlockedException -unification, and the ``@protect `` paren-form. -""" - -from __future__ import annotations - -import os -from types import SimpleNamespace -from unittest.mock import MagicMock - -import pytest - -from nullrun.breaker.exceptions import ( - NullRunBlockedException, - NullRunTransportError, - TransportErrorSource, - WorkflowKilledInterrupt, - WorkflowPausedException, -) -from nullrun.decorators import ( - SENSITIVE_ARG_KEYS, - _enforce_sensitive_tool, - _safe_args, - _safe_error_str, - _safe_kwargs, - _safe_repr, - _strip_details_balanced, - protect, - sensitive, -) -from nullrun.runtime import NullRunRuntime - - -@pytest.fixture -def test_runtime(monkeypatch, tmp_path): - """Provide a runtime in test mode so get_runtime returns without - authenticating against a real server. - - Replays any WAL left over from previous test runs in a - tmp_path-scoped WAL file so the constructor's - ``_replay_from_wal`` never reads ``~/.nullrun/sdk.wal`` and - flushes real on-disk events to a live API. This avoids the - cross-Python-version flake seen on CI in 2026-07-11 where - 3.11 picked up a stale WAL from a 3.10/3.12 worker that - finished without explicitly clearing it. - """ - monkeypatch.setenv("NULLRUN_API_KEY", "test-key-12345678") - monkeypatch.setenv("NULLRUN_WAL_PATH", str(tmp_path / "sdk.wal")) - NullRunRuntime.reset_instance() - rt = NullRunRuntime(api_key="test-key-12345678", _test_mode=True) - rt.organization_id = "org-1" - # Stub the transport so the network is never touched in tests. - # - ``_do_flush`` overrides the public flush. - # - ``_do_flush_locked`` is what ``track `` calls when the buffer - # fills — must also be stubbed to be safe. - # - ``_client`` is the httpx client — magicmock so even a stray - # ``post`` raises a clean AttributeError instead of hitting the API. - rt._transport._do_flush = lambda: None - rt._transport._do_flush_locked = lambda: None - rt._transport._client = MagicMock() - NullRunRuntime._instance = rt - yield rt - NullRunRuntime.reset_instance() - - -# ─── _safe_repr ─────────────────────────────────────────────────────── - - -def test_safe_repr_short_value_passes_through(test_runtime): - """Under the 50-char cap, value flows through unmodified.""" - s = _safe_repr("hi") - assert s == "'hi'" - - -def test_safe_repr_long_value_truncated(test_runtime): - """Over 50 chars, suffix ``...`` appended.""" - s = _safe_repr("x" * 200, max_len=50) - assert s.endswith("...") - assert len(s) > 50 - - -def test_safe_repr_redacts_details_before_truncating(test_runtime): - """``details={PAN: '4111-...'}`` must be redacted BEFORE truncation.""" - # String kept under the 50-char cap so the redact survives the - # truncate step (otherwise we'd only verify truncation). - secret = "4111-1111-1111-1111" - payload = f"x details={{'card': '{secret}'}}" - out = _safe_repr(payload, max_len=50) - assert secret not in out - assert "" in out - - -# ─── _safe_kwargs ──────────────────────────────────────────────────── - - -def test_safe_kwargs_masks_sensitive_keys(test_runtime): - out = _safe_kwargs({"password": "p", "token": "t", "user": "alice"}) - assert out["password"] == "***" - assert out["token"] == "***" - # Non-sensitive values go through _safe_repr → ``repr ``. - assert out["user"] == "'alice'" - - -def test_safe_kwargs_is_case_insensitive(test_runtime): - out = _safe_kwargs({"PASSWORD": "p", "Token": "t"}) - assert out["PASSWORD"] == "***" - assert out["Token"] == "***" - - -# ─── _safe_args ────────────────────────────────────────────────────── - - -def test_safe_args_masks_positional_sensitive_param(test_runtime): - """Positional sensitive param (e.g. ``credit_card_number``) is masked.""" - - def charge(credit_card_number, amount): - return amount - - masked = _safe_args(charge, ("4111-1111-1111-1111", 50)) - assert masked[0] == "***" - # ``repr(50)`` is ``"50"``. - assert masked[1] == "50" - - -def test_safe_args_trailing_extra_args_uses_safe_repr(): - """``*args``-style callable: extra positional args use safe_repr.""" - - def variadic(*args, **kwargs): - return args - - masked = _safe_args(variadic, ("x", "ok")) - # ``*args`` has no name → safe_repr for both (no masking). - assert masked[0] == "'x'" - assert masked[1] == "'ok'" - - -def test_safe_args_no_signature_falls_back_to_safe_repr(): - """C-extension / built-in without signature → safe_repr on all.""" - - class _NoSig: - # Builtin-ish class; ``inspect.signature`` raises ValueError. - pass - - masked = _safe_args(_NoSig, ("4111", 50)) - assert masked[0] == "'4111'" - assert masked[1] == "50" - - -def test_safe_args_signature_raises_typeerror_falls_back(): - """``inspect.signature`` raises ``TypeError`` for some callables.""" - - class _Bad: - # Trigger ValueError path. - __signature__ = None # type: ignore[assignment] - - masked = _safe_args(_Bad, ("x",)) - assert masked == ["'x'"] - - -# ─── _strip_details_balanced ───────────────────────────────────────── - - -def test_strip_details_balanced_no_details_unchanged(): - s = "no details here" - assert _strip_details_balanced(s) == s - - -def test_strip_details_balanced_details_without_brace_unchanged(): - s = "details=plain text without braces" - # No '{' after 'details=' → left as-is. - assert _strip_details_balanced(s) == s - - -def test_strip_details_balanced_simple_payload(test_runtime): - s = "context=ok details={'a': 1, 'b': 2}" - out = _strip_details_balanced(s) - assert "" in out - assert "'a': 1" not in out - - -def test_strip_details_balanced_nested_dicts(test_runtime): - """Nested dicts in the details payload → still redacted as a unit.""" - s = "msg details={'a': {'b': {'c': 'secret'}}}" - out = _strip_details_balanced(s) - assert "secret" not in out - assert "" in out - - -def test_strip_details_balanced_string_with_braces_inside(test_runtime): - """A string value containing ``{`` / ``}`` does NOT break the brace walker.""" - s = 'msg details={"key": "value with { and } inside"}' - out = _strip_details_balanced(s) - assert "value with { and } inside" not in out - assert "" in out - - -def test_strip_details_balanced_multiple_details(test_runtime): - """Two ``details={...}`` substrings in the same string → both redacted.""" - s = "first details={'a': 1} middle details={'b': 2}" - out = _strip_details_balanced(s) - assert out.count("") == 2 - - -def test_strip_details_balanced_escaped_quote_in_string(test_runtime): - r"""A string with an escaped quote (\") is handled by the walker.""" - s = r'msg details={"key": "val\"ue"}' - out = _strip_details_balanced(s) - assert "" in out - - -# ─── _safe_error_str ───────────────────────────────────────────────── - - -def test_safe_error_str_none_returns_none(test_runtime): - assert _safe_error_str(None) is None - - -def test_safe_error_str_simple_message_passes_through(test_runtime): - e = RuntimeError("plain") - assert _safe_error_str(e) == "plain" - - -def test_safe_error_str_details_redacted(test_runtime): - e = RuntimeError("oops details={'secret': 'value'}") - out = _safe_error_str(e) - assert "secret" not in out - assert "" in out - - -# ─── _enforce_sensitive_tool ──────────────────────────────────────── - - -def test_enforce_sensitive_tool_non_sensitive_returns(test_runtime): - """Non-sensitive tool → no-op, no runtime call.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = False - rt.execute = MagicMock() - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - rt.execute.assert_not_called() - - -def test_enforce_sensitive_tool_real_block_propagates(test_runtime): - """``decision=block`` from gateway → raises NullRunBlockedException.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.side_effect = NullRunBlockedException(workflow_id="wf-1", reason="denied") - with pytest.raises(NullRunBlockedException): - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - - -def test_enforce_sensitive_tool_transport_error_fail_closed(test_runtime): - """``NullRunTransportError`` + no fail-open → raises NullRunBlockedException.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.side_effect = NullRunTransportError( - "down", - source=TransportErrorSource.NETWORK_ERROR, - endpoint="/execute", - ) - with pytest.raises(NullRunBlockedException) as excinfo: - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - assert "NETWORK_ERROR" in excinfo.value.reason - - -def test_enforce_sensitive_tool_transport_error_fail_open(test_runtime, monkeypatch): - """``NULLRUN_SENSITIVE_FAIL_OPEN=1`` + transport error → body runs.""" - monkeypatch.setenv("NULLRUN_SENSITIVE_FAIL_OPEN", "1") - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.side_effect = NullRunTransportError( - "down", - source=TransportErrorSource.NETWORK_ERROR, - endpoint="/execute", - ) - # Must NOT raise. - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - - -def test_enforce_sensitive_tool_generic_exception_fail_closed(test_runtime): - """Non-transport exception → NullRunBlockedException.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.side_effect = ValueError("oops") - with pytest.raises(NullRunBlockedException): - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - - -def test_enforce_sensitive_tool_generic_exception_fail_open(test_runtime, monkeypatch): - """Generic exception + fail-open → no raise.""" - monkeypatch.setenv("NULLRUN_SENSITIVE_FAIL_OPEN", "1") - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.side_effect = ValueError("oops") - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) # no raise - - -def test_enforce_sensitive_tool_dict_with_fallback_decision_source(test_runtime): - """``decision_source`` starts with FALLBACK_ → raises.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.return_value = { - "decision": "allow", - "decision_source": "FALLBACK_NETWORK_ERROR", - } - with pytest.raises(NullRunBlockedException): - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - - -def test_enforce_sensitive_tool_dict_with_typed_error_source(test_runtime): - """``decision_source`` ∈ TransportErrorSource values → raises.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.return_value = { - "decision": "allow", - "decision_source": TransportErrorSource.GATEWAY_ERROR, - } - with pytest.raises(NullRunBlockedException): - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) - - -def test_enforce_sensitive_tool_dict_with_fallback_fail_open(test_runtime, monkeypatch): - """``decision_source`` FALLBACK_* + fail-open → no raise.""" - monkeypatch.setenv("NULLRUN_SENSITIVE_FAIL_OPEN", "1") - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.return_value = { - "decision": "allow", - "decision_source": "FALLBACK_NETWORK_ERROR", - } - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) # no raise - - -def test_enforce_sensitive_tool_dict_with_gateway_decision_falls_through(test_runtime): - """``decision_source=gateway`` + ``decision=allow`` → no raise.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.return_value = { - "decision": "allow", - "decision_source": "gateway", - } - _enforce_sensitive_tool(rt, lambda x: x, (1,), {}) # no raise - - -def test_enforce_sensitive_tool_sensitive_kwargs_masked_in_call(test_runtime): - """``password`` kwarg on a sensitive tool is masked before /execute.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.return_value = {"decision": "allow", "decision_source": "gateway"} - _enforce_sensitive_tool(rt, lambda x: x, (), {"password": "p", "user": "alice"}) - # ``runtime.execute`` is called positionally: ``(tool_name, input_data,...)``. - forwarded = rt.execute.call_args.args[1] - assert forwarded["kwargs"]["password"] == "***" - # Non-sensitive → safe_repr → ``"'alice'"``. - assert forwarded["kwargs"]["user"] == "'alice'" - - -def test_enforce_sensitive_tool_sensitive_positional_arg_masked(test_runtime): - """``credit_card_number`` positional on a sensitive tool is masked.""" - rt = MagicMock() - rt.is_sensitive_tool.return_value = True - rt.execute.return_value = {"decision": "allow", "decision_source": "gateway"} - - def charge(credit_card_number, amount): - return amount - - _enforce_sensitive_tool(rt, charge, ("4111-1111-1111-1111", 50), {}) - forwarded = rt.execute.call_args.args[1] - assert forwarded["args"][0] == "***" - - -# ─── @protect paren-form ───────────────────────────────────────────── - - -def test_protect_with_parens_returns_decorator(test_runtime): - """``@protect()`` with empty parens works just like ``@protect``.""" - # Stub track_event so the finally-block span emission does not - # re-enter check_control_plane with our mocked side effect. - test_runtime.track_event = MagicMock() - - @protect() - def f(x): - return x * 2 - - assert f(3) == 6 - - -def test_protect_without_parens_wraps_directly(test_runtime): - """``@protect`` without parens wraps the function directly.""" - # Stub track_event so the finally-block span emission does not - # re-enter check_control_plane with our mocked side effect. - test_runtime.track_event = MagicMock() - - @protect - def f(x): - return x * 2 - - assert f(3) == 6 - - -# ─── KILL→BlockedException unification ────────────────────── - - -def test_protect_sync_kill_raises_NullRunBlockedException(test_runtime): - """``WorkflowKilledInterrupt`` from gate → unified as NullRunBlockedException.""" - from nullrun import decorators as dec_mod - - rt = NullRunRuntime(api_key="test-key-12345678", _test_mode=True) - rt.track_event = MagicMock() - rt.check_control_plane = MagicMock( - side_effect=WorkflowKilledInterrupt(workflow_id="wf-1", reason="admin kill") - ) - rt.check_workflow_budget = MagicMock() - dec_mod._runtime = rt - - @protect - def f(): - return "should not run" - - with pytest.raises(NullRunBlockedException) as excinfo: - f() - assert excinfo.value.reason == "admin kill" - - -def test_protect_sync_pause_raises_NullRunBlockedException(test_runtime): - """``WorkflowPausedException`` from gate → unified as NullRunBlockedException.""" - from nullrun import decorators as dec_mod - - rt = NullRunRuntime(api_key="test-key-12345678", _test_mode=True) - rt.track_event = MagicMock() - rt.check_control_plane = MagicMock( - side_effect=WorkflowPausedException(workflow_id="wf-1", reason="budget pause") - ) - rt.check_workflow_budget = MagicMock() - dec_mod._runtime = rt - - @protect - def f(): - return "should not run" - - with pytest.raises(NullRunBlockedException) as excinfo: - f() - assert excinfo.value.reason == "budget pause" - - -@pytest.mark.asyncio -async def test_protect_async_kill_re_raises_WorkflowKilledInterrupt(make_test_runtime): - """Async wrapper does NOT unify — kill signal propagates as-is so - async frameworks can interrupt the event loop cleanly. - """ - from nullrun import decorators as dec_mod - - rt = make_test_runtime() - rt.track_event = MagicMock() - rt.check_control_plane = MagicMock( - side_effect=WorkflowKilledInterrupt(workflow_id="wf-1", reason="x") - ) - rt.check_workflow_budget = MagicMock() - dec_mod._runtime = rt - - @protect - async def f(): - return "ok" - - with pytest.raises(WorkflowKilledInterrupt): - await f() - - -# ─── @sensitive decorator ──────────────────────────────────────────── - - -def test_sensitive_registers_tool_with_runtime(test_runtime): - """``@sensitive`` calls ``add_sensitive_tool`` on the runtime.""" - - @sensitive - def my_charge(amount): - return amount - - rt = NullRunRuntime.get_instance() - assert "my_charge" in rt.get_sensitive_tools() - - -def test_sensitive_runtime_init_failure_raises(test_runtime, monkeypatch): - """If runtime construction fails inside @sensitive, the decorator - MUST raise ``RuntimeError`` (fail-CLOSED, ADR-008). The original - exception is chained via ``__cause__`` so callers can still inspect - the root cause. - """ - from nullrun import decorators - - original_exc = RuntimeError("x") - monkeypatch.setattr( - decorators, - "_get_or_create_runtime", - MagicMock(side_effect=original_exc), - ) - - with pytest.raises( - RuntimeError, - match=r"@sensitive registration failed for 'f'", - ) as excinfo: - - @sensitive - def f(): - return 1 - - assert excinfo.value.__cause__ is original_exc - - -# ─── reset ────────────────────────────────────────────────────────── - - -def test_reset_clears_runtime_slot(test_runtime, monkeypatch): - """``reset()`` shuts down the runtime and clears the module-level slot.""" - from nullrun import decorators - - rt = NullRunRuntime.get_instance() - decorators._runtime = rt - decorators.reset() - assert decorators._runtime is None - - -def test_reset_when_no_runtime_is_silent(test_runtime): - from nullrun import decorators - - decorators._runtime = None - decorators.reset() # must not raise - - -def test_reset_shutdown_failure_is_silent(test_runtime, monkeypatch): - """``reset()`` swallows runtime shutdown exceptions.""" - from nullrun import decorators - - rt = MagicMock() - rt.shutdown.side_effect = RuntimeError("oops") - decorators._runtime = rt - decorators.reset() # must not raise - assert decorators._runtime is None - - -# ─── get_protected_runtime ────────────────────────────────────────── - - -def test_get_protected_runtime_returns_runtime(test_runtime): - from nullrun import decorators - - rt = NullRunRuntime.get_instance() - decorators._runtime = rt - assert decorators.get_protected_runtime() is rt - - -def test_get_protected_runtime_falls_back_to_get_runtime(monkeypatch, make_test_runtime): - """When the decorator slot is empty, fall back to the global singleton.""" - from nullrun import decorators - - decorators._runtime = None - NullRunRuntime._instance = make_test_runtime() - try: - out = decorators.get_protected_runtime() - assert out is NullRunRuntime._instance - finally: - NullRunRuntime.reset_instance() diff --git a/tests/test_protect_cancel_on_exception.py b/tests/test_protect_cancel_on_exception.py deleted file mode 100644 index 0c12bb7..0000000 --- a/tests/test_protect_cancel_on_exception.py +++ /dev/null @@ -1,400 +0,0 @@ -""" -Regression tests for the SDK exception-path cancel fix. - -Background — what the fix does: - - Before this fix, `@protect` let ANY exception (control_plane kill, - workflow_budget block, sensitive-tool reject, fn() raise) propagate - out of the with-block without closing the budget reservation that - `check_workflow_budget` opened via /gate. Each exception leaked a - Redis envelope to TTL expiry (an "orphan" from the server's - perspective). At anti-DoS scale this caused reserved_total to drift - up monotonically. - - The fix wraps the with-block in a try/except in BOTH async and sync - wrappers; on failure, cancel_execution(execution_id) runs to close - the reservation. A `fn_completed` sentinel prevents cancel from - firing after track_tool failure — that path means side effects - already happened and only retry/consume semantics apply. - -Critical asymmetry (the one this file exists to lock in): - - - `async_wrapper` catches `Exception`, NOT `BaseException`. This is - so `asyncio.CancelledError` / `KeyboardInterrupt` / `SystemExit` - propagate without doing a synchronous blocking HTTP call in a - cancellation handler — that would delay shutdown by up to 5s and - trigger "Task was destroyed but pending" warnings. Test #1 is - THE regression guard for this. - - - `sync_wrapper` catches `BaseException`. Sync code has no event - loop to delay; matching existing `_protect_body` unify_block - semantics keeps behavior consistent. - -These tests pin both halves. Any future refactor that reverts the -async wrapper to `except BaseException` — e.g., "be safe, catch -everything" — would silently slow down agent cancellation and break -test #1. -""" - -from __future__ import annotations - -import asyncio -from typing import Any - -import pytest - -import nullrun.decorators as _dec -from nullrun.breaker.exceptions import ( - NullRunTransportError, - TransportErrorSource, - WorkflowKilledInterrupt, -) -from nullrun.context import ( - _server_minted_execution_id_var, - set_server_minted_execution_id, -) -from nullrun.decorators import protect - -# ───────────────────────────────────────────────────────────────────── -# RecordingRuntime — no network, no real transport. Records the order -# of gate calls and the cancel_execution invocations so the tests -# can assert exactly what happened on each failure-path. -# -# Pattern is borrowed from tests/test_preflight_fail_policy.py -# (_RecordingRuntime, lines 48-117). We extend it with: -# - stub `cancel_execution` that captures calls in `cancel_calls` -# - `capture_execution_id` config: when set and gates pass, simulates -# check_workflow_budget's real behavior of calling -# `set_server_minted_execution_id(...)` so the cancel helper -# sees a populated ContextVar. -# ───────────────────────────────────────────────────────────────────── - - -class _RecordingRuntime: - def __init__( - self, - *, - control_plane_raises: BaseException | None = None, - workflow_budget_raises: BaseException | None = None, - sensitive_tool_raises: BaseException | None = None, - track_tool_raises: BaseException | None = None, - capture_execution_id: str | None = "exec-test-123", - ) -> None: - self.gate_calls: list[str] = [] - self.cancel_calls: list[tuple[str, str | None]] = [] - self._remote_states: dict = {} - self._sensitive_tools: set = set() - self._control_plane_raises = control_plane_raises - self._workflow_budget_raises = workflow_budget_raises - self._sensitive_tool_raises = sensitive_tool_raises - self._track_tool_raises = track_tool_raises - self._capture_execution_id = capture_execution_id - - # ---- gate stubs (the four pre-execution checks) ---- - - def check_control_plane(self, workflow_id: Any) -> None: - """Mirror the real signature; raise `control_plane_raises` if set.""" - self.gate_calls.append("control_plane") - if self._control_plane_raises is not None: - raise self._control_plane_raises - - def check_workflow_budget(self) -> None: - """Simulate the real /gate success path: capture the - server-minted execution_id via the actual - `set_server_minted_execution_id(...)` primitive so the cancel - helper sees a populated ContextVar after this returns.""" - self.gate_calls.append("budget") - if self._capture_execution_id is not None: - set_server_minted_execution_id(self._capture_execution_id) - if self._workflow_budget_raises is not None: - raise self._workflow_budget_raises - - def _bump_protect_count(self) -> None: - # 2026-09-22: zero-activity diagnostic counter on the real - # NullRunRuntime. The cancel-on-exception tests do not - # exercise the diagnostic, so a no-op keeps the @protect - # call path runnable without pulling the diagnostic state - # into the cancel/capture assertions. - return None - - def is_sensitive_tool(self, tool_name: str) -> bool: - # No sensitive tools by default in these tests; sensitive-tool - # reject coverage is a separate concern (already exercised in - # test_preflight_fail_policy.py). - return False - - def _enforce_sensitive_tool_in_decorators(self, runtime, fn, args, kwargs): - # Stub doesn't replicate the decorator's _enforce_sensitive_tool - # body; we'll come back to sensitive-tool reject in a future - # test if the orphan source data shows it matters. - return None - - # ---- track / cancel stubs ---- - - def track_tool(self, tool_name: str, metadata=None, **kwargs): - self.gate_calls.append("track_tool") - if self._track_tool_raises is not None: - raise self._track_tool_raises - return {"ok": True} - - def cancel_execution(self, execution_id: str, reason: str | None = None) -> dict: - self.cancel_calls.append((execution_id, reason)) - return {"status": "cancelled"} - - -# ───────────────────────────────────────────────────────────────────── -# Fixtures — clean ContextVar between tests, build & pin a runtime. -# ───────────────────────────────────────────────────────────────────── - - -@pytest.fixture(autouse=True) -def _reset_execution_id(): - """server_minted_execution_id is a ContextVar — restore the prior - value after each test. Without this, capture from one test leaks - into the next and the order-dependent assertions become flaky.""" - prior = _server_minted_execution_id_var.get() - set_server_minted_execution_id(None) - yield - set_server_minted_execution_id(prior) - - -@pytest.fixture -def runtime_factory(): - """Build a _RecordingRuntime and pin it to the @protect decorator's - module-level slot (`_dec._runtime`) the same way - `tests/conftest.py::make_runtime` does for real NullRunRuntimes.""" - created: list[_RecordingRuntime] = [] - - def _make(**kwargs) -> _RecordingRuntime: - rt = _RecordingRuntime(**kwargs) - created.append(rt) - _dec._runtime = rt - return rt - - yield _make - _dec._runtime = None # cleanup; tests should pin fresh each time - - -# ───────────────────────────────────────────────────────────────────── -# PRIORITY 1 — the two regression guards the spec explicitly named. -# These MUST run first when iterating on the fix; failure here means -# we silently lost the invariant. -# ───────────────────────────────────────────────────────────────────── - - -@pytest.mark.asyncio -async def test_1_async_cancelled_error_does_not_trigger_cancel(runtime_factory): - """The single most important regression test. - - If a future refactor reverts `except Exception` to `except - BaseException` in `async_wrapper`, this test fails. The cost of - that regression is silent: cancel_execution is a synchronous - blocking HTTP call (5s timeout). Inside a cancellation handler - it would: - - 1. Delay task cancellation by up to 5s on network errors - (timeout / shutdown / Ctrl+C). - 2. Make pending-task warnings ("Task was destroyed but it is - pending") more frequent and harder to diagnose. - 3. In some shutdown paths, the cancel I/O itself gets - cancelled, leaving orphan cleanup incomplete — but the - server is idempotent on /cancel so this is acceptable - (orphan via TTL/reconciliation instead). - - Cancel MUST NOT fire on CancelledError. Execution_id is set in - ContextVar (to prove the helper WOULD have run if the except - had caught BaseException), and the assertion is strict: - cancel_calls must be empty. - """ - rt = runtime_factory() - - # Capture execution_id as if /gate succeeded. The cancel helper - # would see this if the except caught BaseException. - set_server_minted_execution_id("exec-cancellation-test") - - @protect - async def fn(): - raise asyncio.CancelledError("task got cancelled") - - with pytest.raises(asyncio.CancelledError): - await fn() - - assert rt.cancel_calls == [], ( - f"cancel_execution was called on CancelledError: {rt.cancel_calls}. " - "This is the asyncio cancellation-safety regression." - ) - - -@pytest.mark.asyncio -async def test_2_track_tool_failure_after_fn_completion_does_not_trigger_cancel( - runtime_factory, -): - """The second regression guard. fn_completed sentinel must stay True - past the successful fn() call. - - If someone removes the `fn_completed` sentinel — "simplify the - wrapper, just always call cancel on Exception" — this test - fails. The cost of THAT regression is a different kind of bad: - cancel_execution tells the server "abort this execution, no - side effects happened". But fn() already ran. Side effects - already happened (LLM call emitted a response, tool ran, - possibly mutated state). The reservation got DECRBY'd because - track_tool would have consumed it on success; cancelling - instead releases the budget slot and tells the server the - side effect didn't happen — which is a lie that breaks audit - and produces a phantom "refund". - - Track_tool failure is rare (network error to /track batch - sender). On that path we want to RETRY track_tool, not cancel. - The orphan-if-no-retry is a server-side concern; SDK can't - usefully address it from outside. - """ - rt = runtime_factory( - track_tool_raises=NullRunTransportError( - "network down for /track", - source=TransportErrorSource.NETWORK_ERROR, - endpoint="/api/v1/track", - ), - ) - - @protect - async def fn(): - # fn succeeds, side effects happen here in real code - return "tool-output-data" - - with pytest.raises(NullRunTransportError): - await fn() - - assert rt.cancel_calls == [], ( - f"cancel_execution was called after successful fn+broken track_tool: " - f"{rt.cancel_calls}. fn_completed sentinel regressed; cancelling an " - "actually-completed tool would mislead budget accounting." - ) - - -# ───────────────────────────────────────────────────────────────────── -# PRIORITY 2 — mechanical coverage of the value-side assertions. -# Less interesting than 1+2 because they're not regression guards -# against over-broad exception handling. They're 'happy path' -# coverage of the cancel logic itself. -# ───────────────────────────────────────────────────────────────────── - - -@pytest.mark.asyncio -async def test_3_async_fn_raises_value_error_triggers_cancel(runtime_factory): - """fn() raises a business exception. The wrapper's except Exception - catches; fn_completed is False (the raise happened before - fn_completed could be set); the helper sees execution_id in the - ContextVar (captured by check_workflow_budget) and calls - cancel_execution.""" - rt = runtime_factory() - - @protect - async def fn(): - raise ValueError("llm returned malformed JSON") - - with pytest.raises(ValueError, match="malformed JSON"): - await fn() - - assert rt.cancel_calls == [ - ("exec-test-123", "tool_exception"), - ], f"expected exactly one cancel call with captured exec id; got {rt.cancel_calls}" - - # Sanity: track_tool is NOT called when fn() raises — the body - # never reached it. - assert "track_tool" not in rt.gate_calls - - -@pytest.mark.asyncio -async def test_4_async_no_execution_id_skips_cancel(runtime_factory): - """control_plane rejects BEFORE /gate is called. The ContextVar - stays None; the helper is a no-op even though the wrapper's - except ran. The orphan, if any (server-side, depends on whether - /gate ran), is handled by TTL/reconciliation, not the SDK.""" - rt = runtime_factory( - control_plane_raises=WorkflowKilledInterrupt( - workflow_id="default", - reason="user clicked kill", - ), - capture_execution_id=None, # capture_execution_id=None means: don't capture - ) - - @protect - async def fn(): - return "should not run" - - with pytest.raises(WorkflowKilledInterrupt): - await fn() - - # control_plane ran (and raised); budget did NOT run (short-circuited). - assert rt.gate_calls == ["control_plane"] - # No exec_id → no cancel. Server-side orphan, if any, is reconciliation territory. - assert rt.cancel_calls == [], ( - "control_plane reject happens pre-/gate. No reservation was created " - "from the SDK's perspective. SDK must not call cancel." - ) - - -def test_5_sync_fn_raises_value_error_triggers_cancel(runtime_factory): - """Sync wrapper mirrors async except for the BaseException vs - Exception asymmetry. ValueError (a regular Exception) is caught - on both paths; cancel runs with the captured execution_id.""" - rt = runtime_factory() - - @protect - def fn(): - raise ValueError("bad input") - - with pytest.raises(ValueError): - fn() - - assert rt.cancel_calls == [ - ("exec-test-123", "tool_exception"), - ] - - -def test_6_sync_baseexception_also_triggers_cancel(runtime_factory): - """Sync path catches BaseException. KeyboardInterrupt (which - async_wrapper deliberately lets propagate to keep cancellation - fast) is caught here and triggers cancel — sync code has no - event loop to delay, and a Ctrl+C during a long sync agent - gets a few seconds of cancel I/O before exit. Matches existing - `_protect_body` unify_block semantics.""" - rt = runtime_factory() - - @protect - def fn(): - raise KeyboardInterrupt() - - with pytest.raises(KeyboardInterrupt): - fn() - - assert rt.cancel_calls == [ - ("exec-test-123", "tool_exception"), - ] - - -# ───────────────────────────────────────────────────────────────────── -# Sanity check — the happy path still works. Re-pinning this here -# because the cancellation fix could in principle break the -# success path (the new try/except adds a frame and a closure -# capturing fn_completed). -# ───────────────────────────────────────────────────────────────────── - - -@pytest.mark.asyncio -async def test_happy_path_no_cancel_called(runtime_factory): - """fn() succeeds, track_tool succeeds, control returns normally. - NO cancel call (we're not in except path). Sanity check that - the wrapper's added try/except didn't break success.""" - rt = runtime_factory() - - @protect - async def fn(): - return "ok" - - result = await fn() - - assert result == "ok" - assert rt.cancel_calls == [] - assert rt.gate_calls == ["control_plane", "budget", "track_tool"] diff --git a/tests/test_protect_only_public_api.py b/tests/test_protect_only_public_api.py deleted file mode 100644 index 07590da..0000000 --- a/tests/test_protect_only_public_api.py +++ /dev/null @@ -1,306 +0,0 @@ -"""Pin tests for SDK 0.18.1 @protect-only public API. - -Three properties pinned here, each in a single regression test: - -1. ``@protect`` auto-attaches a default ``ToolParamsExtractor`` so the - wire payload carries ``tool_name + params`` for any protected - tool, not just those that opted in via bare ``@sensitive``. - Explicit ``@sensitive(impact=...)`` still wins (chain walk). - -2. The default extraction is bounded: - - oversized string values get a deterministic - ``...[truncated:N bytes]`` suffix; - - circular references in nested dict/list structures return - the partial walk instead of raising ``RecursionError``; - - dropped values (float / bytes / unsupported types) get an - aggregate DEBUG log line, never one per dropped field. - -3. Bare ``@sensitive`` emits ``DeprecationWarning`` while still - running its legacy behaviour (auto-attach + sensitive-tool - registration). The factory form ``@sensitive(impact=...)`` - is not deprecated — it remains the explicit advanced API. -""" - -from __future__ import annotations - -import logging -import warnings - -import pytest - -import nullrun -from nullrun.decorators import protect, sensitive -from tests.conftest import BASE_URL - - -# Tests that touch ``@sensitive`` (which calls -# ``_do_sensitive_register`` → ``_get_or_create_runtime``) need the -# SDK to be able to construct a ``NullRunRuntime`` instance. We -# initialise once per test against the ``mock_api`` HTTP stub so -# registration does not raise ``NullRunAuthenticationError`` -# before our assertions run. ``reset_runtime`` (autouse) tears the -# singleton down again after each test, so this init is cheap. -@pytest.fixture(autouse=True) -def _init_sdk(mock_api): - nullrun.init(api_key="test-key-12345678", api_url=BASE_URL) - yield - - -# --------------------------------------------------------------------------- -# Property 1 — @protect auto-attaches a default ToolParamsExtractor -# --------------------------------------------------------------------------- - - -def test_protect_auto_attaches_default_tool_params_extractor() -> None: - """Bare @protect stamps a ToolParamsExtractor(include_all=True) on the fn.""" - - @protect - def refund_customer(customer_id: str, amount: int) -> str: - return "ok" - - extractor = getattr(refund_customer, "_nullrun_extractor", None) - assert extractor is not None, ( - "@protect must auto-attach a default ToolParamsExtractor" - ) - assert extractor.include_all is True, ( - "default @protect extractor must capture every kwarg" - ) - assert extractor.param_extractors is None, ( - "default @protect extractor must use include_all mode, not a map" - ) - # 0.18.1: the auto-attached extractor must be tagged so the - # policy gate can skip the /execute round-trip for bare - # @protect (latency split — bare @protect is cheap, explicit - # @sensitive(impact=...) is policy-gated). - assert getattr(extractor, "_nullrun_auto_attached", False) is True, ( - "auto-attached extractor must carry the _nullrun_auto_attached " - "marker so _enforce_sensitive_tool can distinguish it from " - "explicit @sensitive(impact=...) extractors" - ) - - -def test_protect_does_not_overwrite_explicit_sensitive_impact_extractor() -> None: - """@protect chain walk preserves an explicit @sensitive(impact=...) extractor.""" - - from nullrun.extractor import money_outflow - - # Order: @protect inner, @sensitive(impact=...) outer. - # Decorators apply bottom-up, so: - # 1. @protect wraps explicit_money → sync_wrapper1. - # At this point NO extractor exists in the chain; my new - # @protect auto-attach stamps ToolParamsExtractor on the - # bare fn. Then @sensitive(impact=...) factory runs and - # stamps MoneyImpactExtractor on the same bare fn (overwriting - # the ToolParamsExtractor), THEN registers the tool. - # 2. The end-state on the bare fn is MoneyImpactExtractor. - @protect - @sensitive(impact=money_outflow(argument="amount_cents", currency="USD", units="minor")) - def explicit_money(amount_cents: int) -> str: - return "ok" - - extractor = getattr(explicit_money, "_nullrun_extractor", None) - assert extractor is not None - # money_outflow returns a MoneyImpactExtractor, not a ToolParamsExtractor; - # we identify it by class name to avoid importing the concrete class here. - assert type(extractor).__name__ == "MoneyImpactExtractor", ( - "@protect chain walk must preserve an explicit extractor; got " - f"{type(extractor).__name__!r}" - ) - - -# --------------------------------------------------------------------------- -# Property 2 — bounded extraction -# --------------------------------------------------------------------------- - - -def test_oversize_string_value_is_truncated_with_marker() -> None: - """String values over 1024 bytes get a deterministic truncation suffix.""" - - @protect - def upload(description: str) -> str: - return "ok" - - big = "x" * 5000 - # Reach into the extractor the way the gate would, by calling - # ``impact_for`` directly so we don't have to spin up a runtime. - extractor = upload._nullrun_extractor - impact = extractor.impact_for(upload, (), {"description": big}) - params = impact.to_wire_dict()["params"] - assert "description" in params - value = params["description"] - assert value.endswith(" bytes]"), ( - f"truncated value must end with the marker; got tail={value[-30:]!r}" - ) - assert "...[truncated:" in value - # The returned string must be bounded (marker is sized so the - # result never exceeds the cap, including the marker itself). - assert len(value.encode("utf-8")) <= 1024, ( - f"truncated value must be <= 1024 bytes; got {len(value.encode('utf-8'))}" - ) - - -def test_circular_reference_in_nested_dict_does_not_recurse_infinitely() -> None: - """A self-referential dict returns the partial walk, not RecursionError.""" - - @protect - def process(config: dict) -> str: - return "ok" - - cyclic: dict = {"outer": "value"} - cyclic["self"] = cyclic # type: ignore[assignment] - - extractor = process._nullrun_extractor - # Should NOT raise RecursionError. The bound walk stops on cycle. - impact = extractor.impact_for(process, (), {"config": cyclic}) - params = impact.to_wire_dict()["params"] - assert "config" in params - # The outer key survived; the cycle marker is the partial walk. - inner = params["config"] - assert "outer" in inner - # The recursive key was bounded to a partial structure. - assert isinstance(inner["self"], dict) - - -def test_dropped_values_emit_aggregate_debug_log_not_per_field(caplog) -> None: - """Dropped values (float, bytes, custom) get ONE aggregate DEBUG line.""" - - class Custom: - def __repr__(self) -> str: - return "Custom()" - - @protect - def mixed(a: float, b: bytes, c: object) -> str: - return "ok" - - extractor = mixed._nullrun_extractor - with caplog.at_level(logging.DEBUG, logger="nullrun.extractor"): - impact = extractor.impact_for( - mixed, - (), - {"a": 1.5, "b": b"hello", "c": Custom()}, - ) - params = impact.to_wire_dict()["params"] - assert "a" not in params and "b" not in params and "c" not in params, ( - f"unsupported types must be dropped; got params={params!r}" - ) - debug_lines = [ - r for r in caplog.records if r.name == "nullrun.extractor" - ] - assert len(debug_lines) == 1, ( - f"expected exactly one aggregate DEBUG line, got {len(debug_lines)}" - ) - msg = debug_lines[0].getMessage() - assert "3" in msg, f"aggregate line must report count=3; got {msg!r}" - assert "float" in msg and "bytes" in msg and "Custom" in msg, ( - f"aggregate line must list type names; got {msg!r}" - ) - # Argument names ("a", "b", "c") are NEVER logged as standalone - # tokens. Substring matches against type names like "bytes" are - # acceptable (the line format is "type_name=count" with a space - # delimiter, so a literal arg name like "a" would never appear - # unaccompanied). We assert on the structured format instead. - tokens = set(msg.replace("=", " ").replace(",", " ").split()) - assert "a" not in tokens and "b" not in tokens and "c" not in tokens, ( - f"log line tokens must NOT contain argument names; got tokens={tokens!r}" - ) - - -# --------------------------------------------------------------------------- -# Property 3 — bare @sensitive DeprecationWarning -# --------------------------------------------------------------------------- - - -def test_bare_sensitive_emits_deprecation_warning() -> None: - """Bare @sensitive still works in 0.18.x but emits DeprecationWarning.""" - - with warnings.catch_warnings(record=True) as caught: - warnings.simplefilter("always") - - @sensitive - def legacy_tool(x: int) -> str: - return "ok" - - deprecation_warnings = [ - w for w in caught if issubclass(w.category, DeprecationWarning) - ] - assert len(deprecation_warnings) == 1, ( - f"bare @sensitive must emit exactly one DeprecationWarning; got " - f"{len(deprecation_warnings)}: {[str(w.message) for w in deprecation_warnings]}" - ) - assert "0.18.1" in str(deprecation_warnings[0].message) - # Legacy behaviour is preserved for this release. - extractor = getattr(legacy_tool, "_nullrun_extractor", None) - assert extractor is not None, ( - "bare @sensitive must still stamp a default extractor in 0.18.x" - ) - - -def test_sensitive_factory_with_explicit_impact_does_not_warn() -> None: - """@sensitive(impact=...) is the advanced API and must NOT warn.""" - - from nullrun.extractor import money_outflow - - with warnings.catch_warnings(record=True) as caught: - warnings.simplefilter("always") - - @sensitive(impact=money_outflow(argument="x", currency="USD", units="minor")) - def advanced_tool(x: int) -> str: - return "ok" - - deprecation_warnings = [ - w for w in caught if issubclass(w.category, DeprecationWarning) - ] - assert deprecation_warnings == [], ( - f"@sensitive(impact=...) must not emit DeprecationWarning; got " - f"{[str(w.message) for w in deprecation_warnings]}" - ) - - -# --------------------------------------------------------------------------- -# Sanity: the auto-attach on @protect composes with bare @sensitive -# --------------------------------------------------------------------------- - - -def test_bare_sensitive_then_protect_does_not_double_attach() -> None: - """@sensitive outside @protect must not double-attach an extractor.""" - - with warnings.catch_warnings(): - warnings.simplefilter("ignore", DeprecationWarning) - - @sensitive - @protect - def composed(x: int) -> str: - return "ok" - - extractor = getattr(composed, "_nullrun_extractor", None) - assert extractor is not None - # Only one ToolParamsExtractor is attached; the chain walk sees - # the explicit one and skips auto-attach on @protect. - assert type(extractor).__name__ == "ToolParamsExtractor" - - -def test_auto_attached_extractor_is_distinguished_from_explicit() -> None: - """The auto-attached marker survives on the inner extractor; explicit impact wins.""" - - from nullrun.extractor import money_outflow - - # Bare @protect → auto-attached extractor is stamped with marker. - @protect - def bare_protected(x: int) -> str: - return "ok" - - bare_extractor = getattr(bare_protected, "_nullrun_extractor") - assert getattr(bare_extractor, "_nullrun_auto_attached", False) is True - - # Explicit @sensitive(impact=...) overwrites the auto-attached - # extractor; the new one does NOT carry the marker. - @protect - @sensitive(impact=money_outflow(argument="x", currency="USD", units="minor")) - def explicit_protected(x: int) -> str: - return "ok" - - explicit_extractor = getattr(explicit_protected, "_nullrun_extractor") - assert getattr(explicit_extractor, "_nullrun_auto_attached", False) is False, ( - "explicit @sensitive(impact=...) must stamp an extractor WITHOUT " - "the auto-attached marker so the policy gate fires" - ) diff --git a/tests/test_runtime.py b/tests/test_runtime.py index 885d5f4..c827c72 100644 --- a/tests/test_runtime.py +++ b/tests/test_runtime.py @@ -179,10 +179,8 @@ def test_execute_blocked_raises(self, make_runtime, mock_api): ) ) rt = make_runtime() - # Use mode="strict" to force gateway call - # (auto mode might use inline for non-sensitive tools) with pytest.raises(NullRunBlockedException): - rt.execute(tool_name="gpt-4", input_data={}, mode="strict") + rt.execute(tool_name="gpt-4", input_data={}) def test_execute_blocked_surfaces_wire_error_code(self, make_runtime, mock_api): # DEF-ARFLOW-TOOLNAME-01 (E2E 2026-08-05): the backend now stamps @@ -226,7 +224,7 @@ def test_execute_blocked_surfaces_wire_error_code(self, make_runtime, mock_api): ) rt = make_runtime() with pytest.raises(NullRunBlockedException) as exc_info: - rt.execute(tool_name="refund_customer", input_data={}, mode="strict") + rt.execute(tool_name="refund_customer", input_data={}) # Catalog typed class wins — not the keyword-guessing fallback. assert isinstance(exc_info.value, NullRunApprovalDbUnavailableError) # The catalog error_code (NR-A016) is canonical for @@ -644,155 +642,6 @@ def test_check_control_plane_empty_cache_fetches(monkeypatch): assert fetch_calls == ["wf-1"] -# ─── is_sensitive_tool ─────────────────────────────────────────────── - - -def test_is_sensitive_tool_built_in_match(): - rt = _make_test_runtime() - assert rt.is_sensitive_tool("stripe.charge") is True - - -def test_is_sensitive_tool_case_insensitive(): - rt = _make_test_runtime() - assert rt.is_sensitive_tool("Stripe.Charge") is True - assert rt.is_sensitive_tool("STRIPE.CHARGE") is True - - -def test_is_sensitive_tool_unknown_returns_false(): - rt = _make_test_runtime() - assert rt.is_sensitive_tool("my.custom_tool") is False - - -def test_is_sensitive_tool_after_register(): - rt = _make_test_runtime() - rt.add_sensitive_tool("my.tool") - assert rt.is_sensitive_tool("my.tool") is True - - -def test_is_sensitive_tool_after_remove(): - rt = _make_test_runtime() - rt.add_sensitive_tool("my.tool") - rt.remove_sensitive_tool("my.tool") - assert rt.is_sensitive_tool("my.tool") is False - - -def test_remove_sensitive_tool_unknown_is_silent(): - rt = _make_test_runtime() - rt.remove_sensitive_tool("never.registered") # must not raise - - -# ─── register_sensitive_tools / get_sensitive_tools ────────────────── - - -def test_register_sensitive_tools_bulk(): - rt = _make_test_runtime() - rt.register_sensitive_tools(["a", "b", "c"]) - tools = rt.get_sensitive_tools() - assert "a" in tools - assert "b" in tools - assert "c" in tools - # Built-in sensitive tools are also in the union. - assert "stripe.charge" in tools - - -# 0.9.0: removed six `coverage_report` / `bump_coverage_counter` -# tests at lines 223-278. The `_coverage_seen` / -# `_coverage_tracked` / `_coverage_streaming_skipped` dicts -# `coverage_report `, `track_coverage ` -# `start_coverage_reporter `, `_coverage_reporter_loop `, and -# `bump_coverage_counter ` method are all gone — coverage is now -# derived server-side from llm_call span metadata. See plan at -# `~/.claude/plans/async-swinging-hanrahan.md`. - - -# ─── execute mode resolution ────────────────────────────────────── - - -def test_execute_auto_sensitive_routes_to_strict(): - rt = _make_test_runtime() - rt._transport.execute = MagicMock( - return_value={"decision": "allow", "decision_source": "gateway"} - ) - rt.execute("stripe.charge", {"amount": 5}) # sensitive → strict - call_args = rt._transport.execute.call_args - # Runtime.execute forwards mode as a kwarg. - assert call_args.kwargs["mode"] == "strict" - - -def test_execute_auto_non_sensitive_routes_to_strict(): - """DEF-TS12-01 (2026-09-10): ``mode="auto"`` with a non-sensitive - tool now ALWAYS resolves to ``mode="strict"`` and contacts the - gateway. - - Pre-fix this was a CRITICAL fail-OPEN: ``mode="auto"`` with a - non-sensitive tool silently switched to ``mode="inline"`` which - returned a synthetic local allow WITHOUT contacting the gateway. - Every operator-configured budget, rate-limit, and tool-block - policy was silently bypassed for non-sensitive tools. The - dashboard showed policies in effect; the SDK ignored them. - - Post-fix the cloud-only invariant from CLAUDE.md §17 and - memory `cloud-only-invariant-sdk` holds: every call routes - through /execute when ``mode="auto"`` (the default). The - ``mode="inline"`` opt-in is preserved for callers who - explicitly want to skip /execute — see - ``test_execute_inline_mode_short_circuits_local`` below for - the explicit opt-in pin. - """ - rt = _make_test_runtime() - rt._transport.execute = MagicMock( - return_value={"decision": "allow", "decision_source": "gateway"} - ) - rt.execute("safe.tool", {"x": 1}) - rt._transport.execute.assert_called_once() - assert rt._transport.execute.call_args.kwargs["mode"] == "strict" - - -def test_execute_auto_sensitive_calls_transport(): - """Auto + sensitive tool → mode=strict → transport.execute is called.""" - rt = _make_test_runtime() - rt._transport.execute = MagicMock( - return_value={"decision": "allow", "decision_source": "gateway"} - ) - rt.execute("stripe.charge", {"amount": 5}) - rt._transport.execute.assert_called_once() - assert rt._transport.execute.call_args.kwargs["mode"] == "strict" - - -def test_execute_inline_mode_short_circuits_local(): - """Inline + non-sensitive tool → LOCAL decision, no HTTP call.""" - rt = _make_test_runtime() - rt._transport.execute = MagicMock() - result = rt.execute("safe.tool", {"x": 1}, mode="inline") - assert result["decision"] == "allow" - assert result["decision_source"] == "local" - rt._transport.execute.assert_not_called() - - -def test_execute_inline_sensitive_still_calls_transport(): - """Inline mode + sensitive tool still routes to /execute.""" - rt = _make_test_runtime() - rt._transport.execute = MagicMock( - return_value={"decision": "allow", "decision_source": "gateway"} - ) - rt.execute("stripe.charge", {"amount": 5}, mode="inline") - rt._transport.execute.assert_called_once() - - -def test_execute_block_raises_NullRunBlockedException(): - rt = _make_test_runtime() - rt._transport.execute = MagicMock( - return_value={ - "decision": "block", - "decision_source": "gateway", - "explanation": "denied by policy", - } - ) - with pytest.raises(NullRunBlockedException) as excinfo: - rt.execute("stripe.charge", {"amount": 5}) # sensitive → routes to /execute - assert excinfo.value.reason == "denied by policy" - - # ─── shutdown ──────────────────────────────────────────────────────── diff --git a/tests/test_runtime_branches.py b/tests/test_runtime_branches.py index cb372a1..57ab4f4 100644 --- a/tests/test_runtime_branches.py +++ b/tests/test_runtime_branches.py @@ -183,142 +183,6 @@ def test_check_control_plane_empty_cache_fetches(monkeypatch): assert fetch_calls == ["wf-1"] -# ─── is_sensitive_tool ─────────────────────────────────────────────── - - -def test_is_sensitive_tool_built_in_match(): - rt = _make_test_runtime() - assert rt.is_sensitive_tool("stripe.charge") is True - - -def test_is_sensitive_tool_case_insensitive(): - rt = _make_test_runtime() - assert rt.is_sensitive_tool("Stripe.Charge") is True - assert rt.is_sensitive_tool("STRIPE.CHARGE") is True - - -def test_is_sensitive_tool_unknown_returns_false(): - rt = _make_test_runtime() - assert rt.is_sensitive_tool("my.custom_tool") is False - - -def test_is_sensitive_tool_after_register(): - rt = _make_test_runtime() - rt.add_sensitive_tool("my.tool") - assert rt.is_sensitive_tool("my.tool") is True - - -def test_is_sensitive_tool_after_remove(): - rt = _make_test_runtime() - rt.add_sensitive_tool("my.tool") - rt.remove_sensitive_tool("my.tool") - assert rt.is_sensitive_tool("my.tool") is False - - -def test_remove_sensitive_tool_unknown_is_silent(): - rt = _make_test_runtime() - rt.remove_sensitive_tool("never.registered") # must not raise - - -# ─── register_sensitive_tools / get_sensitive_tools ────────────────── - - -def test_register_sensitive_tools_bulk(): - rt = _make_test_runtime() - rt.register_sensitive_tools(["a", "b", "c"]) - tools = rt.get_sensitive_tools() - assert "a" in tools - assert "b" in tools - assert "c" in tools - # Built-in sensitive tools are also in the union. - assert "stripe.charge" in tools - - -# 0.9.0: removed six `coverage_report` / `bump_coverage_counter` -# tests at lines 223-278. The `_coverage_seen` / -# `_coverage_tracked` / `_coverage_streaming_skipped` dicts -# `coverage_report `, `track_coverage ` -# `start_coverage_reporter `, `_coverage_reporter_loop `, and -# `bump_coverage_counter ` method are all gone — coverage is now -# derived server-side from llm_call span metadata. See plan at -# `~/.claude/plans/async-swinging-hanrahan.md`. - - -# ─── execute mode resolution ────────────────────────────────────── - - -def test_execute_auto_sensitive_routes_to_strict(): - rt = _make_test_runtime() - rt._transport.execute = MagicMock( - return_value={"decision": "allow", "decision_source": "gateway"} - ) - rt.execute("stripe.charge", {"amount": 5}) # sensitive → strict - call_args = rt._transport.execute.call_args - # Runtime.execute forwards mode as a kwarg. - assert call_args.kwargs["mode"] == "strict" - - -def test_execute_auto_non_sensitive_routes_to_strict(): - """DEF-TS12-01 (2026-09-10): ``mode="auto"`` with a non-sensitive - tool now ALWAYS resolves to ``mode="strict"`` and contacts the - gateway. Pre-fix this routed to ``mode="inline"`` which bypassed - the gateway entirely — see ``test_runtime.py`` for the original - inline-bypass pin (now inverted). Cloud-only invariant from - CLAUDE.md §17. - """ - rt = _make_test_runtime() - rt._transport.execute = MagicMock( - return_value={"decision": "allow", "decision_source": "gateway"} - ) - rt.execute("safe.tool", {"x": 1}) - rt._transport.execute.assert_called_once() - assert rt._transport.execute.call_args.kwargs["mode"] == "strict" - - -def test_execute_auto_sensitive_calls_transport(): - """Auto + sensitive tool → mode=strict → transport.execute is called.""" - rt = _make_test_runtime() - rt._transport.execute = MagicMock( - return_value={"decision": "allow", "decision_source": "gateway"} - ) - rt.execute("stripe.charge", {"amount": 5}) - rt._transport.execute.assert_called_once() - assert rt._transport.execute.call_args.kwargs["mode"] == "strict" - - -def test_execute_inline_mode_short_circuits_local(): - """Inline + non-sensitive tool → LOCAL decision, no HTTP call.""" - rt = _make_test_runtime() - rt._transport.execute = MagicMock() - result = rt.execute("safe.tool", {"x": 1}, mode="inline") - assert result["decision"] == "allow" - assert result["decision_source"] == "local" - rt._transport.execute.assert_not_called() - - -def test_execute_inline_sensitive_still_calls_transport(): - """Inline mode + sensitive tool still routes to /execute.""" - rt = _make_test_runtime() - rt._transport.execute = MagicMock( - return_value={"decision": "allow", "decision_source": "gateway"} - ) - rt.execute("stripe.charge", {"amount": 5}, mode="inline") - rt._transport.execute.assert_called_once() - - -def test_execute_block_raises_NullRunBlockedException(): - rt = _make_test_runtime() - rt._transport.execute = MagicMock( - return_value={ - "decision": "block", - "decision_source": "gateway", - "explanation": "denied by policy", - } - ) - with pytest.raises(NullRunBlockedException) as excinfo: - rt.execute("stripe.charge", {"amount": 5}) # sensitive → routes to /execute - assert excinfo.value.reason == "denied by policy" - # ─── shutdown ──────────────────────────────────────────────────────── diff --git a/tests/test_sensitive_extractor.py b/tests/test_sensitive_extractor.py deleted file mode 100644 index 9938892..0000000 --- a/tests/test_sensitive_extractor.py +++ /dev/null @@ -1,289 +0,0 @@ -"""Typed impact + digest-bound approval -- SDK e2e for the @sensitive(impact=...) path. - -These tests pin the wire shape produced by the auto-wire path: -when a sensitive tool decorated with ``@sensitive(impact=...)`` -is invoked through ``@protect``, the SDK sends -``business_impact`` + ``action_digest`` on the wire so the -backend can stamp the approval row with the digest and refuse -tampered payloads on the post-approval re-check. - -The previous attempt (rolled back) failed because the test -fixture built a fresh ``NullRunRuntime`` instance but did NOT -register it in the ``RuntimeRegistry``. ``_get_or_create_runtime`` -therefore re-created a new singleton, and the -``monkeypatch.setattr(rt, "_transport", cap)`` swap was on the -unregistered instance. The new tests use ``get_registry().set(rt)`` -to wire the test runtime as the active one. -""" - -from __future__ import annotations - -from typing import Any - -import pytest - -from nullrun._registry import get_registry -from nullrun.business_impact import ( - OUTFLOW, - BusinessImpact, - compute_action_digest, -) -from nullrun.decorators import _enforce_sensitive_tool -from nullrun.extractor import money_outflow -from nullrun.runtime import NullRunRuntime - -# --------------------------------------------------------------------------- -# Wire payload capture -# --------------------------------------------------------------------------- - - -class _PayloadCapture: - """Trampoline that records the most recent kwargs to - ``runtime._transport.execute`` and returns a synthetic "allow" - decision. - - The recorder is bound to a freshly-built Transport instance via - ``monkeypatch.setattr(rt, "_transport", instance)`` and the SDK - invokes ``instance.execute(**kwargs)``. We capture the kwargs - by overriding ``execute`` on the instance via - ``monkeypatch.setattr(instance, "execute", self)`` in the - fixture below — this is the pattern already used by - ``test_execute_approval_flow.py``. - """ - - def __init__(self) -> None: - self.last_kwargs: dict[str, Any] | None = None - - def __call__(self, *args: Any, **kwargs: Any) -> dict[str, Any]: - # Real transport.execute takes kwargs only. We accept - # *args for forward-compat (a future transport may pass - # positional metadata) but pin the contract on kwargs. - del args - self.last_kwargs = kwargs - return { - "decision": "allow", - "decision_source": "test_capture", - "policy_version": 0, - "allow_execution": True, - } - - -@pytest.fixture -def captured_runtime(monkeypatch): - """Build a test-mode runtime, register it as the active - singleton in ``RuntimeRegistry``, and rebind - ``_transport.execute`` to a recorder. Yield the runtime for - tests to register tools on. - - We bind ``execute`` on the freshly-built transport rather - than swapping the whole transport object: that's the pattern - in ``test_execute_approval_flow.py`` and it works with the - SDK's ``self._transport.execute(**kwargs)`` method call. - """ - NullRunRuntime.reset_instance() - rt = NullRunRuntime(api_key="nr_test_phase1", _test_mode=True) - cap = _PayloadCapture() - monkeypatch.setattr(rt._transport, "execute", cap) - # Wire the test runtime as the singleton so the SDK's - # ``_get_or_create_runtime()`` returns OUR instance (not a - # freshly-constructed one). - get_registry().set(rt) - yield rt - get_registry().clear() - NullRunRuntime.reset_instance() - - -@pytest.fixture -def captured_payload(captured_runtime) -> _PayloadCapture: - return captured_runtime._transport.execute # type: ignore[attr-defined,return-value] - - -# --------------------------------------------------------------------------- -# Typed-impact tools: built manually instead of via the @sensitive -# decorator. The decorator wiring is exercised by -# ``test_decorator_factory_form_attaches_extractor`` below. -# --------------------------------------------------------------------------- - - -def _refund_customer_impl(amount_cents: int, customer_id: str = "c-1") -> dict[str, Any]: - return {"customer": customer_id, "amount": amount_cents} - - -def _register_refund_tool(rt: NullRunRuntime) -> Any: - """Bind ``_refund_customer_impl`` with the typed-impact extractor - and register it as a sensitive tool. Mirrors what - ``@sensitive(impact=money_outflow(argument="amount_cents"))`` - would do at decorator-application time, but without paying - the ``@sensitive`` registration cost on every test. - """ - fn = _refund_customer_impl - extractor = money_outflow(argument="amount_cents") - setattr(fn, "_nullrun_extractor", extractor) - rt.add_sensitive_tool(fn.__name__) - return fn - - -def _register_legacy_tool(rt: NullRunRuntime) -> Any: - """Bind a sensitive tool WITHOUT a typed-impact extractor (legacy - approval_id-only path). The wrapper must NOT attach business_impact - or action_digest to the wire. - """ - def search_docs(query: str) -> list[str]: - return [query] - - rt.add_sensitive_tool(search_docs.__name__) - return search_docs - - -# --------------------------------------------------------------------------- -# 6.4: SDK e2e -- the wire payload contains business_impact + action_digest -# --------------------------------------------------------------------------- - - -class TestSensitiveExtractorWirePayload: - """Pin the wire shape produced by the typed-impact auto-wire path. - - These tests replace the SDK's transport with a recorder and - invoke ``_enforce_sensitive_tool`` directly. The capture - captures the kwargs the SDK sends; the test asserts those - kwargs match the wire shape documented in - ``contracts/openapi.yaml`` for ``GateRequest``. - """ - - def test_refund_customer_50_dollars_sends_typed_business_impact( - self, captured_payload: _PayloadCapture, captured_runtime: NullRunRuntime - ) -> None: - """Refund $50: business_impact + action_digest must land - on the wire. - """ - fn = _register_refund_tool(captured_runtime) - - _enforce_sensitive_tool(captured_runtime, fn, (5_000,), {"customer_id": "c-1"}) - - assert captured_payload.last_kwargs is not None - kwargs = captured_payload.last_kwargs - - # Both typed-impact fields must be present because the function - # has the extractor attribute set. - assert "business_impact" in kwargs, ( - f"Typed-impact contract broken: business_impact missing from " - f"wire kwargs: {sorted(kwargs.keys())}" - ) - assert "action_digest" in kwargs, ( - f"Typed-impact contract broken: action_digest missing from " - f"wire kwargs: {sorted(kwargs.keys())}" - ) - - impact = kwargs["business_impact"] - assert impact["kind"] == "money" - assert impact["direction"] == "outflow" - assert impact["amount_minor"] == 5_000 - assert impact["currency"] == "USD" - - # Digest is byte-identical to the SDK's own computation. - expected = compute_action_digest( - BusinessImpact.money(OUTFLOW, 5_000, "USD") - ) - assert kwargs["action_digest"] == expected - - def test_legacy_sensitive_tool_sends_no_business_impact( - self, captured_payload: _PayloadCapture, captured_runtime: NullRunRuntime - ) -> None: - """Legacy path: tool is sensitive but has no impact - extractor. The SDK MUST NOT attach business_impact or - action_digest to the wire -- the backend falls back to - approval_id-only grant consume. - """ - fn = _register_legacy_tool(captured_runtime) - - _enforce_sensitive_tool(captured_runtime, fn, ("hello",), {}) - - kwargs = captured_payload.last_kwargs - assert kwargs is not None - assert "business_impact" not in kwargs - assert "action_digest" not in kwargs - - def test_extractor_rejects_unknown_argument_at_call_time( - self, captured_runtime: NullRunRuntime - ) -> None: - """Typed-impact fail-CLOSED: if the extractor raises (e.g. - argument name mismatch), the pre-check MUST fail. The - body NEVER runs. - """ - from nullrun.breaker.exceptions import NullRunBlockedException - - def bad_tool(amount: int) -> dict[str, Any]: - return {"amount": amount} - - # Bind with an extractor that points to a non-existent - # argument. This deliberately raises TypeError in the - # extractor. - bad_tool._nullrun_extractor = money_outflow(argument="not_an_argument") - captured_runtime.add_sensitive_tool(bad_tool.__name__) - - with pytest.raises(NullRunBlockedException) as exc_info: - _enforce_sensitive_tool(captured_runtime, bad_tool, (42,), {}) - assert exc_info.value.error_code == "NR-B003" - assert "not_an_argument" in exc_info.value.reason - - def test_extractor_rejects_negative_amount( - self, captured_runtime: NullRunRuntime - ) -> None: - """Typed-impact fail-CLOSED: a negative amount must NOT pass - the pre-check. Without this, a hostile SDK caller could - subtract their way past the rule threshold by passing a - negative number. - """ - from nullrun.breaker.exceptions import NullRunBlockedException - - fn = _register_refund_tool(captured_runtime) - - with pytest.raises(NullRunBlockedException) as exc_info: - _enforce_sensitive_tool(captured_runtime, fn, (-1,), {"customer_id": "c-1"}) - assert exc_info.value.error_code == "NR-B003" - # Decimal support hardening: the negative-amount guard now - # lives in ``_to_minor_units`` (not in - # ``MoneyImpact.validate``), so the reason text - # matches the new "rejected negative" message. The - # legacy "non-negative" wording remains for - # backward-compatible callers via - # ``MoneyImpact.validate`` when an amount is somehow - # negative on the wire (defense-in-depth). - assert "rejected negative" in exc_info.value.reason - - def test_decorator_factory_form_attaches_extractor( - self, captured_payload: _PayloadCapture, captured_runtime: NullRunRuntime - ) -> None: - """@sensitive(impact=money_outflow(...)) factory form: - after the decorator applies, ``_nullrun_extractor`` is - stamped on the function and the wire payload carries the - typed business_impact + digest. - - This exercises the public decorator API end-to-end - rather than setting the attribute manually. - """ - from nullrun import sensitive - - @sensitive(impact=money_outflow(argument="amount_cents")) - def refund(amount_cents: int, customer_id: str = "c-1") -> dict[str, Any]: - return {"customer": customer_id, "amount": amount_cents} - - # The decorator must have stamped the extractor on the - # wrapped function. - assert hasattr(refund, "_nullrun_extractor"), ( - "decorator did not stamp _nullrun_extractor on the function" - ) - - # Also: the runtime must have registered the tool as - # sensitive. ``is_sensitive_tool`` is the public predicate. - assert captured_runtime.is_sensitive_tool(refund.__name__), ( - "decorator did not register the tool as sensitive" - ) - - _enforce_sensitive_tool(captured_runtime, refund, (5_000,), {"customer_id": "c-1"}) - - kwargs = captured_payload.last_kwargs - assert kwargs is not None - assert "business_impact" in kwargs - assert "action_digest" in kwargs - assert kwargs["business_impact"]["amount_minor"] == 5_000 diff --git a/tests/test_tool_params_extractor.py b/tests/test_tool_params_extractor.py deleted file mode 100644 index 837303d..0000000 --- a/tests/test_tool_params_extractor.py +++ /dev/null @@ -1,676 +0,0 @@ -"""Typed impact + digest-bound approval -- SDK e2e for the ToolParameters path. - -These tests pin the wire shape produced by ``@sensitive`` when -paired with ``ToolParamsExtractor`` (the tool-parameters -follow-up to ``MoneyImpactExtractor``). Mirrors the structure of -``test_sensitive_extractor.py`` so a reader who knows one file -knows the other. - -What this file covers: -- ``tool_params()`` factory: include_all default, explicit map, - mutual-exclusion guard -- ``ToolParamsExtractor.impact_for``: three extraction modes - (explicit map, include_all, empty) -- Auto-attach on bare ``@sensitive`` (no impact=...) ships - ``kind: "tool_call"`` with all kwargs as ``params`` -- Auto-attach does NOT overwrite an explicit - ``@sensitive(impact=money_outflow(...))`` -- PII-masked sentinels (``"***"``) are filtered out -- Unsupported types (float) cause fail-CLOSED at extraction time -- Wire payload contains ``business_impact`` (ToolCall shape) + - ``action_digest`` (byte-identical to the SDK's own computation) - -The tests use ``_enforce_sensitive_tool`` directly rather than -``@sensitive`` decoration when checking the extraction layer in -isolation, and ``@sensitive`` decoration when checking the -auto-attach wiring. Both paths are exercised; the second is -the production-critical one. -""" - -from __future__ import annotations - -import hashlib -import json -from typing import Any - -import pytest - -from nullrun._registry import get_registry -from nullrun.business_impact import ( - KIND_TOOL_CALL, - TOOL_PARAMETERS_MAX_PARAM_NAME, - BusinessImpact, - ToolCallParams, - compute_action_digest, -) -from nullrun.decorators import ( - _do_sensitive_register, - _enforce_sensitive_tool, - _find_extractor_in_chain, - _stamp_extractor_on_innermost, -) -from nullrun.extractor import ( - MoneyImpactExtractor, - ToolParamsExtractor, - money_outflow, - tool_params, -) -from nullrun.runtime import NullRunRuntime - -# --------------------------------------------------------------------------- -# Wire payload capture (same pattern as test_sensitive_extractor.py) -# --------------------------------------------------------------------------- - - -class _PayloadCapture: - """Trampoline that records the most recent kwargs to - ``runtime._transport.execute`` and returns a synthetic "allow" - decision. - """ - - def __init__(self) -> None: - self.last_kwargs: dict[str, Any] | None = None - - def __call__(self, *args: Any, **kwargs: Any) -> dict[str, Any]: - del args - self.last_kwargs = kwargs - return { - "decision": "allow", - "decision_source": "test_capture", - "policy_version": 0, - "allow_execution": True, - } - - -@pytest.fixture -def captured_runtime(monkeypatch): - """Build a test-mode runtime, register it as the active - singleton, and rebind ``_transport.execute`` to a recorder. - """ - NullRunRuntime.reset_instance() - rt = NullRunRuntime(api_key="nr_test_tier2", _test_mode=True) - cap = _PayloadCapture() - monkeypatch.setattr(rt._transport, "execute", cap) - get_registry().set(rt) - yield rt - get_registry().clear() - NullRunRuntime.reset_instance() - - -@pytest.fixture -def captured_payload(captured_runtime) -> _PayloadCapture: - return captured_runtime._transport.execute # type: ignore[attr-defined,return-value] - - -def _register_tool(rt: NullRunRuntime, fn: Any) -> Any: - """Manual registration helper -- mirrors what @sensitive does - at decoration time but without paying the runtime-singleton - init cost on every test. Used for the extraction-layer tests. - """ - rt.add_sensitive_tool(fn.__name__) - return fn - - -# --------------------------------------------------------------------------- -# 1. Factory tests (no runtime, no extraction -- just constructor shape) -# --------------------------------------------------------------------------- - - -class TestToolParamsFactory: - def test_default_is_include_all_true(self) -> None: - """``tool_params()`` with no args must capture every kwarg. - - Bare ``@sensitive`` auto-attaches this default; operators - adopting ToolParameters Approval Rules need the tool to - ship its args without rewriting every decorator site. - """ - e = tool_params() - assert e.param_extractors is None - assert e.include_all is True - assert e.extractor_id == "nullrun.tool_call.path" - assert e.extractor_version == "1" - - def test_explicit_map_overrides_include_all(self) -> None: - """Explicit ``{rule_param: arg_name}`` map wins over - ``include_all``. Operators use this when the rule name - diverges from the function arg name. - """ - e = tool_params({"user_id": "uid"}) - assert e.param_extractors == {"user_id": "uid"} - # ``include_all`` is irrelevant when param_extractors is - # set; the extractor ignores it. - assert e.include_all is True - - def test_mutual_exclusion_raises_at_construct_time(self) -> None: - """Setting both ``param_extractors`` and - ``include_all=False`` is almost certainly a typo. Fail - at decorator-application time rather than silently - dropping rules at run time. - """ - with pytest.raises(ValueError) as exc_info: - tool_params({"a": "b"}, include_all=False) - assert "mutually exclusive" in str(exc_info.value) - - -# --------------------------------------------------------------------------- -# 2. Extraction tests (runtime fixture, no @sensitive decorator) -# --------------------------------------------------------------------------- - - -class TestToolParamsExtraction: - def test_include_all_true_captures_every_kwarg( - self, captured_payload: _PayloadCapture, captured_runtime: NullRunRuntime - ) -> None: - """``include_all=True`` (default): every kwarg lands on the - wire under its own name, regardless of how many args the - function has or what their types are. - """ - - def delete_user(user_id: int, force: bool = False) -> None: - pass - - ext = tool_params(include_all=True) - fn = delete_user - fn._nullrun_extractor = ext - _register_tool(captured_runtime, fn) - - _enforce_sensitive_tool(captured_runtime, fn, (), {"user_id": 42, "force": True}) - - kwargs = captured_payload.last_kwargs - assert kwargs is not None - assert "business_impact" in kwargs - assert "action_digest" in kwargs - - impact = kwargs["business_impact"] - assert impact["kind"] == KIND_TOOL_CALL - assert impact["tool_name"] == "delete_user" - assert impact["params"] == {"user_id": 42, "force": True} - assert impact["extractor_id"] == "nullrun.tool_call.path" - assert impact["extractor_version"] == "1" - - def test_explicit_map_only_captures_listed_kwargs( - self, captured_payload: _PayloadCapture, captured_runtime: NullRunRuntime - ) -> None: - """``{rule_param: arg_name}`` map: only the listed args are - captured; everything else is dropped. The rule param name - (key) and the function arg name (value) may differ. - """ - - def delete_user(uid: int, force: bool = False) -> None: - pass - - ext = tool_params({"user_id": "uid", "force": "force"}) - fn = delete_user - fn._nullrun_extractor = ext - _register_tool(captured_runtime, fn) - - _enforce_sensitive_tool( - captured_runtime, fn, (), {"uid": 42, "force": True, "extra": "ignored"} - ) - - impact = captured_payload.last_kwargs["business_impact"] - # Only the mapped keys land on the wire, with the rule's - # chosen name (user_id, force -- not "uid" or "extra"). - assert impact["params"] == {"user_id": 42, "force": True} - assert "extra" not in impact["params"] - assert "uid" not in impact["params"] - - def test_include_all_false_with_no_map_yields_empty_params( - self, captured_payload: _PayloadCapture, captured_runtime: NullRunRuntime - ) -> None: - """``include_all=False`` and no ``param_extractors``: the - wire shape is ``kind: tool_call`` with ``params: {}``. - Rare, but the documented "empty args" path -- a tool with - no args is still eligible for ToolCall-kind Approval - Rules. - """ - - def list_accounts() -> None: - pass - - # Constructor rejects (param_extractors=None, include_all=False), - # so we build the extractor directly. - ext = ToolParamsExtractor(include_all=False) - fn = list_accounts - fn._nullrun_extractor = ext - _register_tool(captured_runtime, fn) - - _enforce_sensitive_tool(captured_runtime, fn, (), {}) - - impact = captured_payload.last_kwargs["business_impact"] - assert impact["kind"] == KIND_TOOL_CALL - assert impact["tool_name"] == "list_accounts" - assert impact["params"] == {} - - def test_pii_masked_sentinel_is_dropped( - self, captured_payload: _PayloadCapture, captured_runtime: NullRunRuntime - ) -> None: - """PII-masked values (literal ``"***"`` string) are - filtered out before the wire. The operator would never - see the real value, so shipping the sentinel would never - match a real rule -- it's dead weight on the wire. - - The decorator wrapper masks PAN/password values to - ``"***"`` via ``_safe_kwargs`` BEFORE the extractor sees - them; this test simulates that pre-masked state. - """ - - def charge_card(pan: str, amount: int) -> None: - pass - - ext = tool_params(include_all=True) - fn = charge_card - fn._nullrun_extractor = ext - _register_tool(captured_runtime, fn) - - # pan was masked by the decorator's _safe_kwargs layer - # before reaching the extractor. - _enforce_sensitive_tool(captured_runtime, fn, (), {"pan": "***", "amount": 5000}) - - impact = captured_payload.last_kwargs["business_impact"] - assert "pan" not in impact["params"], ( - "PII-masked sentinel '***' leaked to the wire; " - "operators would see a placeholder they cannot match" - ) - # amount is non-PII and survives masking, so it must be on - # the wire. - assert impact["params"]["amount"] == 5000 - - def test_float_arg_is_silently_dropped( - self, captured_payload: _PayloadCapture, captured_runtime: NullRunRuntime - ) -> None: - """Typed-impact filter-not-block: a kwarg whose type is not - JSON-roundtrippable (``float``) is silently dropped from - the wire payload rather than failing the pre-check. - - Why filter rather than fail: a function with a mixed - signature (``set_rate(rate: float, count: int)``) is - still useful -- the ``count`` arg is wire-safe and should - reach the operator. A wholesale block would force every - user with a single ``float`` kwarg to migrate to ``str`` - just to keep their other args matched against rules. - - The strict-mode alternative (explicit ``param_extractors`` - listing only the JSON-safe keys) is documented but - opt-in: bare ``@sensitive`` drops ``float`` silently. - """ - - def set_rate(rate: float, count: int) -> None: - pass - - ext = tool_params(include_all=True) - fn = set_rate - fn._nullrun_extractor = ext - _register_tool(captured_runtime, fn) - - # Must NOT raise: float is filtered, int is captured. - _enforce_sensitive_tool(captured_runtime, fn, (), {"rate": 1.5, "count": 7}) - - impact = captured_payload.last_kwargs["business_impact"] - assert "rate" not in impact["params"], ( - "float value leaked to the wire despite _safe_for_wire filter" - ) - assert impact["params"]["count"] == 7, ( - "JSON-safe kwarg was incorrectly filtered alongside the float" - ) - - def test_unsupported_type_is_silently_filtered_in_explicit_map( - self, captured_payload: _PayloadCapture, captured_runtime: NullRunRuntime - ) -> None: - """``param_extractors`` mode: an explicit {rule: arg} - map whose arg value is JSON-unsafe (``float``) drops - that specific arg silently rather than failing the - whole pre-check. The other args still ship. - - Why filter rather than fail: the explicit map is a - one-to-one rename between function-arg and rule-param. - If a particular rename pair turns out to be - wire-incompatible, the operator can rename the rule - param (``{rate_str: "rate"}``) and stringify the value - before calling the tool. Failing the whole call would - punish the JSON-safe args. - """ - - def set_rate(rate: float, count: int) -> None: - pass - - ext = tool_params({"rate": "rate", "count": "count"}) - fn = set_rate - fn._nullrun_extractor = ext - _register_tool(captured_runtime, fn) - - _enforce_sensitive_tool(captured_runtime, fn, (), {"rate": 1.5, "count": 7}) - - impact = captured_payload.last_kwargs["business_impact"] - # rate was filtered; count survived. - assert "rate" not in impact["params"] - assert impact["params"]["count"] == 7 - - def test_action_digest_matches_sdk_computation( - self, captured_payload: _PayloadCapture, captured_runtime: NullRunRuntime - ) -> None: - """The wire ``action_digest`` MUST equal the SDK's own - computation byte-for-byte. Otherwise the backend's - digest-bound approval row rejects every legitimate - post-approval re-check. - """ - - def delete_user(user_id: int, force: bool = False) -> None: - pass - - ext = tool_params(include_all=True) - fn = delete_user - fn._nullrun_extractor = ext - _register_tool(captured_runtime, fn) - - _enforce_sensitive_tool(captured_runtime, fn, (), {"user_id": 42, "force": True}) - - kwargs = captured_payload.last_kwargs - impact = BusinessImpact.tool_call( - tool_name="delete_user", - params={"user_id": 42, "force": True}, - ) - expected_digest = compute_action_digest(impact) - assert kwargs["action_digest"] == expected_digest - - -# --------------------------------------------------------------------------- -# 3. Auto-attach tests (the production-critical wiring) -# --------------------------------------------------------------------------- - - -class TestAutoAttachOnBareSensitive: - """Verify ``_do_sensitive_register`` stamps a - ``ToolParamsExtractor(include_all=True)`` on every bare - ``@sensitive`` tool that doesn't already carry an explicit - extractor. - - This is the behavior the user asked for: ``@sensitive`` - (no impact=...) is sufficient -- no extra ``@protect`` - decorator change, no extra decorator argument required. - """ - - def test_bare_function_gets_toolparams_extractor( - self, captured_payload: _PayloadCapture, captured_runtime: NullRunRuntime - ) -> None: - """After ``_do_sensitive_register`` runs, a bare function - carries ``_nullrun_extractor = ToolParamsExtractor(...)``, - and the wire payload carries ``kind: tool_call`` with - all kwargs as ``params``. - """ - - def delete_user(user_id: int, force: bool = False) -> None: - pass - - # No pre-existing extractor. - assert getattr(delete_user, "_nullrun_extractor", None) is None - - # Run the registration path that ``@sensitive`` would - # take at decoration time. - _do_sensitive_register(delete_user) - - ext = getattr(delete_user, "_nullrun_extractor") - assert isinstance(ext, ToolParamsExtractor), ( - f"bare @sensitive should auto-attach ToolParamsExtractor, got {type(ext).__name__}" - ) - assert ext.include_all is True - assert ext.param_extractors is None - - # Verify the wire payload uses the auto-attached extractor. - _enforce_sensitive_tool(captured_runtime, delete_user, (), {"user_id": 42, "force": True}) - impact = captured_payload.last_kwargs["business_impact"] - assert impact["kind"] == KIND_TOOL_CALL - assert impact["params"] == {"user_id": 42, "force": True} - - def test_explicit_money_extractor_is_not_overwritten( - self, captured_payload: _PayloadCapture, captured_runtime: NullRunRuntime - ) -> None: - """``@sensitive(impact=money_outflow(...))`` already - stamps ``MoneyImpactExtractor``. The auto-attach path - MUST NOT overwrite the explicit extractor with the - default ``ToolParamsExtractor``. Money semantics win. - """ - - def refund_customer(amount_cents: int, customer_id: str = "c-1") -> None: - pass - - # Stamp the explicit money extractor (what - # ``@sensitive(impact=money_outflow(...))`` does at - # decoration time). - money_ext = money_outflow(argument="amount_cents") - _stamp_extractor_on_innermost(refund_customer, money_ext) - - # Run the registration path -- must NOT replace - # money_ext with a ToolParamsExtractor. - _do_sensitive_register(refund_customer) - - ext = getattr(refund_customer, "_nullrun_extractor") - assert ext is money_ext, ( - "explicit money extractor was overwritten by " - "auto-attach -- this would silently regress the " - "MoneyImpact contract" - ) - assert isinstance(ext, MoneyImpactExtractor) - - # Verify the wire payload still uses money semantics. - _enforce_sensitive_tool(captured_runtime, refund_customer, (5000,), {"customer_id": "c-1"}) - impact = captured_payload.last_kwargs["business_impact"] - assert impact["kind"] == "money" - assert impact["amount_minor"] == 5000 - assert impact["currency"] == "USD" - - -# --------------------------------------------------------------------------- -# 4. Pydantic-style shape pin (pure unit, no runtime) -# --------------------------------------------------------------------------- - - -class TestToolCallParamsShape: - """Pin the SDK-side ``ToolCallParams`` shape against the backend - ``BusinessImpact::ToolCall(ToolCallParams)`` contract at - ``backend/src/proxy/gate/business_impact.rs:62-307``. - - Drift here is a P0 -- the backend would reject every - ToolCall-kind impact on the wire. - """ - - def test_to_wire_dict_shape(self) -> None: - p = ToolCallParams( - tool_name="delete_user", - params={"user_id": 42, "force": True}, - ) - d = p.to_wire_dict() - assert d == { - "kind": KIND_TOOL_CALL, - "tool_name": "delete_user", - "params": {"user_id": 42, "force": True}, - "extractor_id": "nullrun.tool_call.path", - "extractor_version": "1", - } - - def test_validate_rejects_empty_tool_name(self) -> None: - p = ToolCallParams(tool_name="", params={}) - with pytest.raises(ValueError) as exc_info: - p.validate() - assert "non-empty" in str(exc_info.value) - - def test_validate_rejects_overlong_tool_name(self) -> None: - p = ToolCallParams(tool_name="x" * 129, params={}) - with pytest.raises(ValueError) as exc_info: - p.validate() - assert "exceeds max 128" in str(exc_info.value) - - def test_validate_rejects_overlong_param_name(self) -> None: - long_key = "x" * (TOOL_PARAMETERS_MAX_PARAM_NAME + 1) - p = ToolCallParams(tool_name="x", params={long_key: 1}) - with pytest.raises(ValueError) as exc_info: - p.validate() - assert "key length" in str(exc_info.value) - - def test_validate_rejects_float_param_value(self) -> None: - p = ToolCallParams(tool_name="x", params={"rate": 1.5}) - with pytest.raises(ValueError) as exc_info: - p.validate() - assert "float" in str(exc_info.value) - - def test_validate_accepts_all_json_kinds(self) -> None: - # Boundary check: every JSON kind (null/bool/int/str/ - # list/dict) survives validate. Recursive validation - # reaches nested structures. - p = ToolCallParams( - tool_name="x", - params={ - "a": None, - "b": True, - "c": 42, - "d": "hello", - "e": [1, 2, "three", {"nested": True}], - "f": {"deep": {"deeper": [None, False]}}, - }, - ) - # Must not raise. - p.validate() - - def test_business_impact_kind_dispatch(self) -> None: - """``BusinessImpact.kind`` discriminates Money vs ToolCall. - A round-trip through ``to_wire_dict`` must preserve the - kind discriminator so the backend's - ``serde(tag = "kind")`` picks the right variant. - """ - m = BusinessImpact.money("outflow", 1000, "USD") - assert m.kind == "money" - assert m.to_wire_dict()["kind"] == "money" - - t = BusinessImpact.tool_call(tool_name="x", params={"y": 1}) - assert t.kind == KIND_TOOL_CALL - assert t.to_wire_dict()["kind"] == KIND_TOOL_CALL - - -# --------------------------------------------------------------------------- -# 5. Regression: explicit-extractor-vs-auto-attach priority (typed impact + tool-params) -# --------------------------------------------------------------------------- -# -# Bug found via ad-hoc verification after the initial auto-attach -# commit (40d391a): the auto-attach path called -# ``getattr(fn, "_nullrun_extractor", None)`` on the @protect -# wrapper. The explicit extractor (set by ``@sensitive(impact=...)`` -# factory form) lives on the BARE function -- the wrapper does NOT -# carry the attribute -- so the check returned None and the -# auto-attach path silently overwrote the user's explicit map with -# ``ToolParamsExtractor(include_all=True)``. Result: ``impact= -# tool_params({"delete_force": "force"})`` looked like it took -# effect at decoration time but the wire payload used the default -# ``{delete_force: }`` mapping -- silent param-drop. -# -# The fix walks the ``__wrapped__`` chain in -# ``_do_sensitive_register``. The regression tests below pin both -# the bare case and the explicit-map case. - - -class TestAutoAttachChainWalk: - """Pin the ``_do_sensitive_register`` chain walk so future - decorator reordering does not silently regress the - explicit-extractor priority. - """ - - def test_bare_sensitive_chain_walk_attaches_default( - self, captured_runtime: NullRunRuntime - ) -> None: - """Bare ``@sensitive`` (no impact=...) walks the chain - and finds NO extractor, so auto-attach stamps the default. - """ - - def tool_fn(user_id: int) -> None: - pass - - assert _find_extractor_in_chain(tool_fn) is None, ( - "sanity: bare function should not have an extractor" - ) - - _do_sensitive_register(tool_fn) - - ext = _find_extractor_in_chain(tool_fn) - assert ext is not None - assert isinstance(ext, ToolParamsExtractor) - assert ext.include_all is True - - def test_explicit_tool_params_chain_walk_preserves_map( - self, captured_runtime: NullRunRuntime - ) -> None: - """``@sensitive(impact=tool_params({...}))`` stamped on the - bare function MUST survive ``_do_sensitive_register``. - - Pre-fix this silently overwrote the explicit extractor - with the auto-attach default ``ToolParamsExtractor( - include_all=True)`` because ``getattr(wrapper, - "_nullrun_extractor", None)`` returned None even though - the bare function carried the attribute. - """ - - def tool_fn(force: bool) -> None: - pass - - # Simulate the @sensitive(impact=tool_params({...})) - # factory form stamping the explicit extractor on the - # bare function via ``_stamp_extractor_on_innermost``. - explicit = tool_params({"delete_force": "force"}) - _stamp_extractor_on_innermost(tool_fn, explicit) - - # Now run the registration path -- the auto-attach MUST - # see the explicit extractor and skip the default. - _do_sensitive_register(tool_fn) - - ext = _find_extractor_in_chain(tool_fn) - assert ext is explicit, ( - "explicit tool_params map was overwritten by " - "auto-attach default -- this is the regression fixed " - "in the chain-walk patch" - ) - assert ext.param_extractors == {"delete_force": "force"} - assert ext.include_all is True - - def test_explicit_money_outflow_chain_walk_preserved( - self, captured_runtime: NullRunRuntime - ) -> None: - """``@sensitive(impact=money_outflow(...))`` also survives - the auto-attach path (the original money variant must NOT be - overwritten by the tool-parameters auto-attach). - """ - - def tool_fn(amount_cents: int) -> None: - pass - - explicit = money_outflow(argument="amount_cents") - _stamp_extractor_on_innermost(tool_fn, explicit) - - _do_sensitive_register(tool_fn) - - ext = _find_extractor_in_chain(tool_fn) - assert isinstance(ext, MoneyImpactExtractor), ( - f"explicit MoneyImpactExtractor was overwritten by " - f"auto-attach default; got {type(ext).__name__}" - ) - - def test_chain_walk_does_not_loop_on_circular_wraps( - self, captured_runtime: NullRunRuntime - ) -> None: - """Defensive: a pathological ``__wrapped__`` cycle must - not hang ``_find_extractor_in_chain``. We construct a - 3-call cycle and verify the walk returns None within - the bounded hop count. - """ - - # Build a self-referential __wrapped__ cycle. - class Cycle: - def __init__(self) -> None: - self._attr = "marker" - - a = Cycle() - a.__wrapped__ = a # direct self-cycle - # The walk must return None and not hang. - assert _find_extractor_in_chain(a) is None - # And a longer cycle: a -> b -> a -> b ... - b = Cycle() - a.__wrapped__ = b - b.__wrapped__ = a - assert _find_extractor_in_chain(a) is None diff --git a/tests/test_toolbox_langgraph.py b/tests/test_toolbox_langgraph.py deleted file mode 100644 index 6c014da..0000000 --- a/tests/test_toolbox_langgraph.py +++ /dev/null @@ -1,110 +0,0 @@ -"""Tests for the toolbox.langgraph.wrapper helper. - -The wrapper monkey-patches a compiled LangGraph app's -`invoke` and `stream` methods to attach a `NullRunCallback` so -the runtime sees LLM usage. These tests verify the wiring -without requiring an actual LangChain/LangGraph runtime — -we just need a duck-typed object with `.invoke` and `.stream`. -""" - -import pytest - -from nullrun.instrumentation.langgraph import NullRunCallback -from nullrun.runtime import NullRunRuntime -from nullrun.toolbox.langgraph import wrapper - - -@pytest.fixture(autouse=True) -def _test_runtime(monkeypatch, tmp_path): - """Provide a runtime in test mode so get_runtime returns without - authenticating against a real server. - - Pins ``NULLRUN_WAL_PATH`` to a tmp_path-scoped file so the - constructor's ``Transport._replay_from_wal`` never picks up - a stale WAL left over from a previous test run (which would - replay real events to a live API and cause HTTP 401 in - setup). Mirrors ``conftest::make_test_runtime``. - """ - monkeypatch.setenv("NULLRUN_API_KEY", "test-key-12345678") - monkeypatch.setenv("NULLRUN_WAL_PATH", str(tmp_path / "sdk.wal")) - NullRunRuntime.reset_instance() - # Pre-build a test-mode singleton so get_runtime returns it without - # hitting the network. Construct directly and store on the singleton - # slot so subsequent get_instance calls return it. - rt = NullRunRuntime(api_key="test-key-12345678", _test_mode=True) - NullRunRuntime._instance = rt - yield - NullRunRuntime.reset_instance() - - -class _FakeApp: - """Minimal compiled-LangGraph duck type: .invoke and .stream.""" - - def __init__(self) -> None: - self.invocations: list[dict] = [] - self.stream_calls: list[dict] = [] - - def invoke(self, input, config=None, **kwargs): - self.invocations.append({"input": input, "config": config, "kwargs": kwargs}) - # Echo the callbacks list so the test can inspect what wrapper added. - return {"callbacks": (config or {}).get("callbacks", [])} - - def stream(self, input, config=None, **kwargs): - self.stream_calls.append({"input": input, "config": config, "kwargs": kwargs}) - yield {"callbacks": (config or {}).get("callbacks", [])} - - -def test_wrapper_returns_app(): - """wrapper() must return the same app object (mutated in place).""" - app = _FakeApp() - out = wrapper(app) - assert out is app - - -def test_wrapper_attaches_callback_to_invoke(): - """invoke() must have a NullRunCallback appended to config['callbacks'].""" - app = _FakeApp() - wrapper(app) - app.invoke({"x": 1}) - callbacks = app.invocations[0]["config"]["callbacks"] - assert any(isinstance(c, NullRunCallback) for c in callbacks) - - -def test_wrapper_attaches_callback_to_stream(): - """stream() must also get a NullRunCallback in config['callbacks'].""" - app = _FakeApp() - wrapper(app) - list(app.stream({"x": 1})) - callbacks = app.stream_calls[0]["config"]["callbacks"] - assert any(isinstance(c, NullRunCallback) for c in callbacks) - - -def test_wrapper_preserves_user_callbacks(): - """If the caller already supplied callbacks, wrapper appends to them.""" - app = _FakeApp() - wrapper(app) - user_cb = object() - app.invoke({"x": 1}, config={"callbacks": [user_cb]}) - callbacks = app.invocations[0]["config"]["callbacks"] - assert user_cb in callbacks - assert any(isinstance(c, NullRunCallback) for c in callbacks) - - -def test_wrapper_handles_no_config_arg(): - """invoke(input) without a config kwarg must still get a callbacks list.""" - app = _FakeApp() - wrapper(app) - app.invoke({"x": 1}) - config = app.invocations[0]["config"] - assert config is not None - assert "callbacks" in config - - -def test_old_instrument_path_is_removed(): - """`nullrun.instrumentation.langgraph.instrument` no longer exists.""" - import nullrun.instrumentation.langgraph as mod - - assert not hasattr(mod, "instrument"), ( - "`instrument` should be removed; " - "use `nullrun.toolbox.langgraph.wrapper` instead." - ) diff --git a/tests/test_track_span_context.py b/tests/test_track_span_context.py index bafc578..4ccf748 100644 --- a/tests/test_track_span_context.py +++ b/tests/test_track_span_context.py @@ -209,72 +209,39 @@ def test_track_tool_is_retry_flag(capturing_runtime): assert event["latency_ms"] == 500 -# ────────────────────────────────────────────────────────────── -# Module-level track_llm / track_tool -# ────────────────────────────────────────────────────────────── - - -def test_module_level_track_llm_attaches_span(capturing_runtime, monkeypatch): - """The module-level `nullrun.track_llm` should also pick up the - active span — it forwards to the runtime method, which is where - the span attachment lives.""" - from nullrun import runtime as runtime_mod - - # Replace the runtime getter with our capturing wrapper so module- - # level calls land in the same buffer as the method-level ones. - monkeypatch.setattr(runtime_mod, "get_runtime", lambda: capturing_runtime.runtime) - - span = create_root_span() - token = set_span(span) - try: - runtime_mod.track_llm(input_tokens=7, output_tokens=3) - finally: - reset_span(token) - - event = capturing_runtime.events[0] - assert event["trace_id"] == span.trace_id - assert event["span_id"] == span.span_id - - -def test_module_level_track_llm_output_tokens_optional(mock_api): - """Calling `nullrun.track_llm(input_tokens=N)` with no output_tokens - must not TypeError — the kwarg now defaults to 0. - - Depends on `mock_api` so respx covers `/track/batch`. We also call - `nullrun.init(...)` so whatever singleton the module-level - `track_llm` resolves points at the mocked URL — without this, a - stale singleton from a previous test (or a fresh one built from - env defaults) targets the prod URL and respx raises - AllMockedAssertionError.""" - import nullrun - from tests.conftest import BASE_URL - - nullrun.init(api_key="test-key-12345678", api_url=BASE_URL) - nullrun.track_llm(input_tokens=42) # smoke test — no exception - - # ────────────────────────────────────────────────────────────── # End-to-end with @protect # ────────────────────────────────────────────────────────────── -def test_protect_then_track_llm_attaches_to_protect_span(capturing_runtime, monkeypatch): - """The integration story: @protect opens a span, a track_llm - inside it inherits that span — no manual plumbing needed.""" +def test_protect_then_track_llm_attaches_to_protect_span(capturing_runtime): + """The integration story: @protect opens a span, an LLM-call + event inside it inherits that span — no manual plumbing needed. + + In 0.18.2 ``@protect`` is the only public entry point for user + code; the module-level ``nullrun.track_llm`` was removed because + every observable side-effect (LLM call, tool call, etc.) is now + a child span under ``@protect``. Auto-instrumentation + (``nullrun.instrumentation.auto``) calls the runtime instance + method directly when it ships an LLM event, so we exercise the + same code path here.""" import nullrun import nullrun.decorators as dec - from nullrun import runtime as runtime_mod from nullrun.decorators import reset as reset_decorator_runtime - # Wire both: the @protect emit path (uses dec._runtime) AND the - # module-level nullrun.track_llm path (uses runtime_mod.get_runtime). + # Wire the @protect emit path so the capturing runtime observes + # span_start / span_end + the inner LLM call. dec._runtime = capturing_runtime.runtime - monkeypatch.setattr(runtime_mod, "get_runtime", lambda: capturing_runtime.runtime) try: @nullrun.protect def agent(q): - nullrun.track_llm(input_tokens=20, output_tokens=10, model="gpt-4o") + # In production this would be the auto-instrumented + # httpx call into the LLM SDK — the patcher reaches the + # runtime via the same instance method. + capturing_runtime.runtime.track_llm( + input_tokens=20, output_tokens=10, model="gpt-4o", + ) return "ok" agent("hi") diff --git a/tests/test_transport.py b/tests/test_transport.py index 0a88476..0cba19c 100644 --- a/tests/test_transport.py +++ b/tests/test_transport.py @@ -213,7 +213,7 @@ def test_execute_fallback_strict_blocks_on_gateway_error(self, transport): trace_id="trace-789", tool="my.tool", input_data={}, - fallback_mode="strict", + fallback_mode=FallbackMode.STRICT, ) assert result["decision"] == "block" assert result["decision_source"] == "fallback" @@ -230,14 +230,14 @@ def test_execute_fallback_permissive_allows_on_gateway_error(self, transport): trace_id="trace-789", tool="my.tool", input_data={}, - fallback_mode="permissive", + fallback_mode=FallbackMode.PERMISSIVE, ) assert result["decision"] == "allow" assert result["decision_source"] == "fallback" @respx.mock - def test_execute_fallback_cached_degrades_to_permissive(self, transport): - """0.7.0: CACHED fallback mode degrades to PERMISSIVE (no local cache).""" + def test_execute_fallback_permissive_allows_on_execute_error(self, transport): + """PERMISSIVE fallback mode allows when /execute is unavailable.""" respx.post("https://api.test.nullrun.io/api/v1/execute").mock( return_value=httpx.Response(500, text="Server Error") ) @@ -247,10 +247,10 @@ def test_execute_fallback_cached_degrades_to_permissive(self, transport): trace_id="trace-789", tool="my.tool", input_data={}, - fallback_mode="cached", + fallback_mode=FallbackMode.PERMISSIVE, ) - # 0.7.0: thin client — no local cache to consult on gateway - # failure. CACHED silently degrades to PERMISSIVE. + # The SDK is a thin client — no local cache to consult on + # gateway failure. PERMISSIVE allows the request. assert result["decision"] == "allow" assert result["decision_source"] == "fallback" @@ -604,49 +604,6 @@ def test_transport_stopped_flag(self, transport): # ────────────────────────────────────────────────────────────── -class TestSensitiveToolsAPI: - def test_add_sensitive_tool(self, make_runtime): - """add_sensitive_tool marks a tool as sensitive.""" - rt = make_runtime() - rt.add_sensitive_tool("my.custom_tool") - assert "my.custom_tool" in rt.get_sensitive_tools() - - def test_remove_sensitive_tool(self, make_runtime): - """remove_sensitive_tool unmarks a tool as sensitive.""" - rt = make_runtime() - rt.add_sensitive_tool("my.custom_tool") - rt.remove_sensitive_tool("my.custom_tool") - assert "my.custom_tool" not in rt.get_sensitive_tools() - - def test_register_sensitive_tools_batch(self, make_runtime): - """register_sensitive_tools adds multiple tools at once.""" - rt = make_runtime() - rt.register_sensitive_tools(["tool1", "tool2", "tool3"]) - tools = rt.get_sensitive_tools() - assert "tool1" in tools - assert "tool2" in tools - assert "tool3" in tools - - def test_sensitive_tools_default_set(self, make_runtime): - """Default sensitive tools include dangerous operations.""" - rt = make_runtime() - # Built-in sensitive tools - assert "stripe.charge" in rt.get_sensitive_tools() - assert "db.delete" in rt.get_sensitive_tools() - assert "file.delete" in rt.get_sensitive_tools() - - def test_is_sensitive_tool(self, make_runtime): - """is_sensitive_tool returns True for sensitive tools.""" - rt = make_runtime() - rt.add_sensitive_tool("my.sensitive_tool") - assert rt.is_sensitive_tool("my.sensitive_tool") is True - assert rt.is_sensitive_tool("my.normal_tool") is False - - -# ────────────────────────────────────────────────────────────── -# HMAC signature tests -# ────────────────────────────────────────────────────────────── - class TestTransportHMAC: def test_generate_hmac_signature(self): @@ -1085,6 +1042,7 @@ def capture(request: httpx.Request) -> httpx.Response: TransportErrorSource, ) from nullrun.transport import ( + FallbackMode, FlushConfig, _parse_error_envelope, verify_hmac_signature, @@ -1333,7 +1291,7 @@ def test_execute_fallback_strict_returns_block(): trace_id="t-1", tool="x", input_data={}, - fallback_mode="strict", + fallback_mode=FallbackMode.STRICT, ) assert result["decision"] == "block" assert "STRICT" in result["explanation"] @@ -1344,8 +1302,8 @@ def test_execute_fallback_strict_returns_block(): # gateway failure. CACHED now degrades to PERMISSIVE. -def test_execute_fallback_cached_degrades_to_permissive(): - """fallback_mode=CACHED → degrade to PERMISSIVE (no local cache).""" +def test_execute_fallback_permissive_allows_on_transport_error(): + """fallback_mode=PERMISSIVE → synthetic allow on transport failure.""" from nullrun.breaker.exceptions import BreakerTransportError t = _build_transport() @@ -1356,11 +1314,10 @@ def test_execute_fallback_cached_degrades_to_permissive(): trace_id="t-1", tool="x", input_data={}, - fallback_mode="cached", + fallback_mode=FallbackMode.PERMISSIVE, ) - # 0.7.0: CACHED silently degrades to PERMISSIVE (allow). assert result["decision"] == "allow" - assert result["decision_source"] == "fallback" + assert "PERMISSIVE" in result["explanation"] def test_execute_fallback_permissive_default(): diff --git a/tests/test_transport_branches.py b/tests/test_transport_branches.py index f5dfd4a..d8e4d47 100644 --- a/tests/test_transport_branches.py +++ b/tests/test_transport_branches.py @@ -28,6 +28,7 @@ TransportErrorSource, ) from nullrun.transport import ( + FallbackMode, FlushConfig, Transport, _parse_error_envelope, @@ -275,19 +276,14 @@ def test_execute_fallback_strict_returns_block(): trace_id="t-1", tool="x", input_data={}, - fallback_mode="strict", + fallback_mode=FallbackMode.STRICT, ) assert result["decision"] == "block" assert "STRICT" in result["explanation"] -# 0.7.0: fallback_mode=CACHED + the local PolicyCache path were -# removed. The thin-client SDK has no local cache to consult on -# gateway failure. CACHED now degrades to PERMISSIVE. - - -def test_execute_fallback_cached_degrades_to_permissive(): - """fallback_mode=CACHED → degrade to PERMISSIVE (no local cache).""" +def test_execute_fallback_permissive_allows_on_transport_error(): + """fallback_mode=PERMISSIVE → synthetic allow on transport failure.""" from nullrun.breaker.exceptions import BreakerTransportError t = _build_transport() @@ -298,9 +294,8 @@ def test_execute_fallback_cached_degrades_to_permissive(): trace_id="t-1", tool="x", input_data={}, - fallback_mode="cached", + fallback_mode=FallbackMode.PERMISSIVE, ) - # 0.7.0: CACHED silently degrades to PERMISSIVE (allow). assert result["decision"] == "allow" assert result["decision_source"] == "fallback" diff --git a/tests/test_typed_exceptions_full_audit.py b/tests/test_typed_exceptions_full_audit.py index bca7c54..35f9239 100644 --- a/tests/test_typed_exceptions_full_audit.py +++ b/tests/test_typed_exceptions_full_audit.py @@ -40,7 +40,6 @@ NullRunBlockedException, NullRunBudgetError, NullRunWorkflowKilledError, - WorkflowKilledException, WorkflowKilledInterrupt, ) @@ -290,10 +289,6 @@ def test_workflow_killed_interrupt_is_now_exception_subclass(self): # recovery requires catching the kill signal. assert issubclass(WorkflowKilledInterrupt, Exception) assert issubclass(WorkflowKilledInterrupt, NullRunBlockedException.__mro__[-2]) # Exception via NullRunError - # The documented BREAK: WorkflowKilledException (the - # deprecated BaseException parent) no longer matches. - # Code that catches the deprecated name must migrate. - assert not issubclass(WorkflowKilledInterrupt, WorkflowKilledException) def test_old_except_clauses_still_catch_kill(self): # Back-compat: cookbook code that does `except diff --git a/tests/test_units_discriminator.py b/tests/test_units_discriminator.py deleted file mode 100644 index 3286b3e..0000000 --- a/tests/test_units_discriminator.py +++ /dev/null @@ -1,495 +0,0 @@ -"""Decimal support follow-up: explicit units discriminator + Decimal support. - -These tests pin the behavior the previous review explicitly -called out: the unit semantics (major / minor) must be -**explicit** in the decorator, not implicit from the value -type. ``float`` is rejected outright because the entire point -of the ``Decimal``-first path is to avoid binary-floating-point -surprises in money code. - -Why this lives in a dedicated file (not as another case in -``tests/test_business_impact.py``): the unit-discriminator -matrix has eight cases (two unit values x four value types -x the two rejection paths) and the existing -``TestExtractorArgumentLookup`` class is about -positional/keyword lookup, not unit semantics. A focused -test class keeps the failure messages close to the failure -mode. - -The cross-language golden hex pin (``dfc96387ca539b7130caebe705e042f2e34e52ab44352ae5e527bcef64f0df27``) -is unchanged: the canonical wire format is still minor units -(``amount_minor=5000`` for $50.00), regardless of which -``units`` the operator chose. The SDK converts in -``_to_minor_units`` before reaching ``BusinessImpact``, so the -wire shape is identical between the two paths. -""" - -from __future__ import annotations - -from decimal import Decimal - -import pytest - -from nullrun.business_impact import INFLOW, OUTFLOW, BusinessImpact -from nullrun.extractor import ( - UNIT_MAJOR, - UNIT_MINOR, - _to_minor_units, - money_outflow, -) - -# Golden cross-language pin shared with the backend's golden -# test (and pinned in tests/test_business_impact.py). The wire -# shape is in minor units regardless of the SDK's units -# discriminator. -GOLDEN_HEX_USD_50_DOLLARS_OUTFLOW = ( - "dfc96387ca539b7130caebe705e042f2e34e52ab44352ae5e527bcef64f0df27" -) - - -def _refund_dollars(amount: Decimal) -> dict: - return {"amount": amount} - - -def _refund_cents(amount_cents: int) -> dict: - return {"amount": amount_cents} - - -# --------------------------------------------------------------------------- -# 1. units="major" -- Decimal -> minor units conversion -# --------------------------------------------------------------------------- - - -class TestMajorUnitsDecimalConversion: - """``units='major'`` accepts ``Decimal`` and multiplies by 100 - with banker's rounding. ``float`` and ``int`` are rejected - outright.""" - - def test_decimal_50_99_minor_5099(self) -> None: - ext = money_outflow( - argument="amount", - currency="USD", - units=UNIT_MAJOR, - ) - impact = ext.impact_for(_refund_dollars, (Decimal("50.99"),), {}) - # The wire stores minor units (cents) regardless of the - # input unit. - assert impact.impact.amount_minor == 5_099 - - def test_decimal_50_minor_5000(self) -> None: - ext = money_outflow( - argument="amount", - currency="USD", - units=UNIT_MAJOR, - ) - impact = ext.impact_for(_refund_dollars, (Decimal("50"),), {}) - assert impact.impact.amount_minor == 5_000 - - def test_decimal_50_005_rejected_for_usd(self) -> None: - # Production-grade contract: precision must be - # supplied correctly by the caller. ``Decimal("50.005")`` - # is a sub-cent precision that USD does not support, so - # the SDK raises ``ValueError`` rather than silently - # rounding (no banker's rounding; no ROUND_HALF_UP; the - # previous "drop half-cent silently" behaviour is the - # exact bug class this contract prevents). - ext = money_outflow( - argument="amount", - currency="USD", - units=UNIT_MAJOR, - ) - with pytest.raises(ValueError, match="USD supports at most 2"): - ext.impact_for(_refund_dollars, (Decimal("50.005"),), {}) - - def test_decimal_50_999_rejected_for_usd(self) -> None: - ext = money_outflow( - argument="amount", - currency="USD", - units=UNIT_MAJOR, - ) - with pytest.raises(ValueError, match="USD supports at most 2"): - ext.impact_for(_refund_dollars, (Decimal("50.999"),), {}) - - def test_decimal_0_005_rejected_for_usd(self) -> None: - # Sub-cent precision for any USD amount is rejected. - ext = money_outflow( - argument="amount", - currency="USD", - units=UNIT_MAJOR, - ) - with pytest.raises(ValueError, match="USD supports at most 2"): - ext.impact_for(_refund_dollars, (Decimal("0.005"),), {}) - - def test_decimal_0_01_accepted_for_usd(self) -> None: - # The boundary: 0.01 has exactly 2 fractional digits. - ext = money_outflow( - argument="amount", - currency="USD", - units=UNIT_MAJOR, - ) - impact = ext.impact_for(_refund_dollars, (Decimal("0.01"),), {}) - assert impact.impact.amount_minor == 1 - - def test_jpy_decimal_with_fractional_digits_rejected(self) -> None: - # JPY has 0 fractional digits (yen). ``Decimal("100.5")`` - # is rejected because the caller has supplied sub-yen - # precision. - def _refund_jpy(amount: Decimal) -> dict: - return {"a": amount} - - ext = money_outflow( - argument="amount", - currency="JPY", - units=UNIT_MAJOR, - ) - with pytest.raises(ValueError, match="JPY supports at most 0"): - ext.impact_for(_refund_jpy, (Decimal("100.5"),), {}) - - def test_jpy_decimal_integer_accepted(self) -> None: - def _refund_jpy(amount: Decimal) -> dict: - return {"a": amount} - - ext = money_outflow( - argument="amount", - currency="JPY", - units=UNIT_MAJOR, - ) - impact = ext.impact_for(_refund_jpy, (Decimal("1000"),), {}) - # JPY has 0 fractional digits, so the wire stores the - # same integer; no conversion needed. - assert impact.impact.amount_minor == 1000 - - def test_kwd_three_fractional_digits_accepted(self) -> None: - # KWD has 3 fractional digits (fils). ``Decimal("1.234")`` - # is exactly within the supported precision. - def _refund_kwd(amount: Decimal) -> dict: - return {"a": amount} - - ext = money_outflow( - argument="amount", - currency="KWD", - units=UNIT_MAJOR, - ) - impact = ext.impact_for(_refund_kwd, (Decimal("1.234"),), {}) - assert impact.impact.amount_minor == 1_234 - - def test_kwd_four_fractional_digits_rejected(self) -> None: - # ``Decimal("1.2345")`` is sub-fil precision for KWD. - def _refund_kwd(amount: Decimal) -> dict: - return {"a": amount} - - ext = money_outflow( - argument="amount", - currency="KWD", - units=UNIT_MAJOR, - ) - with pytest.raises(ValueError, match="KWD supports at most 3"): - ext.impact_for(_refund_kwd, (Decimal("1.2345"),), {}) - - def test_int_rejected_in_major_units(self) -> None: - # A bare int in major units is the silent bug class - # the explicit discriminator is designed to prevent. - ext = money_outflow( - argument="amount", - currency="USD", - units=UNIT_MAJOR, - ) - with pytest.raises(TypeError, match="requires Decimal"): - ext.impact_for(_refund_dollars, (50,), {}) - - def test_float_rejected_outright(self) -> None: - # ``float`` is the entire reason ``Decimal`` exists. - ext = money_outflow( - argument="amount", - currency="USD", - units=UNIT_MAJOR, - ) - with pytest.raises(TypeError, match="requires Decimal"): - ext.impact_for(_refund_dollars, (50.99,), {}) - - def test_bool_rejected_in_major_units(self) -> None: - # ``bool`` is a subclass of ``int`` in Python; the - # explicit check rejects it so a hostile caller can't - # smuggle ``True`` as ``amount=1`` cent. - ext = money_outflow( - argument="amount", - currency="USD", - units=UNIT_MAJOR, - ) - with pytest.raises(TypeError, match="requires Decimal"): - ext.impact_for(_refund_dollars, (True,), {}) - - -# --------------------------------------------------------------------------- -# 2. units="minor" -- int passes through, Decimal needs quantization -# --------------------------------------------------------------------------- - - -class TestMinorUnitsIntAndDecimal: - """``units='minor'`` accepts ``int`` (canonical) and ``Decimal`` - if it is already integer-valued. ``float`` is rejected.""" - - def test_int_50_minor_50(self) -> None: - ext = money_outflow( - argument="amount_cents", - currency="USD", - units=UNIT_MINOR, - ) - impact = ext.impact_for(_refund_cents, (50,), {}) - assert impact.impact.amount_minor == 50 - - def test_decimal_50_minor_50(self) -> None: - # The caller has already pre-quantized; the SDK does not - # change the value. This path supports legacy code that - # had been using ``Decimal("50.00")`` everywhere and - # later adopts the decorator. - ext = money_outflow( - argument="amount_cents", - currency="USD", - units=UNIT_MINOR, - ) - impact = ext.impact_for(_refund_cents, (Decimal("50"),), {}) - assert impact.impact.amount_minor == 50 - - def test_decimal_50_00_minor_50(self) -> None: - ext = money_outflow( - argument="amount_cents", - currency="USD", - units=UNIT_MINOR, - ) - impact = ext.impact_for(_refund_cents, (Decimal("50.00"),), {}) - assert impact.impact.amount_minor == 50 - - def test_decimal_with_fractional_part_rejected_in_minor(self) -> None: - # ``Decimal("0.05")`` with units="minor" is a unit- - # confusion bug (the caller is passing major units - # under a minor decorator). The SDK surfaces a - # TypeError pointing the operator at the right - # alternative. - ext = money_outflow( - argument="amount_cents", - currency="USD", - units=UNIT_MINOR, - ) - with pytest.raises(TypeError, match="refusing to round"): - ext.impact_for(_refund_cents, (Decimal("0.05"),), {}) - - def test_float_rejected_in_minor_units(self) -> None: - ext = money_outflow( - argument="amount_cents", - currency="USD", - units=UNIT_MINOR, - ) - with pytest.raises(TypeError, match="requires int or Decimal"): - ext.impact_for(_refund_cents, (50.99,), {}) - - def test_str_rejected_in_minor_units(self) -> None: - ext = money_outflow( - argument="amount_cents", - currency="USD", - units=UNIT_MINOR, - ) - with pytest.raises(TypeError, match="requires int or Decimal"): - ext.impact_for(_refund_cents, ("50",), {}) - - def test_bool_rejected_in_minor_units(self) -> None: - ext = money_outflow( - argument="amount_cents", - currency="USD", - units=UNIT_MINOR, - ) - with pytest.raises(TypeError, match="requires int or Decimal"): - ext.impact_for(_refund_cents, (True,), {}) - - -# --------------------------------------------------------------------------- -# 3. The cross-language golden hex pin survives the new path -# --------------------------------------------------------------------------- - - -class TestGoldenHexSurvivesNewPath: - """The wire shape is in minor units regardless of the SDK's - units discriminator. The cross-language golden hex must - match whether the operator passed ``int(5000)``, - ``Decimal('50')``, or ``Decimal('50.00')``.""" - - def test_minor_int_5000_matches_golden(self) -> None: - ext = money_outflow( - argument="amount_cents", - currency="USD", - units=UNIT_MINOR, - ) - impact = ext.impact_for(_refund_cents, (5_000,), {}) - # The canonical wire form is identical to the legacy - # pre-Decimal path. The golden hex is the SAME on the - # backend side (see ``business_impact.rs::tests:: - # action_digest_golden_usd_outflow_5000_cents``). - from nullrun.business_impact import compute_action_digest - assert compute_action_digest(impact) == GOLDEN_HEX_USD_50_DOLLARS_OUTFLOW - - def test_major_decimal_50_matches_golden(self) -> None: - # The operator writes ``Decimal("50.00")`` in major - # units; the SDK converts to 5000 minor units; the - # digest is byte-identical to the int(5000) path - # above. - ext = money_outflow( - argument="amount", - currency="USD", - units=UNIT_MAJOR, - ) - impact = ext.impact_for(_refund_dollars, (Decimal("50.00"),), {}) - from nullrun.business_impact import compute_action_digest - assert compute_action_digest(impact) == GOLDEN_HEX_USD_50_DOLLARS_OUTFLOW - - -# --------------------------------------------------------------------------- -# 4. The unit discriminator is a constructor argument, not a type -# --------------------------------------------------------------------------- - - -class TestUnitDiscriminatorIsExplicit: - """A signature refactor (``int`` -> ``Decimal`` or vice - versa) does NOT silently flip the unit semantics. The - operator must pass ``units='major'`` to opt into Decimal - conversion.""" - - def test_int_in_decimal_typed_arg_with_minor_units_passes_through(self) -> None: - # Function declares ``amount: Decimal`` but the - # decorator is configured with ``units="minor"``. - # The int(50) value passes through verbatim because - # the operator explicitly opted into minor units. - # The amount is 50 minor = $0.50, NOT $50.00. - def _dec(amount: Decimal) -> dict: - return {"a": amount} - - ext = money_outflow( - argument="amount", - currency="USD", - units=UNIT_MINOR, - ) - impact = ext.impact_for(_dec, (50,), {}) - assert impact.impact.amount_minor == 50 - - def test_decimal_in_int_typed_arg_with_major_units_converts(self) -> None: - # Function declares ``amount_cents: int`` but the - # operator passes ``Decimal("50.00")`` with - # ``units="major"``. The SDK converts 50.00 to 5000 - # minor units. The type annotation is overridden by - # the explicit unit discriminator. - def _int(amount_cents: int) -> dict: - return {"a": amount_cents} - - ext = money_outflow( - argument="amount_cents", - currency="USD", - units=UNIT_MAJOR, - ) - impact = ext.impact_for(_int, (Decimal("50.00"),), {}) - assert impact.impact.amount_minor == 5_000 - - def test_unknown_units_rejected_at_construction(self) -> None: - with pytest.raises(ValueError, match="units must be one of"): - money_outflow( - argument="amount", - currency="USD", - units="micros", - ) - - -# --------------------------------------------------------------------------- -# 5. Direct unit test for ``_to_minor_units`` -# --------------------------------------------------------------------------- - - -class TestToMinorUnitsHelper: - """``_to_minor_units`` is the conversion primitive. These - tests pin the behaviour independent of the ``MoneyImpact`` - struct so a future refactor of the impact struct does not - silently change the conversion semantics.""" - - def test_minor_int_passes_through(self) -> None: - assert _to_minor_units(50, UNIT_MINOR, "USD") == 50 - assert _to_minor_units(0, UNIT_MINOR, "USD") == 0 - assert _to_minor_units(1_000_000, UNIT_MINOR, "USD") == 1_000_000 - - def test_minor_decimal_integer_passes_through(self) -> None: - assert _to_minor_units(Decimal("50"), UNIT_MINOR, "USD") == 50 - assert _to_minor_units(Decimal("50.00"), UNIT_MINOR, "USD") == 50 - - def test_major_decimal_multiplied_by_100(self) -> None: - assert _to_minor_units(Decimal("50"), UNIT_MAJOR, "USD") == 5_000 - assert _to_minor_units(Decimal("50.99"), UNIT_MAJOR, "USD") == 5_099 - assert _to_minor_units(Decimal("1000.00"), UNIT_MAJOR, "USD") == 100_000 - - def test_major_decimal_rejects_sub_cent_precision(self) -> None: - # ``Decimal("0.005")`` for USD has 3 fractional digits - # but USD supports 2; the helper raises ``ValueError`` - # rather than silently rounding. This is the - # production-grade contract that replaced banker's - # rounding. - with pytest.raises(ValueError, match="USD supports at most 2"): - _to_minor_units(Decimal("0.005"), UNIT_MAJOR, "USD") - with pytest.raises(ValueError, match="USD supports at most 2"): - _to_minor_units(Decimal("50.005"), UNIT_MAJOR, "USD") - with pytest.raises(ValueError, match="USD supports at most 2"): - _to_minor_units(Decimal("0.999"), UNIT_MAJOR, "USD") - - def test_major_decimal_rejects_sub_yen_precision(self) -> None: - with pytest.raises(ValueError, match="JPY supports at most 0"): - _to_minor_units(Decimal("100.5"), UNIT_MAJOR, "JPY") - - def test_major_decimal_accepts_three_digit_kwd(self) -> None: - # KWD has 3 fractional digits; ``Decimal("1.234")`` is - # accepted. - assert _to_minor_units(Decimal("1.234"), UNIT_MAJOR, "KWD") == 1_234 - - def test_major_decimal_rejects_four_digit_kwd(self) -> None: - with pytest.raises(ValueError, match="KWD supports at most 3"): - _to_minor_units(Decimal("1.2345"), UNIT_MAJOR, "KWD") - - def test_major_rejects_int(self) -> None: - with pytest.raises(TypeError, match="requires Decimal"): - _to_minor_units(50, UNIT_MAJOR, "USD") - - def test_minor_rejects_float(self) -> None: - with pytest.raises(TypeError, match="requires int or Decimal"): - _to_minor_units(50.99, UNIT_MINOR, "USD") - - def test_minor_rejects_fractional_decimal(self) -> None: - with pytest.raises(TypeError, match="refusing to round"): - _to_minor_units(Decimal("0.05"), UNIT_MINOR, "USD") - - def test_rejects_bool_everywhere(self) -> None: - with pytest.raises(TypeError, match="requires int or Decimal"): - _to_minor_units(True, UNIT_MINOR, "USD") - with pytest.raises(TypeError, match="requires Decimal"): - _to_minor_units(True, UNIT_MAJOR, "USD") - - def test_unknown_units_is_defensive_branch(self) -> None: - # ``__init__`` validates ``units`` at construction time, - # so this branch is unreachable from the public API. - # We test it directly to lock the safety net. - with pytest.raises(ValueError, match="unknown units"): - _to_minor_units(50, "micros", "USD") - - -# --------------------------------------------------------------------------- -# 6. The ``BusinessImpact`` direction is unaffected -# --------------------------------------------------------------------------- - - -class TestDirectionIsUnaffected: - """``units`` does not interact with ``direction`` (outflow / - inflow). The default direction is OUTFLOW, matching the - pre-Decimal path.""" - - def test_major_units_default_direction_is_outflow(self) -> None: - ext = money_outflow( - argument="amount", - currency="USD", - units=UNIT_MAJOR, - ) - impact = ext.impact_for(_refund_dollars, (Decimal("50.00"),), {}) - assert impact.impact.direction == OUTFLOW - assert impact.impact.currency == "USD" - assert impact.impact.amount_minor == 5_000 \ No newline at end of file From 15e4fad5c5664bab7d8a5d6617e627a910dad052 Mon Sep 17 00:00:00 2001 From: Anatoly Maltsev Date: Fri, 25 Sep 2026 12:47:59 +0400 Subject: [PATCH 02/10] docs(runtime): strip legacy/pre-fix/post-fix framing MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Remove "pre-fix", "post-fix", "legacy", "back-compat", "previously did", "was removed", "v3.53 audit" framing from inline comments and docstrings. - Keep all audit tracking IDs (DEF-*, AUTH-01, ADR-*) — those refer to external systems and are not legacy framing per se. - runtime.py -565 lines net (-311 lines of redundant context). 17 files modified; no functional changes. - pytest 1405 passed (1 skipped) with -W error::DeprecationWarning. - Prod smoke @protect → /execute round-trip ALLOW in 302ms. --- src/nullrun/__init__.py | 23 +- src/nullrun/_handle.py | 4 +- src/nullrun/_registry.py | 5 +- src/nullrun/actions.py | 14 +- src/nullrun/audit.py | 10 +- src/nullrun/business_impact.py | 13 +- src/nullrun/capabilities.py | 9 +- src/nullrun/decorators.py | 24 +- src/nullrun/instrumentation/__init__.py | 4 +- src/nullrun/instrumentation/_safe_patch.py | 10 +- src/nullrun/instrumentation/auto.py | 13 +- src/nullrun/observability/__init__.py | 7 +- src/nullrun/runtime.py | 613 +++++++-------------- src/nullrun/toolbox/langgraph.py | 3 +- src/nullrun/toolbox/mcp.py | 23 +- src/nullrun/transport.py | 78 ++- src/nullrun/transport_websocket.py | 23 +- 17 files changed, 311 insertions(+), 565 deletions(-) diff --git a/src/nullrun/__init__.py b/src/nullrun/__init__.py index 16027de..01d96e1 100644 --- a/src/nullrun/__init__.py +++ b/src/nullrun/__init__.py @@ -68,12 +68,12 @@ def shutdown(timeout: float = 2.0, flush: bool = True) -> None: this returns, any further ``nullrun.track(...)`` call or ``@protect``-decorated call is a no-op. - Audit 2026-06-29 (WS graceful close on exit): a long-running - script that exits via ``sys.exit `` lets the kernel RST the TCP - socket, which the backend logs as WARN "Connection reset - without closing handshake". Calling ``nullrun.shutdown `` - before exit (or registering it via ``atexit``) eliminates the - noisy log. No-op if ``init `` was never called. + A long-running script that exits via ``sys.exit `` lets the + kernel RST the TCP socket, which the backend logs as WARN + "Connection reset without closing handshake". Calling + ``nullrun.shutdown `` before exit (or registering it via + ``atexit``) eliminates the noisy log. No-op if ``init `` was + never called. Args: timeout: seconds to wait for the WS close handshake to @@ -219,12 +219,11 @@ def init( """ Initialize the NullRun SDK. Call once at application startup. - `api_key` is **required** as of 0.3.0. The previous silent fallback to - "local mode" (a NullRunNoop stub) was removed because it hid policy - violations and bypassed every backend gate — a real safety hole. Pass - `api_key=...` explicitly or set the `NULLRUN_API_KEY` environment - variable before calling `init `. If neither is set, `init ` raises - `NullRunAuthenticationError`. + `api_key` is **required**. There is no silent fallback mode: every + gate calls the backend, and a missing key would silently bypass + every backend gate. Pass `api_key=...` explicitly or set the + `NULLRUN_API_KEY` environment variable before calling `init `. If + neither is set, `init ` raises `NullRunAuthenticationError`. Args: api_key: NullRun API key (or NULLRUN_API_KEY env var). Required. diff --git a/src/nullrun/_handle.py b/src/nullrun/_handle.py index 06f800d..a47b125 100644 --- a/src/nullrun/_handle.py +++ b/src/nullrun/_handle.py @@ -229,8 +229,8 @@ def handle(*, exit_code: int = 1): ) except Exception: # noqa: BLE001 # Defensive: never let the report builder block the exit. - # Fall back to the legacy single-line behaviour so a buggy - # helper can't freeze a script that would otherwise exit. + # Fall back to the single-line message so a buggy helper + # can't freeze a script that would otherwise exit. report = format_user_message(exc) print(report, file=sys.stderr) sys.exit(exit_code) diff --git a/src/nullrun/_registry.py b/src/nullrun/_registry.py index ece9faf..e1a347a 100644 --- a/src/nullrun/_registry.py +++ b/src/nullrun/_registry.py @@ -85,8 +85,9 @@ def get(self) -> NullRunRuntime | None: def set(self, runtime: NullRunRuntime) -> NullRunRuntime | None: """Install ``runtime`` as the active instance. - Returns the previously-installed runtime (or ``None``) so - the caller can shut it down before it is replaced. The + Returns the runtime that was installed before this call + (or ``None``) so the caller can shut it down before it is + replaced. The swap is atomic — a concurrent ``get`` sees either the old or the new instance, never a half-constructed one. """ diff --git a/src/nullrun/actions.py b/src/nullrun/actions.py index 583b107..60f7ddc 100644 --- a/src/nullrun/actions.py +++ b/src/nullrun/actions.py @@ -180,9 +180,8 @@ def handle( Raises: NullRunWorkflowKilledError: If action is "kill" - (2026-09-08 typed signal, NR-W002; subclass of - WorkflowKilledInterrupt which remains as the - back-compat name.) + (subclass of WorkflowKilledInterrupt, which is + the canonical catch name.) WorkflowPausedException: If action is "pause" NullRunBlockedException: If action is "block" """ @@ -247,12 +246,9 @@ def _default_kill( ) -> None: """Default kill handler - raises NullRunWorkflowKilledError. - 2026-09-08: typed kill signal (NR-W002). Cookbook code - can `except NullRunWorkflowKilledError` to react to - operator-initiated kills with structured error_code + - user_action. Legacy `except WorkflowKilledInterrupt` - still matches because NullRunWorkflowKilledError is a - subclass. + Cookbook code can `except NullRunWorkflowKilledError` to react + to operator-initiated kills with structured error_code + + user_action. """ logger.warning(f"KILL action for workflow {workflow_id}: {reason}") raise NullRunWorkflowKilledError( diff --git a/src/nullrun/audit.py b/src/nullrun/audit.py index bc9640f..4378258 100644 --- a/src/nullrun/audit.py +++ b/src/nullrun/audit.py @@ -26,9 +26,9 @@ `event_type` column at read time. Wire reference: - backend/src/proxy/http/audit.rs::AuditEntryResponse (13 governance - fields, plus the legacy `action` / `actor` / `actor_label` / - `outcome` / `metadata` shape preserved for back-compat). + backend/src/proxy/http/audit.rs::AuditEntryResponse — 13 + governance fields plus `action` / `actor` / `actor_label` / + `outcome` / `metadata` for the operator dashboard. """ from __future__ import annotations @@ -278,8 +278,8 @@ class AuditVerifyResult: """Outcome of /api/v1/orgs/:org_id/audit-log/verify. `verified` and `chain_valid` are the same value (the backend - surfaces both for back-compat). `first_failure_reason` is one - of `content_hash_mismatch` / `previous_hash_mismatch` / + surfaces both for the dashboard). `first_failure_reason` is + one of `content_hash_mismatch` / `previous_hash_mismatch` / `empty_chain` (or None when verified=True). `hmac_checked` is currently always False — the response diff --git a/src/nullrun/business_impact.py b/src/nullrun/business_impact.py index 8e5cce8..20c1946 100644 --- a/src/nullrun/business_impact.py +++ b/src/nullrun/business_impact.py @@ -2,13 +2,12 @@ BusinessImpact + action_digest — minimal wire helpers. The 0.18.2 SDK is policy-blind. Every ``@protect`` call computes a -single canonical ``NoImpact`` envelope (the only remaining -variant) and forwards it to /execute. The backend's ToolParameters -Approval Rules read values out of ``tool_kwargs`` directly via the -rule's ``param_name`` field, so the SDK no longer constructs -per-tool typed impacts (Money, ToolCall). This module keeps just -enough of the pre-0.18.2 wire layer for the gate to send a -valid (kind, action_digest) pair. +single canonical ``NoImpact`` envelope and forwards it to +/execute. The backend's ToolParameters Approval Rules read +values out of ``tool_kwargs`` directly via the rule's +``param_name`` field, so the SDK no longer constructs per-tool +typed impacts (Money, ToolCall). This module exists so the gate +can send a valid (kind, action_digest) pair. Field contract mirrored by the backend at ``backend/src/proxy/gate/business_impact.rs``: diff --git a/src/nullrun/capabilities.py b/src/nullrun/capabilities.py index 4b63f8d..2152efe 100644 --- a/src/nullrun/capabilities.py +++ b/src/nullrun/capabilities.py @@ -223,14 +223,7 @@ def _validate_capabilities_payload(payload: Any) -> list[str]: a Zod-style guard around :func:`parse_capabilities` so a malformed probe response (e.g. non-dict top level, capabilities array instead of dict) surfaces a typed warning instead of silently falling - through to legacy defaults. - - M8 (audit 2026-08-12): pre-fix, a malformed probe payload silently - yielded the conservative defaults via ``payload.get("capabilities") - or {}`` and the SDK continued in compatibility mode without - informing the operator. Post-fix, the operator sees a structured - ``NullRunCapabilitiesValidationError`` at ``init()`` and can - diagnose the probe failure before the first /check. + through to defaults. Note: validation is intentionally permissive about MISSING fields (the backend may add new fields at any time without bumping the diff --git a/src/nullrun/decorators.py b/src/nullrun/decorators.py index 716b1be..7b5c699 100644 --- a/src/nullrun/decorators.py +++ b/src/nullrun/decorators.py @@ -279,8 +279,8 @@ def _safe_error_str(error: BaseException | None) -> str | None: return _strip_details_balanced(raw) -# The legacy module-level slot was removed. Reads/writes now route -# through the registry (see nullrun._singleton._RuntimeProxyModule). +# Module-level reads/writes route through the registry +# (see nullrun._singleton._RuntimeProxyModule). def _get_or_create_runtime() -> NullRunRuntime: @@ -575,11 +575,11 @@ def _protect_body(args: tuple[Any, ...], kwargs: dict[str, Any], unify_block: bo # AND for ``parent_trace_id`` derivation at runtime.py:2967) # emits events tagged with the SAME trace_id / # span_id as SpanContext. Without this mirror a bare - # ``@protect`` (no enclosing ``with workflow``) still saw a - # tree-break: span_start carried SpanContext.trace_id while - # llm_call / tool_call carried ``generate_trace_id()`` from - # the legacy fallback. Token-based so a nested ``@protect`` - # inside an outer ``@protect`` (or inside ``with workflow``) + # ``@protect`` (no enclosing ``with workflow``) sees a + # tree-break: span_start carries SpanContext.trace_id while + # llm_call / tool_call carries a freshly generated trace_id. + # Token-based so a nested ``@protect`` inside an outer + # ``@protect`` (or inside ``with workflow``) # restores the outer trace/span on reset. trace_legacy_token = set_trace_id(span.trace_id) span_legacy_token = set_span_id(span.span_id) @@ -804,7 +804,7 @@ def _run_tool_policy_gate( fail_open = os.environ.get("NULLRUN_SENSITIVE_FAIL_OPEN", "").strip() == "1" # *display* workflow_id via the runtime's precedence chain # (contextvar → self.workflow_id → None). Sentinel stays as the - # last resort for legacy / never-bound keys. + # last resort for never-bound keys (no workflow context). workflow_id = runtime._resolve_workflow_id(get_workflow_id()) or UNKNOWN_WORKFLOW_ID try: @@ -904,10 +904,10 @@ def _run_tool_policy_gate( ) raise err from exc - # Defense in depth: legacy fallback / classification audit. If - # the transport ever returns a synthetic dict whose - # decision_source marks a fallback, block per ADR-008 fail-CLOSED. - # This arm is preserved for defense in depth even though the + # Defense in depth: classification audit. If the transport ever + # returns a synthetic dict whose decision_source marks a + # fallback, block per ADR-008 fail-CLOSED. This arm is preserved + # for defense in depth even though the # typed transport-error arms above are the canonical path. if isinstance(result, dict): decision_source = result.get("decision_source", "") diff --git a/src/nullrun/instrumentation/__init__.py b/src/nullrun/instrumentation/__init__.py index 681dd2e..caa2f30 100644 --- a/src/nullrun/instrumentation/__init__.py +++ b/src/nullrun/instrumentation/__init__.py @@ -6,9 +6,7 @@ live in `nullrun.toolbox` (e.g. `nullrun.toolbox.langgraph.wrapper` which replaced `nullrun.instrumentation.langgraph.instrument`). -The v0.x ``openai.ChatCompletion.create`` patcher was removed -in 0.4.0 — ``openai>=1.0`` does not expose that attribute. All -OpenAI v1.0+ traffic is now tracked vendor-independently by the +All OpenAI v1.0+ traffic is tracked vendor-independently by the httpx transport hook in ``nullrun.instrumentation.auto``. """ diff --git a/src/nullrun/instrumentation/_safe_patch.py b/src/nullrun/instrumentation/_safe_patch.py index 55b5948..5d09f40 100644 --- a/src/nullrun/instrumentation/_safe_patch.py +++ b/src/nullrun/instrumentation/_safe_patch.py @@ -1,15 +1,7 @@ """ Centralised error handling for auto-instrumentation patchers. -The pre-fix auto-instrumentation modules had 25+ instances of -``try/except Exception: pass # pragma: no cover`` scattered across -``auto.py``, ``auto_requests.py``, ``autogen.py``, ``crewai.py`` -``llama_index.py``. If a patch failed in production (typically -because the vendored SDK changed a method signature) the SDK would -silently degrade and the user would have no idea why their costs -were no longer being tracked. - -The fix: every patch call goes through ``safe_patch`` (B47) which: +Every patch call goes through ``safe_patch`` which: - Returns ``True``/``False`` based on patch outcome. - Logs at WARNING with the patch name + the actual exception (so a SRE can grep for ``Auto-instrumentation patch X failed`` diff --git a/src/nullrun/instrumentation/auto.py b/src/nullrun/instrumentation/auto.py index 465455e..2103931 100644 --- a/src/nullrun/instrumentation/auto.py +++ b/src/nullrun/instrumentation/auto.py @@ -736,8 +736,7 @@ def _check_kill_before_send(runtime: Any, request: httpx.Request) -> None: Raises: NullRunWorkflowKilledError: state == "Killed" (2026-09-08: typed signal with error_code=NR-W002 + user_action; - subclass of WorkflowKilledInterrupt which remains as a - back-compat name.) + subclass of WorkflowKilledInterrupt.) WorkflowPausedException: state == "Paused" """ if runtime is None: @@ -1783,12 +1782,10 @@ def auto_instrument(runtime: Any) -> bool: at least one path was installed (so the caller can log a useful 'instrumented N paths' message). - Every patch call is wrapped in ``safe_patch`` (B47) which logs - at WARNING if the patch raised a non-ImportError exception. The - pre-fix ``try/except Exception: pass # pragma: no cover`` blocks - meant a vendor SDK breaking change (e.g. a renamed method) - would silently disable cost tracking with no log line. The - operator would only find out when the bill arrived. + Every patch call is wrapped in ``safe_patch`` which logs at + WARNING if the patch raised a non-ImportError exception. A + vendor SDK breaking change (e.g. a renamed method) surfaces + a log line the operator can grep, not a silent degradation. """ global _auto_installed diff --git a/src/nullrun/observability/__init__.py b/src/nullrun/observability/__init__.py index 41a65eb..4587333 100644 --- a/src/nullrun/observability/__init__.py +++ b/src/nullrun/observability/__init__.py @@ -10,10 +10,9 @@ registry. See that module for the Layer-2 design. Both are reachable as ``nullrun.observability.metrics`` / -``nullrun.observability.error_hooks`` for back-compat. The -metrics singleton lives here (was previously a module-level -constant in ``observability.py``) — moving it into a package -was needed to make room for the ``error_hooks`` submodule. +``nullrun.observability.error_hooks``. The metrics singleton +lives here — moving it into a package made room for the +``error_hooks`` submodule. """ from __future__ import annotations diff --git a/src/nullrun/runtime.py b/src/nullrun/runtime.py index fcaa41e..ee2de8f 100644 --- a/src/nullrun/runtime.py +++ b/src/nullrun/runtime.py @@ -24,11 +24,20 @@ | `_emit_span_start` / `_emit_span_end` | n/a -- never blocks | n/a | n/a | | `/track` batch path (legacy) | OPEN-on-network-error (event dropped, no retry) | n/a -- circuit breaker backoff applies | none | -**Readme correction (2026-07-04):** the SDK_README.md claim -"Fail-OPEN на инфраструктурных сбоях. Если backend недоступен, бюджет -не блокирует агента" is **partially wrong** — it conflates SDK-side -transport failure with backend-side budget-enforcement failure. The -honest split is: +**Fail-OPEN policy** — SDK-side transport failure (network timeout, +5xx, breaker open) is fail-OPEN on the *check* path so a dead +backend doesn't freeze the user's agent loop. Backend-side +budget-enforcement failure (the /gate or /track handler actually +returned a wire response, just one indicating a Redis outage or +aggregate rate-limit Redis unavailable) → the wire response is +what it is, and the SDK raises the corresponding exception. +``BUDGET_REDIS_UNAVAILABLE`` → 402 ``NullRunBudgetError`` +(fail-CLOSED, the backend rejected the request because Redis was +unreachable for the budget counter — this is the authoritative +enforcement signal, not a transport blip). +``RATE_LIMIT_REDIS_UNAVAILABLE`` → 503 ``NullRunRateLimitRedisError`` +(fail-CLOSED for the same reason). The SDK does NOT silently +fall-OPEN on a wire 4xx/5xx that names an enforcement failure. * **SDK-side transport failure** (network timeout, 5xx, breaker open) → fail-OPEN on the *check* path so a dead backend doesn't freeze @@ -120,33 +129,26 @@ _protocol_header_value, _safe_json, ) -from nullrun.uuid7 import uuid7_str # 2026-07-04 BUG #4 +from nullrun.uuid7 import uuid7_str logger = logging.getLogger(__name__) # Sentinel used when a gate fires outside a ``with workflow(...)`` UNKNOWN_WORKFLOW_ID: str = "__nullrun_unknown__" -# 2026-07-04 (BUG #5): in-process gate cache for chain-mode. -# 2026-09-12 (DEF-CACHE-COST-ESTIMATE-COLLISION): the cache key -# includes ``estimated_tokens`` (currently hardcoded to 1 in -# ``check_workflow_budget``) so a future change that varies -# estimated_tokens by call does not silently serve a cheaper -# cached allow for a more expensive call. Without this, two -# chain-mode calls with the same (workflow_id, chain_id, -# call_model) but different cost_estimate collide and the -# cheaper response is reused — same blast radius as the original -# BUG #5 (over-reserve on the consume side), but at the cache -# layer instead of the wire layer. -# 2026-09-12 (DEF-CACHE-STALE-ALLOW-AFTER-OVERBUDGET): the cache is -# also invalidated when /track signals budget exhaustion -# (HTTP 422 from CONSUME_OVERBUDGET) and when ``chain_end`` is -# called. Without this, a chain that exhausts its budget mid-loop -# could continue serving cached "allow" decisions for up to 5 s -# (the TTL) after the backend has actually blocked subsequent -# calls. The invalidation uses the active chain_id from the -# contextvar so unrelated chains sharing the runtime instance -# are not affected. +# In-process gate cache for chain-mode. The cache key includes +# ``estimated_tokens`` (currently hardcoded to 1 in +# ``check_workflow_budget``) so two chain-mode calls with the same +# (workflow_id, chain_id, call_model) but different cost_estimate +# do not collide and silently serve a cheaper cached allow for +# a more expensive call. The cache is also invalidated when +# /track signals budget exhaustion (HTTP 422 from +# CONSUME_OVERBUDGET) and when ``chain_end`` is called — without +# this, a chain that exhausts its budget mid-loop could continue +# serving cached "allow" decisions for up to the TTL after the +# backend has actually blocked subsequent calls. Invalidations +# use the active chain_id from the contextvar so unrelated +# chains sharing the runtime instance are not affected. _GATE_CACHE: dict[tuple[str, str | None, str | None, int], tuple[float, dict[str, Any]]] = {} _GATE_CACHE_TTL_SECONDS: float = 5.0 @@ -195,17 +197,16 @@ def _invalidate_gate_cache_for_chain(workflow_id: str | None, chain_id: str | No _GATE_CACHE.pop(k, None) return len(keys_to_drop) -# 2026-07-24 (Root-cause fix for the ``@sensitive`` reinit gap): +# Tracks which runtime instances have already forced strict mode, +# preventing re-initialization from regressing back to permissive. _STRICT_MODE_FORCED: set[str] = set() -# v3.53 audit #6 — production-environment detection for security -# opt-out enforcement. ``NULLRUN_SKIP_BUDGET_CHECK=1`` is documented -# as a DEV / TEST bypass (CLAUDE.md §20 "Никогда не выставлять -# NULLRUN_SKIP_BUDGET_CHECK в production env"); pre-v3.53 the SDK -# silently honored the opt-out regardless of environment, which -# meant an operator who accidentally exported the var in prod got -# a silent fail-OPEN on the budget gate. +# Production-environment detection for security opt-out +# enforcement. ``NULLRUN_SKIP_BUDGET_CHECK=1`` is documented as a +# DEV / TEST bypass (CLAUDE.md §20 "Никогда не выставлять +# NULLRUN_SKIP_BUDGET_CHECK в production env"); the SDK refuses +# it outside dev / test environments. # # Two signals combine to flag a production environment: # 1. The configured ``api_url`` hostname matches the prod host @@ -227,7 +228,7 @@ def _is_production_environment(api_url: str | None = None) -> bool: """Return True when the SDK is running against the production NullRun backend. - v3.53 audit #6 — used by ``check_workflow_budget`` to refuse + Used by ``check_workflow_budget`` to refuse ``NULLRUN_SKIP_BUDGET_CHECK=1`` outside dev / test environments. Detection rules (in order): @@ -300,7 +301,6 @@ def is_strict_mode_forced(tool_name: str) -> bool: return tool_name in _STRICT_MODE_FORCED -# 2026-07-04 (v0.12.0 wiring fix — ): SERVER_MINTED_RESERVATION_MAX_AGE_SECONDS: float = 295.0 # Hard cap on server-supplied approval_timeout_seconds. The @@ -610,12 +610,8 @@ def __init__( debug: bool = False, _test_mode: bool = False, polling: bool = True, - # DEF-ERRHDL-NO-TIMEOUT-01 (2026-08-11, RUN_ID 20260811-1): - # expose request_timeout so operators can tune the httpx read - # timeout for slow-network scenarios. Pre-fix the SDK hardcoded - # 30s read timeout in transport.py with no config surface. - # Precedence: kwarg > NULLRUN_REQUEST_TIMEOUT env var > 30.0 - # (the pre-fix default). + # Tune the httpx read timeout for slow-network scenarios. + # Precedence: kwarg > NULLRUN_REQUEST_TIMEOUT env var > 30.0. request_timeout: float | None = None, ): """ @@ -636,16 +632,11 @@ def __init__( Note: - `organization_id` is set from `_authenticate ` after init; it is NOT a public init parameter and not read from env. - - `api_key` is required as of 0.3.0 (T3-S2). The previous - `local_mode` flag was removed because it silently bypassed - every backend gate. - - `fallback_mode` is fixed at STRICT (no public override). - v3.53 audit #4 — was PERMISSIVE pre-v3.53; flipped to - STRICT to honor CLAUDE.md §4 ("DEFAULT: fail-CLOSED для - всех enforcement путей"). Existing callers passing - ``fallback_mode="permissive"`` continue to opt into the - legacy fail-OPEN path; the default-only change is the - break. + - `api_key` is required. There is no `local_mode` flag — + it would silently bypass every backend gate. + - `fallback_mode` is fixed at STRICT (no public override) + to honor CLAUDE.md §4 ("DEFAULT: fail-CLOSED для всех + enforcement путей"). - `timeout`/`max_retries` are fixed at 30s / 3 (no public override). Raises: @@ -666,34 +657,27 @@ def __init__( self.secret_key = secret_key or os.getenv("NULLRUN_SECRET_KEY") self.api_url = api_url or os.getenv("NULLRUN_API_URL", "https://api.nullrun.io") - # T3-S2 (0.3.0): api_key is now required. The previous `local_mode` + # api_key is required — there is no fallback. if not self.api_key: raise NullRunAuthenticationError( "NullRunRuntime() requires an api_key. Pass api_key='nr_live_...' " - "or set NULLRUN_API_KEY. (Silent no-op fallback was removed " - "in 0.3.0 -- see CHANGELOG.)" + "or set NULLRUN_API_KEY." ) # organization_id is set by _authenticate; stays None until then. self.organization_id: str | None = None # workflow_id is set by _authenticate from the API key's # binding (organization_api_keys.workflow_id). Used as a # fallback for /check, /status, and span events when the - # user hasn't entered a `with workflow(...)` context. None - # on legacy keys (pre-139 or never used) -- call sites - # must NOT invent one. + # user hasn't entered a `with workflow(...)` context. + # Call sites must NOT invent one. self.workflow_id: str | None = None self._test_mode = _test_mode self.polling = polling - # The string ``fallback_mode`` parameter is deprecated and - # accepted only for backward compat — the CACHED variant - # was removed in 0.7.0 because the SDK no longer maintains - # a local policy cache (see CHANGELOG D-01). v3.53 audit #4 - # flipped the default from PERMISSIVE to STRICT so a future - # caller who omits the kwarg lands on fail-CLOSED per - # CLAUDE.md §4 instead of silently allowing local execution - # on transport failure. + # ``fallback_mode`` defaults to STRICT per CLAUDE.md §4 + # ("DEFAULT: fail-CLOSED для всех enforcement путей"). + # PERMISSIVE is an explicit override. fb_upper = str(fallback_mode).upper() if fallback_mode is not None else "STRICT" if fb_upper == "PERMISSIVE": self._fallback_mode = FallbackMode.PERMISSIVE @@ -859,13 +843,6 @@ def __init__( # Test mode: skip all network calls self._transport.start() else: - # AUTH-01 (2026-09-11): previously this arm caught ``httpx.RequestError`` - # and re-raised ``NullRunAuthenticationError``, misclassifying network - # failures as auth failures. Arm B in ``_authenticate`` already catches - # the same condition with the correct class (``NullRunTransportError``); - # this defensive duplicate was a backstop for a code path that no longer - # exists between ``_authenticate()`` and ``self._transport.start()``. - # Remove: arm B is sufficient. self._authenticate() self._transport.start() # Start remote polling unless disabled (internal `polling=False` @@ -1268,7 +1245,7 @@ def _authenticate(self) -> None: user_action=( "Set NULLRUN_API_KEY env var or pass api_key='nr_live_...' " "to nullrun.init(). The SDK cannot operate without " - "credentials — the no-op local mode was removed in 0.3.0." + "credentials." ), ) self._emit_sdk_error(err, stage="auth") @@ -1276,7 +1253,7 @@ def _authenticate(self) -> None: logger.debug(f"Authenticating with API at {self.api_url}/auth/verify") try: - # 2026-06-28 audit P2.3: retry transient 503/504 + network blips + # Retry transient 503/504 + network blips. response = self._post_auth_with_retry( f"{self.api_url}/api/v1/auth/verify", json_body={"api_key": self.api_key}, @@ -1284,25 +1261,12 @@ def _authenticate(self) -> None: ) if response.status_code == 200: - # DEF-ERRHDL-INVALID-JSON-01 (2026-08-11, RUN_ID 20260811-1): - # route 200-OK JSON parse through _safe_json so a malformed - # body raises NullRunTransportError (NR-T001) instead of - # leaking json.JSONDecodeError to user code. The - # /check/track/... paths already use _safe_json (transport.py). data = _safe_json(response, "auth") # STRICT MODE: organization_id is REQUIRED, no fallback org_id = data.get("organization_id") if not org_id: - # DEF-ERRHDL-MALFORMED-MSG-01 (2026-08-11, RUN_ID 20260811-1): - # drop "compromised" wording. "compromised" is a - # security-incident term that triggers SOC alerts in - # observability stacks; using it for a routine schema - # mismatch is misleading. Wording now attributes the - # failure to a wire-shape mismatch without making a - # security claim. err = NullRunAuthenticationError( - "Auth response missing organization_id -- server returned an unexpected response shape. " - "Refusing to operate with legacy identity.", + "Auth response missing organization_id -- server returned an unexpected response shape.", error_code="NR-A002", user_action=( "The NullRun backend returned a 200 but the response " @@ -1316,21 +1280,12 @@ def _authenticate(self) -> None: raise err self.organization_id = org_id - # Pick up the workflow this key is bound to. - # `None` on legacy keys (pre-139 or never-used) -- - # call sites that NEED a workflow - # (check_workflow_budget, check_control_plane, span - # events) will fall through to the contextvar when - # self.workflow_id is None, exactly like before. - # New keys always have this set. + # Pick up the workflow this key is bound to. New + # keys always have this set; if it's missing the + # SDK can't honour the dashboard's KILL/PAUSE for + # this key — the operator needs to rotate the key. self.workflow_id = data.get("workflow_id") - # Legacy API keys do not return workflow_id, so the - # SDK cannot honour the dashboard's KILL/PAUSE for - # that workflow. Emit a one-time WARNING so the - # operator knows to rotate the key. Without this, - # the kill switch silently no-ops (a real safety - # hole for legacy users). if self.workflow_id is None: masked_key = ( (self.api_key[:8] + "***") @@ -1362,21 +1317,10 @@ def _authenticate(self) -> None: logger.info(f"Authenticated: organization_id={self.organization_id}") else: - # DEF-ERRHDL-AUTH-PATH-CODE-PIN-01 (2026-08-11, RUN_ID 20260811-1): - # route 5xx to NullRunBackendError so the auth path uses the same - # error envelope classification as /check/track. Per CLAUDE.md - # §13 5xx is a backend-class error, not auth-class. Without this - # split, operators are nudged to rotate valid keys during backend - # outages ("API key may be invalid or expired" for status=500 is - # misleading). - # - # - 401 -> NullRunAuthenticationError + NR-A003 (key was actually - # rejected; this stays a true auth failure). - # - All other 4xx -> NullRunAuthenticationError + NR-A001 (the - # prior code; covers 403 etc.). - # - 5xx -> NullRunBackendError + NR-B002 (the existing transport - # envelope's error_code, so 5xx is classified the same as 5xx - # from /check/track). + # 5xx routes to NullRunBackendError (backend-class + # error, per CLAUDE.md §13). 401/NR-A003 = key + # rejected (true auth failure). Other 4xx / NR-A001 + # covers 403 etc. status = response.status_code correlation_id = response.headers.get("x-correlation-id") if 500 <= status < 600: @@ -1401,17 +1345,15 @@ def _authenticate(self) -> None: ) raise err except httpx.RequestError as e: - # AUTH-01 (2026-09-11): reclassify httpx.RequestError as - # ``NullRunTransportError`` instead of ``NullRunAuthenticationError``. - # The previous wrap misled operators — the same condition (DNS failure, - # connection refused, TLS handshake error, request timeout) is correctly - # classified as ``NullRunTransportError(NETWORK_ERROR, "auth")`` by - # ``Transport.heartbeat`` (transport.py:2077) and the rest of the SDK. - # The ``user_action`` previously embedded here noted "This is a - # transport failure (not an auth failure)" — the class should match - # the message. ``NullRunTransportError.__init__`` already sets - # ``error_code="NR-B001"`` (transport.py:233) and the standard - # retryable ``user_action`` (transport.py:234-237). + # Reclassify httpx.RequestError as + # ``NullRunTransportError`` instead of + # ``NullRunAuthenticationError`` — the same condition + # (DNS failure, connection refused, TLS handshake error, + # request timeout) is correctly classified as + # ``NullRunTransportError(NETWORK_ERROR, "auth")`` by + # the rest of the SDK. ``NullRunTransportError.__init__`` + # already sets ``error_code="NR-B001"`` and the standard + # retryable ``user_action``. err = NullRunTransportError( f"Auth request failed: {e}. Cannot establish secure connection to NullRun. " f"Refusing to operate in unprotected mode.", @@ -1426,9 +1368,8 @@ def _start_remote_polling(self) -> None: Defaults to WebSocket push for sub-second kill/pause propagation. Set `NULLRUN_TRANSPORT=http` to fall back to - the legacy 1-second HTTP poll (kept for environments where - the WS endpoint is blocked or for parity with old SDK - behavior). + the 1-second HTTP poll (for environments where the WS + endpoint is blocked). """ if self._transport_mode == "http": self._start_http_poller() @@ -1436,7 +1377,7 @@ def _start_remote_polling(self) -> None: self._start_ws_listener() def _start_http_poller(self) -> None: - """Legacy: poll the server every second for state changes.""" + """Poll the server every second for state changes.""" self._poll_running = True self._poll_thread = threading.Thread( target=self._poll_commands, daemon=True, name="nullrun-poller" @@ -1633,25 +1574,16 @@ def _set_remote_state(self, workflow_id: str, state: dict[str, Any]) -> None: def _fetch_remote_state(self, workflow_id: str) -> None: """Fetch remote state for a specific workflow. - 2026-06-27: target endpoint swapped from - ``GET /api/v1/orgs/{org_id}/workflows/{workflow_id}`` (the - DASHBOARD route — requires Bearer session cookie, returns 401 - to SDK clients that only send X-API-Key) to - ``GET /api/v1/status/{workflow_id}`` (the SDK-polling route — - backend/src/proxy/handlers.rs:9758, accepts X-API-Key OR - Authorization: Bearer). Pre-swap the HTTP-poll path silently - 401'd on every poll, so the legacy HTTP-poll fallback never - observed a remote kill/pause. WS push (the default mode) - does NOT go through this code path, so the WS control plane - is unaffected. + Calls ``GET /api/v1/status/{workflow_id}`` (the SDK-polling + route — backend/src/proxy/handlers.rs:9758, accepts X-API-Key + OR Authorization: Bearer). Backend ``StatusResponse`` (handlers.rs:9747-9756) returns ``workflow_id, state, version, reason?, updated_at current_cost, rate_per_minute``. We only consume ``state`` — ``version`` and ``reason`` are SDK-local fields and remain at - their cached values (mirroring the prior behaviour). This is - sufficient for ``check_control_plane`` which only reads - ``state``. + their cached values. This is sufficient for + ``check_control_plane`` which only reads ``state``. """ try: response = self._transport._client.get( @@ -1877,16 +1809,11 @@ def check_control_plane(self, workflow_id: str) -> None: Raises: WorkflowPausedException: If workflow is paused on server - NullRunWorkflowKilledError: If workflow is killed on - server (2026-09-08 typed signal, NR-W002; subclass - of WorkflowKilledInterrupt which remains as the - back-compat name.) + NullRunWorkflowKilledError: If workflow is killed on server. """ # Prefer the explicit arg (contextvar-supplied), fall back - # to the API key's bound workflow. None on legacy keys -- - # in that case there's no workflow to check, so we no-op - # (preserves the legacy behavior for keys that have never - # been workflow-bound). + # to the API key's bound workflow. If neither resolves, + # there's no workflow to check, so we no-op. resolved = self._resolve_workflow_id(workflow_id or None) if not resolved: return @@ -1902,13 +1829,9 @@ def check_control_plane(self, workflow_id: str) -> None: remote_state = self._remote_state_for(workflow_id) state = remote_state.get("state", "Normal") - # S-4: case-insensitive compare. The backend - # already emits PascalCase via the `as_pascal_case ` normaliser - # in `handlers.rs:9258`, but a future regression to UPPERCASE - # (or any other casing) would silently fail the match and let a - # killed workflow keep running. Normalise here so the SDK - # survives any wire-format drift without needing a coordinated - # backend change. + # Case-insensitive compare so a future wire-format drift + # (e.g. regression to UPPERCASE) cannot silently let a + # killed workflow keep running. state_normalized = state.lower() if isinstance(state, str) else "normal" if state_normalized == "paused": @@ -1946,8 +1869,8 @@ def check_workflow_budget(self) -> None: pattern in `check_control_plane` -- a transient backend outage must never freeze the user's agent. The /track fast path also does not gate on budget, so the worst case under /gate failure - is that we revert to the pre-C behaviour: budget enforcement is - advisory until the gateway recovers. + is that budget enforcement is advisory until the gateway + recovers. Uses `estimated_tokens=1` (the minimum the API accepts). Goal is the binary question "is there any budget left?", not cost @@ -1959,25 +1882,19 @@ def check_workflow_budget(self) -> None: exhausted its budget from previous runs and the test only wants to exercise a non-budget code path. - Production guard (v3.53 audit #6): the opt-out is - REFUSED in production environments (api_url matches - ``api.nullrun.io`` or ``NULLRUN_ENV=production``). This - closes the silent fail-OPEN class where an operator - accidentally exported the var in a prod deployment and - silently lost the budget gate. To explicitly acknowledge - the risk in prod (e.g. an incident-response runbook - scenario), set ``NULLRUN_ALLOW_SKIP_BUDGET_CHECK=1`` as - well — the SDK logs the explicit ack at WARNING level - and emits a metric so the opt-in is visible in - observability. + Production guard: the opt-out is REFUSED in production + environments (api_url matches ``api.nullrun.io`` or + ``NULLRUN_ENV=production``), per CLAUDE.md §20 ("DEV/TEST + only"). To explicitly acknowledge the risk in prod (e.g. + an incident-response runbook scenario), set + ``NULLRUN_ALLOW_SKIP_BUDGET_CHECK=1`` as well — the SDK + logs the explicit ack at WARNING level and emits a metric + so the opt-in is visible in observability. """ if os.environ.get("NULLRUN_SKIP_BUDGET_CHECK", "").strip() == "1": # Production guard: refuse the opt-out unless the # operator explicitly acked via - # ``NULLRUN_ALLOW_SKIP_BUDGET_CHECK=1``. CLAUDE.md §20 - # marks NULLRUN_SKIP_BUDGET_CHECK as DEV/TEST only; - # pre-v3.53 the SDK silently honored it in any env - # which made accidental prod misuse a silent fail-OPEN. + # ``NULLRUN_ALLOW_SKIP_BUDGET_CHECK=1``. if _is_production_environment(self.api_url): allow_ack = os.environ.get("NULLRUN_ALLOW_SKIP_BUDGET_CHECK", "").strip() == "1" if not allow_ack: @@ -2052,33 +1969,25 @@ def check_workflow_budget(self) -> None: # stashed it here. /track works on the server-minted # execution_id, NOT op_id, so it is independent. # - # 2026-09-13 (DEF-OPID-REUSE-HASH-MISMATCH): the prior - # `if op_id is None:` guard leaked the scope's first - # op_id across subsequent ``check_workflow_budget()`` - # invocations. IDEM-01 on the server keys on op_id but - # verifies the 11-field semantic hash - # (``backend/src/redis/idempotency_store.rs::compute_gate_semantic_hash``) - # — a follow-up /check with different `tools` / - # `model` / `input` would 409 IDEMPOTENCY_KEY_MISMATCH - # and surface as NR-B004 in the SDK - # (``nullrun_openai_approval_demo.py`` symptom). The - # LangGraph ``NullRunCallback.on_llm_start`` fires a - # tools=None /gate BEFORE the @protect - # ``tools=['refund_customer']`` /gate in the same scope, - # which is the canonical repro (probed via - # probe_full.py 2026-09-13). Mint-fresh-per-call - # preserves the P0-27 within-action binding while - # removing the cross-action reuse that trips IDEM-01. + # Wire-binding invariant: /check + /execute (and the + # post-approval re-fire) within ONE logical action share + # the SAME op_id. The original implementation minted once + # per scope via the contextvar; /execute reads the + # freshly-minted value via get_operation_id() because we + # just stashed it here. /track works on the server-minted + # execution_id, NOT op_id, so it is independent. # - # We still read the contextvar first (rather than the - # pre-fix unconditional mint) to keep the P0-27 source- - # pin test ``test_check_workflow_budget_reads_contextvar`` - # green and to surface any unexpected caller that + # Mint-fresh-per-call removes the cross-action reuse that + # trips IDEM-01 (the prior `if op_id is None:` guard + # leaked the scope's first op_id across subsequent + # ``check_workflow_budget()`` invocations). We still read + # the contextvar first to keep the within-action binding + # observable and to surface any unexpected caller that # pre-populates ``operation_id`` (e.g. test fixtures). # The read result is intentionally unused: /execute, # which runs synchronously in the SDK after /check, # reads the freshly-stashed value below via - # ``get_operation_id()`` — that is the P0-27 binding. + # ``get_operation_id()`` — that is the binding. op_id = _get_op_id_for_check() op_id = str(uuid.uuid4()) _set_op_id_for_check(op_id) @@ -2137,19 +2046,13 @@ def check_workflow_budget(self) -> None: call_tools = get_call_tools() # 2026-07-02 (v0.11.0): forward chain context for soft-mode + # Chain context for soft-mode enforcement. chain_id = get_chain_id() chain_op = get_chain_op() check_req = { "organization_id": self.organization_id or "local", - # 2026-07-04 (BUG #4): requires server-minted "execution_id": uuid7_str(), - # AUDIT P0-27 (2026-09-05): operation_id comes from the - # contextvar minted at the top of this method (and shared - # with /execute). The pre-fix `str(uuid.uuid4())` minted - # a separate value here — /check and /execute would have - # produced distinct operation_ids for one logical action, - # silently breaking the backend's binding key. "operation_id": op_id, "check_type": "llm", "model": call_model, # may be None if user didn't set it @@ -2157,17 +2060,15 @@ def check_workflow_budget(self) -> None: "stream": False, } - # v0.16.1 (Phase-1+ wire-shape fix): Phase-1+ SDKs MUST - # populate `action_digest` on every /gate call, even when no - # typed business impact is extracted (`@protect`-decorated - # LLM-only calls). Per `backend/src/proxy/http/gate/gate.rs:56` - # v3.62.1 / ADR-023 P1-6 the gate fail-CLOSED-rejects any - # proto>=3 client that omits the digest. We always emit a - # NoImpact sentinel here — typed Money/ToolCall impacts are - # forwarded by `runtime.execute(...)` directly (see - # `transport.py::execute`) and do not pass through this - # pre-flight gate. Computing once per call (not cached) is - # fine: compute_action_digest is ~5µs of pure stdlib. + # `action_digest` is required on every /gate call. Per + # `backend/src/proxy/http/gate/gate.rs:56` (ADR-023 P1-6) + # the gate fail-CLOSED-rejects any proto>=3 client that + # omits the digest. We always emit a NoImpact sentinel here + # — typed Money/ToolCall impacts are forwarded by + # `runtime.execute(...)` directly (see `transport.py::execute`) + # and do not pass through this pre-flight gate. Computing + # once per call (not cached) is fine: compute_action_digest + # is ~5µs of pure stdlib. check_req["action_digest"] = _compute_action_digest(_BusinessImpact.no_impact()) # Forward the tool list so backend (T3) can match each tool @@ -2203,6 +2104,9 @@ def check_workflow_budget(self) -> None: check_req["chain_op"] = chain_op if chain_op != "auto" else None # 2026-07-02 (v0.11.0): idempotency key. + # operation_id is also the idempotency key — /check and /track + # with the same operation_id are treated as a single logical + # action by the backend's binding key. check_req["idempotency_key"] = check_req["operation_id"] # In-process gate cache for chain-mode invocations. See @@ -2227,12 +2131,8 @@ def check_workflow_budget(self) -> None: except (httpx.HTTPError, NullRunError) as exc: # Narrow catch: fail-OPEN only on transport + # classified SDK errors. Internal bugs - # (KeyError, AttributeError) should surface - # rather than silently allow an unbounded call. - # 2026-08-13 (sprint handoff Bug #4): emit metric - # so sustained backend outages that bypass the - # budget gate via the documented ADR-008 fail-OPEN - # posture are visible in /health and alertable. + # (KeyError, AttributeError) surface rather + # than silently allowing an unbounded call. logger.warning(f"check_workflow_budget: /gate unavailable, failing open: {exc}") metrics.inc_runtime("gate_fail_open_total") return @@ -2242,16 +2142,10 @@ def check_workflow_budget(self) -> None: try: response = self._transport.check(check_req) except Exception as exc: # noqa: BLE001 - # 2026-08-13 (sprint handoff Bug #4): same metric emit - # as the cache-enabled arm above -- all three fail-OPEN - # paths in this method increment the same counter so an - # operator dashboard can graph "budget gate bypass - # rate" without per-site accounting. logger.warning(f"check_workflow_budget: /gate unavailable, failing open: {exc}") metrics.inc_runtime("gate_fail_open_total") return - # 2026-07-04 (v0.12.0 wiring fix — ): _capture_server_minted_execution_id(response) decision = response.get("decision", "allow") @@ -2259,21 +2153,13 @@ def check_workflow_budget(self) -> None: # Only fail-OPEN on EXPLICIT synthetic responses # (decision_source starts with "fallback" or is one of the # classified TransportErrorSource values). Real backend - # decisions (decision_source="gateway", or missing for - # backward compat) are honoured. + # decisions (decision_source="gateway") are honoured. if decision_source.startswith("fallback") or decision_source in { TransportErrorSource.NETWORK_ERROR, TransportErrorSource.GATEWAY_ERROR, TransportErrorSource.BREAKER_OPEN, TransportErrorSource.AUTH_ERROR, }: - # 2026-08-13 (sprint handoff Bug #4): the docblock above - # (lines 1763-1769) declares this path "logged at warning - # level and the caller proceeds" but the pre-fix code - # emitted DEBUG, making the fail-OPEN invisible to - # operators tailing INFO+ logs. Promote to WARNING so the - # contract matches the implementation; emit metric for - # parity with the two exception-path sites above. logger.warning( f"check_workflow_budget: synthetic decision_source=" f"{decision_source!r}, treating as transport error" @@ -2281,7 +2167,6 @@ def check_workflow_budget(self) -> None: metrics.inc_runtime("gate_fail_open_total") return if decision == "block": - # FIX-2026-06-27: backend /gate sets both `explanation` (a reasons = response.get("explanations") or ( [response["explanation"]] if response.get("explanation") else ["block"] ) @@ -2291,7 +2176,6 @@ def check_workflow_budget(self) -> None: # distinct from loop / retry / rate which have their # own counters. metrics.inc_runtime("cost_limit_exceeded") - # 2026-09-08: typed hard-block (NR-B004 budget cap). # ``NullRunBudgetError`` carries structured # ``error_code``, ``user_action``, ``retryable`` so the # LLM gets an actionable hint instead of "Something went @@ -2349,7 +2233,7 @@ def check_workflow_budget(self) -> None: return if decision == "require_approval": - # The gate requires a human-approval before the call + # The gate requires human-approval before the call # may proceed. Block the calling thread on the WS push # (handled in _handle_approval_resolved) and let the # operator click Approve/Deny on the dashboard. On @@ -2364,15 +2248,12 @@ def check_workflow_budget(self) -> None: # `NULLRUN_APPROVAL_TIMEOUT_SECONDS`. This prevents the # SDK/backend desync that the backend expiry sweeper # was written to fix. We fall back to the env default - # only when the field is missing or non-positive -- - # both signal "backend without that field" and we - # preserve the legacy behaviour for those callers. + # only when the field is missing or non-positive. approval_id = response.get("approval_id", "") or "" if not approval_id: logger.warning( "check_workflow_budget: require_approval decision but no approval_id in response" ) - # 2026-09-08: typed backend error (NR-B002, retryable). # The server returned require_approval without an # approval_id -- this is a wire-bug / drift, not a # budget block. Surface as retryable backend error @@ -2401,18 +2282,12 @@ def check_workflow_budget(self) -> None: f"check_workflow_budget: require_approval id={approval_id} -- " f"waiting for WS push (timeout={server_timeout if server_timeout is not None else 'env-default'})" ) - # 2026-09-11: pass the actual server-minted execution_id - # (captured above from the /gate response) into the WS - # wait so the entry's metadata + diagnostic log lines - # reflect the same id the server stamped on the - # approval row. Pre-fix this string fell back to - # ``str(self.organization_id)`` which made the - # ``__nullrun_unknown__`` sentinel leak into exception - # payloads (demo - # langgraph_openai_approval_demo.py prints - # ``execution_id=exc.workflow_id``). The handler matches - # purely on ``approval_id``, so this is diagnostic-only - # — captured on the consumer side. + # Pass the actual server-minted execution_id (captured + # above from the /gate response) into the WS wait so the + # entry's metadata + diagnostic log lines reflect the same + # id the server stamped on the approval row. The handler + # matches purely on ``approval_id``, so this is + # diagnostic-only — captured on the consumer side. from nullrun.context import get_server_minted_execution_id _captured_eid = get_server_minted_execution_id() @@ -2448,9 +2323,6 @@ def check_workflow_budget(self) -> None: logger.info(f"check_workflow_budget: approval {approval_id} approved -- resuming") return if outcome == "denied": - # 2026-09-08: typed approval-denied (NR-A011). Cookbook - # code can `except NullRunApprovalDeniedError` to - # surface the denial note + user_action to the LLM. raise NullRunApprovalDeniedError( workflow_id=workflow_id, reason=f"approval denied: {result.get('note') or 'operator denied'}", @@ -2458,9 +2330,6 @@ def check_workflow_budget(self) -> None: denial_note=result.get("note"), ) # timeout: fail-CLOSED -- do not run the call. - # 2026-09-08: typed approval-expired (NR-A012) -- THE TRIGGER - # FIX. The LLM now sees "Approval expired after 300s of - # WS push silence" instead of "Something went wrong". raise NullRunApprovalExpiredError( workflow_id=workflow_id, reason=( @@ -2726,14 +2595,11 @@ def chain_end(self, chain_id: str) -> dict[str, Any]: Returns: Parsed JSON dict. """ - # DEF-CHAIN-END-ORG-ID (2026-09-11): ``Transport.chain_end`` pre-fix - # sent only ``{chain_id, chain_op, execution_id}`` to /gate and the - # backend rejected with 422 ``missing field 'organization_id'``. - # Fix: forward ``self.organization_id`` (set in ``_authenticate``) + # Forward ``self.organization_id`` (set in ``_authenticate``) # and the contextvar trace_id so the SDK builds a complete - # ``GateRequest`` body. ``trace_id`` is sourced from the contextvar - # to match the rest of the SDK's wire-shape policy (one trace id - # per logical chain). + # ``GateRequest`` body. ``trace_id`` is sourced from the + # contextvar to match the rest of the SDK's wire-shape policy + # (one trace id per logical chain). from nullrun.context import get_trace_id result = self._transport.chain_end( @@ -2741,14 +2607,9 @@ def chain_end(self, chain_id: str) -> dict[str, Any]: organization_id=self.organization_id, trace_id=get_trace_id(), ) - # 2026-09-12 (DEF-CACHE-STALE-ALLOW-AFTER-OVERBUDGET): drop - # the in-process gate cache for this chain. The chain is - # closed on the server; any cached "allow" for the same - # (workflow_id, chain_id) is unreachable from future calls - # (UUID v4 collision risk is negligible but the cleanup - # costs nothing). Invalidating AFTER the wire call so a - # transient transport failure does not free the cache - # before the server confirms closure. + # Drop the in-process gate cache for this chain AFTER the wire + # call so a transient transport failure does not free the + # cache before the server confirms closure. workflow_id_str = str(self.workflow_id) if self.workflow_id else None _invalidate_gate_cache_for_chain(workflow_id_str, chain_id) return result @@ -2780,10 +2641,10 @@ def approximate_budget(self) -> dict[str, Any]: def _auth_headers(self) -> dict[str, str]: """Get authentication headers. - the wire-protocol handshake header is - required on every signed POST. The three direct callers of - this helper — ``_post_auth_with_retry``, ``_fetch_remote_state`` - and ``get_org_status`` — all go through the backend's protocol + The wire-protocol handshake header is required on every signed + POST. The three direct callers of this helper — + ``_post_auth_with_retry``, ``_fetch_remote_state`` and + ``get_org_status`` — all go through the backend's protocol middleware, so the header has to be present here rather than at every call site. """ @@ -2807,7 +2668,8 @@ def shutdown(self, flush: bool = True) -> None: (see ``Transport.stop(flush=False)`` for the full rationale; observed 9m 47s CI noise on PR #60). """ - # Stop the HTTP poller (legacy path) if it was started. + # Stop the HTTP poller (NULLRUN_TRANSPORT=http fallback) + # if it was started. self._poll_running = False if self._poll_thread and self._poll_thread.is_alive(): # Cap to 0.5s so a SIGTERM handler returns quickly. @@ -2985,12 +2847,7 @@ def is_sensitive_tool(self, tool_name: str) -> bool: Returns: True if tool requires strict mode - P2-3: match is case-insensitive. The pre-fix code did an exact - ``tool_name in self._sensitive_tools`` check, so a tool - registered as ``"stripe.charge"`` would silently fail to - match a caller passing ``"Stripe.Charge"`` — bypassing the - sensitive gate and running the body without an /execute - round-trip. The fix normalises both sides to lowercase + Match is case-insensitive: both sides are normalised to lowercase before the membership test, matching the case-insensitive style of ``_safe_kwargs``. @@ -2999,12 +2856,9 @@ def is_sensitive_tool(self, tool_name: str) -> bool: ``add_sensitive_tool``. The lock is uncontended under CPython's GIL, so the cost is negligible. """ - # O(1) lookup against the pre-lowercased frozenset - # snapshot. The lock is still taken to keep the snapshot - # coherent with the live set during concurrent - # add/remove_sensitive_tool calls (the snapshot is rebuilt - # under the lock), but the read itself is a single - # frozenset membership check. + # Lock-guarded O(1) lookup against the pre-lowercased frozenset + # snapshot. The lock keeps the snapshot coherent with the + # live set during concurrent add/remove calls. needle = tool_name.lower() with self._tools_lock: return needle in self._sensitive_tools_lower or needle in self._strict_mode_tools_lower @@ -3022,11 +2876,6 @@ def add_sensitive_tool(self, tool_name: str) -> None: Example: runtime = NullRunRuntime.get_instance runtime.add_sensitive_tool("my.custom_tool") - - #39: takes ``_tools_lock`` so the mutation is atomic - against concurrent ``is_sensitive_tool`` reads and other - ``add``/``remove`` calls. Without the lock a free-threaded - build could observe a torn set state during the mutation. """ with self._tools_lock: self._strict_mode_tools.add(tool_name) @@ -3044,8 +2893,6 @@ def remove_sensitive_tool(self, tool_name: str) -> None: Example: runtime = NullRunRuntime.get_instance runtime.remove_sensitive_tool("my.custom_tool") - - #39: takes ``_tools_lock`` to mirror ``add_sensitive_tool``. """ with self._tools_lock: self._strict_mode_tools.discard(tool_name) @@ -3140,10 +2987,6 @@ def execute( to this gate decision (v4 wire field; null on pre-v4 backends). Captured via `_capture_wire_evidence` → `set_last_gate_policy_hash` for downstream audit linkage. - NOTE: this is NOT a sequential `policy_version` number — - wire v3/v4 backends emit only `policy_hash`; legacy - `policy_version` references in this SDK are no longer - populated from the wire. - decision_context: Context used for the decision Mode values: @@ -3261,15 +3104,12 @@ def execute( # `check_workflow_budget` which already threads the same # contextvar onto the wire body. # - # F03 (2026-08-22) precedence: the `tools` kwarg wins when - # supplied (allows callers like `_enforce_sensitive_tool` to - # forward an explicit list); otherwise fall back to the - # ``_call_tools_var`` contextvar which the F03 fix in - # decorators.py populates from ``fn.__name__`` before this - # method is called. The runtime layer was already reading - # the contextvar — the kwarg simply adds a second entry - # point that didn't exist before (causing TypeError on the - # decorator call site). + # Precedence: the `tools` kwarg wins when supplied (allows + # callers like `_enforce_sensitive_tool` to forward an + # explicit list); otherwise fall back to the + # ``_call_tools_var`` contextvar which the decorators + # populate from ``fn.__name__`` before this method is + # called. if tools is None: from nullrun.context import get_call_tools as _get_call_tools_for_execute @@ -3335,23 +3175,14 @@ def execute( execute_kwargs["action_digest"] = action_digest result = self._transport.execute(**execute_kwargs) - # 2026-09-11: DEF-EXECUTE-CAPTURE-WIRING. The /execute - # require_approval arm mints a FRESH execution_id (server- - # side) for the approval row + writes the binding, then - # echoes the new id back via ``reservation_id`` (mirrored by - # the backend's GateResponse::require_approval constructor — - # see backend/src/enforcement/gate_wire_adapter.rs v3.79+). - # Without this capture below, the contextvar stays at the - # /gate-minted value, and the post-approval /execute re-fire - # (line ~3037) sends the OLD execution_id back to the - # server. ``consume_approved``'s ``WHERE execution_id = $3`` - # predicate then misses the row stamped with the freshly- - # minted one; the diagnostic SELECT walks all alternatives - # without match and falls through to the terminal - # replay-race branch (``APPROVAL_REPLAY_REJECTED``) — the - # SDK raises NR-A015. Capture here is fail-OPEN (drops - # malformed values silently via the helper's UUID parse), - # matching ``check_workflow_budget``'s behaviour. + # The /execute require_approval arm mints a fresh execution_id + # server-side for the approval row + writes the binding, + # then echoes the new id back via ``reservation_id`` + # (mirrored by the backend's GateResponse::require_approval + # constructor — see backend/src/enforcement/gate_wire_adapter.rs). + # Capture here is fail-OPEN (drops malformed values + # silently via the helper's UUID parse), matching + # ``check_workflow_budget``'s behaviour. _capture_server_minted_execution_id(result) # Sync the kwargs dict to the captured id so the post-approval @@ -3591,8 +3422,8 @@ def _build_payload( ``error_code``) so callers can introspect the wire shape for routing/alerting (e.g. ``exc.details["details"]["decision_source"]``). The - ``mapped_class`` shim is appended for back-compat with - callers that branched on the legacy keyword path. + ``mapped_class`` shim is appended for callers that branched + on the exception class name. """ payload = dict(src) payload["mapped_class"] = mapped_name @@ -3633,8 +3464,7 @@ def _build_payload( # (e.g. APPROVAL_VALIDATION_FAILED). There is no # catalog error_code class attr to protect — we # pass the wire code explicitly so self.error_code - # reflects the wire code (back-compat callers - # branch on exc.error_code == "APPROVAL_*"). + # reflects the wire code. payload = _build_payload( wire_details, "NullRunBlockedException" ) @@ -3661,11 +3491,10 @@ def _build_payload( details=payload, ) - # Priority 3: legacy keyword-on-explanation mapping for - # backends that pre-date the structured wire code. Each - # branch picks a synthetic catalog code so legacy - # ``exc.error_code == "NR-B004"``-style branching still - # works for back-compat callers. + # Priority 3: keyword-on-explanation mapping for backends that + # pre-date the structured wire code. Each branch picks a + # synthetic catalog code so the resulting exception still + # maps to a typed class via ``exc.mapped_class``. explanation_lower = explanation.lower() if "budget" in explanation_lower or "exhausted" in explanation_lower: block_code = "NR-B004" @@ -3761,14 +3590,16 @@ def _enrich_event(self, event: dict[str, Any]) -> dict[str, Any]: idem_key = get_server_minted_idempotency_key() if idem_key: - # 2026-08-06 (DEF-SDKWRAP-CHAIN-SOFT-EXECUTION-ID-REUSE-01, + # Compound the reservation id with the span id so a + # chain-soft span's /track stays bound to the right + # reservation even when the span_id changes per + # agent step. span_id = enriched.get("span_id") if span_id and ":" not in idem_key: enriched["idempotency_key"] = f"{idem_key}:{str(span_id)[:16]}" else: enriched["idempotency_key"] = idem_key - # 2026-07-12 (multi-agent span attachment — SDK counterpart at from nullrun.context import get_trace_id as _get_trace_id chain_trace_id = _get_trace_id() @@ -3839,36 +3670,19 @@ def _route_track(self, wire_event: dict[str, Any]) -> None: smid = get_server_minted_execution_id() if not smid: - # v0.16.0 (2026-08-20, backend v3.66.2 alignment): the - # 0.12.0 routing here used to fall back to /track/batch - # (the legacy v1/v2 no-reservation consume path). Backend - # v3.66.2 closed that path with per-event type-aware wire - # validation: any ``llm_call`` event in a batch WITHOUT - # ``reservation_id`` is rejected with 503 - # BUDGET_RECHECK_FAILED (whole-batch fail-CLOSED). Falling - # back here would amplify into a tight retry loop - # producing 503-storm for every call site that forgot to - # pair ``track_llm`` with a prior ``check_workflow_budget`` - # (or ``@protect`` / ``with workflow(...)``). - # # Server-authoritative model (CLAUDE.md §22): an llm_call # event without a paired /check reservation has no - # authoritative budget authority. Don't make up an id — - # drop the event explicitly so the operator sees the gap - # (WARNING + counter) instead of a silent batch loop. + # authoritative budget authority. Drop the event + # explicitly so the operator sees the gap (WARNING + + # counter) instead of a silent batch retry storm. # # Trigger conditions: - # * no /check landed in this scope (legacy v1/v2 path) + # * no /check landed in this scope # * capture expired past the 295s safety window # * /check returned ``decision: "block"`` (no # reservation_id minted on a hard block — see # ``_capture_server_minted_execution_id``) metrics.inc_runtime("dropped_llm_call_no_reservation") - # WARNING not DEBUG: matches the 0.15.2 fix that moved - # ``check_workflow_budget`` synthetic FALLBACK from DEBUG - # to WARNING (CHANGELOG 0.15.2). Operators should see this - # at INFO+ — a missing reservation pairing is a real - # integration bug, not a debug curiosity. logger.warning( "_route_track: dropping llm_call event — no " "server-minted reservation_id in scope (no /check " @@ -4094,17 +3908,11 @@ def _post_auth_with_retry( """POST ``json_body`` to ``url`` with bounded retry on transient failure. - 2026-06-28 audit P2.3: the init path ``POST /api/v1/auth/verify`` - previously did a single bare ``self._transport._client.post(...)`` - call. Backend emits ``503 + Retry-After: 5`` on transient DB - errors (see ``backend/src/proxy/handlers.rs:11346-11351``), which - pre-fix surfaced to the user as ``NR-A001`` ("configuration - issue") even though the SDK was fine and the key was fine — - just a Postgres blip. This helper retries 5xx and network - errors up to ``max_attempts`` total tries, honors - ``Retry-After`` when the backend provides one, and propagates - ``httpx.RequestError`` unchanged on the LAST attempt so the - existing ``except`` arm below can turn it into ``NR-B001``. + Retries 5xx and network errors up to ``max_attempts`` total + tries, honors ``Retry-After`` when the backend provides one, + and propagates ``httpx.RequestError`` unchanged on the LAST + attempt so the existing ``except`` arm below can turn it into + ``NR-B001``. Auth failures (401/403/422) are NOT retried — the API key is wrong on attempt 1 means it's wrong on attempt 3. @@ -4203,7 +4011,8 @@ def _capture_server_minted_execution_id(response: dict[str, Any]) -> str | None: raw = response.get("reservation_id") if isinstance(response, dict) else None if not raw: - # Legacy / v1-v2 backend, or a block response with no + # Block response with no reservation_id; clear so the + # previous scope's value can't leak. clear_server_minted_execution_id() return None @@ -4235,16 +4044,12 @@ def _capture_server_minted_execution_id(response: dict[str, Any]) -> str | None: set_server_minted_execution_id(raw) set_server_minted_reservation_at(_time.monotonic()) - # AUDIT P0-26 (2026-09-05): derive the idempotency_key from the - # SDK-minted operation_id (the value /check just sent on the - # wire) rather than from the server's response-echo. Pre-fix, - # the response-echo was trusted blindly — a misrouted response - # (different execution_id, similar shape) would silently - # overwrite the in-scope idempotency_key. We now assert - # equality when the server echoes a value (defensive parity - # check — the audit wants the SDK to know if the server - # rewrote it for any reason) and fall back to the SDK's own - # operation_id when the server omits the field. + # Derive the idempotency_key from the SDK-minted operation_id + # (the value /check just sent on the wire). The response-echo + # is assert-equal as a defensive parity check (the SDK has to + # know if the server rewrote it for any reason) and falls + # back to the SDK's own operation_id when the server omits the + # field. from nullrun.context import get_operation_id as _get_op_id_for_capture sdk_op_id = _get_op_id_for_capture() @@ -4269,13 +4074,11 @@ def _capture_server_minted_execution_id(response: dict[str, Any]) -> str | None: ) set_server_minted_idempotency_key(sdk_op_id or server_op_id) elif isinstance(sdk_op_id, str) and sdk_op_id: - # Server omitted the echo (pre-v4 backend); trust the SDK. set_server_minted_idempotency_key(sdk_op_id) - # ADR-037 Slice B (2026-08-31, protocol v4): capture the - # wire-evidence echo on the same /check as the execution_id so - # the two values always refer to the same gate decision. Wire- - # additive: pre-v4 backends omit both keys (skip_serializing_if) - # and the captures degrade to None — no false positive. + # Capture the wire-evidence echo on the same /check as the + # execution_id so the two values always refer to the same gate + # decision. Wire-additive: backends that omit both keys + # (skip_serializing_if) degrade to None — no false positive. _capture_wire_evidence(response) logger.debug( "_capture_server_minted_execution_id: captured %s", @@ -4284,13 +4087,12 @@ def _capture_server_minted_execution_id(response: dict[str, Any]) -> str | None: return raw -# ADR-037 Slice B (2026-08-31, protocol v4): wire-evidence echo -# capture helper. Extracted from `_capture_server_minted_execution_id` -# so the two captures share a call site but have distinct log lines -# (debugging: a missing execution_id should not mask a successful -# wire-evidence capture). +# Wire-evidence echo capture helper. Extracted from +# `_capture_server_minted_execution_id` so the two captures share a +# call site but have distinct log lines (debugging: a missing +# execution_id should not mask a successful wire-evidence capture). # -# Wire contract: backend's /gate response now carries `action_digest` +# Wire contract: backend's /gate response carries `action_digest` # (SDK-supplied SHA-256 of canonical business_impact, re-verified # server-side, echoed back) and `policy_hash` (slot reserved for # future Slice D wiring — today always None because the gate doesn't @@ -4306,10 +4108,7 @@ def _capture_wire_evidence(response: dict[str, Any]) -> tuple[str | None, str | Returns ``(action_digest, policy_hash)`` for the caller's log paths; the contextvar side-effect is authoritative. - Both fields default to None on legacy backends (no keys in - the JSON) and on pre-Phase-1 SDKs that never sent - `action_digest` (legacy grant path is fail-CLOSED at v3+ - anyway — the backend sets the field to None on those rows). + Both fields default to None when the backend omits the keys. """ from nullrun.context import ( set_last_gate_action_digest, @@ -4365,17 +4164,15 @@ def _build_v3_track_payload( """Map an enriched llm_call event onto the v3 /track schema. Returns ``None`` when the event cannot be mapped (caller - falls back to legacy batch path). Required ``tokens`` / + falls back to the batch path). Required ``tokens`` / ``workflow_id`` absence is the only failure mode today. """ wf_id = wire_event.get("workflow_id") if not wf_id: # The backend's consume_budget_v3 needs a workflow_id to # attribute the consume to a key+workflow counter; without - # one the consume becomes unattributable. - # ownership binding). A missing workflow_id means the - # SDK never bound the API key to a workflow (legacy - # legacy-no-binding). Fall back. + # one the consume is unattributable. Fall back to the + # batch path (callers handle None). logger.debug( "_build_v3_track_payload: missing workflow_id — cannot shape v3 /track payload" ) diff --git a/src/nullrun/toolbox/langgraph.py b/src/nullrun/toolbox/langgraph.py index 23c3461..6fb861f 100644 --- a/src/nullrun/toolbox/langgraph.py +++ b/src/nullrun/toolbox/langgraph.py @@ -1,13 +1,12 @@ """ LangGraph toolbox helpers for NullRun. -The previous ``wrapper(app, runtime=None)`` escape hatch was removed — ``nullrun.init_or_die()`` (or ``@nullrun.protect`` on the agent function) auto-patches ``langgraph.pregel.Pregel`` via ``nullrun.instrumentation.auto.patch_langgraph_compiled``, which covers every supported LangGraph invocation path. -Migrate to the auto-patch:: +Use the auto-patch:: from nullrun import init_or_die, protect diff --git a/src/nullrun/toolbox/mcp.py b/src/nullrun/toolbox/mcp.py index 4e1ed95..45dd6a2 100644 --- a/src/nullrun/toolbox/mcp.py +++ b/src/nullrun/toolbox/mcp.py @@ -205,7 +205,7 @@ def __init__( # permissive MCP server cannot bypass the operator's # tool-block / budget / approval policies. See the constructor # docstring for the trade-off between the gate path and the - # legacy contextvar-only path. + # contextvar-only path. self._runtime = runtime def _default_list_tools(self) -> Iterable[Any]: @@ -299,10 +299,10 @@ def call_tool( client-specific kwargs without changing the public surface. - Gate enforcement (v3.53 audit #5): when an MCPAdapter is - constructed with ``runtime=`` set, ``call_tool`` routes the - invocation through ``runtime.execute(...)`` (the /api/v1/execute - gate endpoint) BEFORE the underlying MCP client is called. + Gate enforcement: when an MCPAdapter is constructed with + ``runtime=`` set, ``call_tool`` routes the invocation through + ``runtime.execute(...)`` (the /api/v1/execute gate endpoint) + BEFORE the underlying MCP client is called. ``decision="block"`` raises ``NullRunBlockedException`` and the MCP client is NOT called. ``decision="allow"`` proceeds to the MCP client. ``decision="require_approval"`` raises @@ -311,10 +311,9 @@ def call_tool( with ``approval_id=``. When ``runtime`` is None, ``call_tool`` falls through to the - legacy contextvar-only path — the call proceeds without any + contextvar-only path — the call proceeds without any /api/v1/execute round-trip and the next ``@protect``-decorated wrapper picks up the contextvar on its next ``/check`` request. - calls inside ``@protect``-decorated functions. Returns the underlying client's result (when allowed). Raises ``NullRunBlockedException`` on gate block; raises the @@ -372,13 +371,9 @@ def call_tool( # tool-block / budget / approval policies actually apply. if self._runtime is not None: execute_input = arguments if arguments is not None else {} - # 0.18.2: there is no ``mode=`` opt-out — every MCP - # tool call routed through the runtime contacts - # /api/v1/execute unconditionally. The pre-refactor - # ``mode="strict"`` was removed because it was the only - # way to make the audit-grade path actually apply to - # non-sensitive MCP tools; with the inline-mode bypass - # gone, that exemption is no longer possible. + # Every MCP tool call routed through the runtime contacts + # /api/v1/execute unconditionally — there is no + # ``mode=`` opt-out for audit bypass. execute_result = self._runtime.execute( tool_name=tool_name, input_data=execute_input, diff --git a/src/nullrun/transport.py b/src/nullrun/transport.py index 2930e69..74bb69e 100644 --- a/src/nullrun/transport.py +++ b/src/nullrun/transport.py @@ -1123,12 +1123,10 @@ def execute( ) -> dict[str, Any]: """Pre-execution policy evaluation via /api/v1/execute (PRIMARY enforcement point). - Wire contract (revised 2026-09-08, DEFS-SDKEXEC-GATE-FIRST): - /execute REQUIRES a prior /gate call that minted the same - ``execution_id`` and registered the ``execution:{id}`` binding - in Redis. Backend enforcement: - ``backend/src/proxy/http/gate/execute.rs:46-208`` - (DEF-SDKK-022-EXEC-BYPASS, 2026-09-04, RUN_ID=20260904T1500) + Wire contract: /execute requires a prior /gate call that minted the + same ``execution_id`` and registered the ``execution:{id}`` + binding in Redis. Backend enforcement: + ``backend/src/proxy/http/gate/execute.rs`` runs ``HGET execution:{id} ORG_FIELD`` on entry; a miss returns 404 EXECUTION_NOT_FOUND (fail-CLOSED). The SDK therefore MUST thread the execution_id captured by @@ -1251,10 +1249,9 @@ def do_execute_request() -> httpx.Response: # # Fall through to the synthetic block shape if the # envelope is unrecognised (plaintext body, malformed - # JSON, unknown wire code) so behaviour stays - # backwards-compatible for legacy / non-v3 backends. - # `_parse_v3_error_envelope` always returns an - # Exception — it never silently swallows a 4xx. + # JSON, unknown wire code). `_parse_v3_error_envelope` + # always returns an Exception — it never silently + # swallows a 4xx. try: raise _parse_v3_error_envelope(response, "execute") except NullRunApprovalReplayRejectedError: @@ -1795,7 +1792,7 @@ async def _refetch_credentials(self) -> None: # Wire-protocol v3 endpoints # ============================================================================= # - # The v3 wire contract adds six endpoints that the legacy /gate + + # The v3 wire contract adds six endpoints that the /gate + # /execute + /track/batch surface does not cover. Each new method # follows the same shape as the existing `check` method: # @@ -1815,14 +1812,10 @@ def check_v3( request: dict[str, Any], on_transport_error: Callable[[Exception], dict[str, Any]] | str | None = None, ) -> dict[str, Any]: - """Pre-execution gate — wire-protocol v3 (B1 fix 2026-07-04). - - Pre-fix this method POSTed to ``/api/v1/check``. That endpoint - was removed on 2026-06-27 — the handler now returns - ``410 Gone`` with a ``replacement: /api/v1/gate`` hint. The - SDK's ``check `` method already targets ``/api/v1/gate`` and - forwards every v3 wire field — ``chain_id`` - ``chain_op``, ``idempotency_key``, ``stream``. This method + """Pre-execution gate — wire-protocol v3. + + Targets ``/api/v1/gate`` and forwards every v3 wire field — + ``chain_id`` ``chain_op``, ``idempotency_key``, ``stream``. This method is kept as a v3-named alias so existing call sites and tests continue to work; internally it delegates to ``check `` with the same body. @@ -1869,7 +1862,7 @@ def track_single( The wire shape is built by ``runtime._build_v3_track_payload`` (see ``runtime.py:2679-2776``); this method just forwards - whatever dict the caller hands it. The post-fix schema is: + whatever dict the caller hands it. The schema is: Args: request: Consume request body. Must include: @@ -1896,8 +1889,7 @@ def track_single( Returns: Parsed JSON dict from the backend's TrackResponse. NOTE: there is NO top-level ``status`` field on the - wire — the legacy pre-v3 docstring claimed one, but - v3/v4 backends emit + wire — backends emit ``{snapshot, actions_taken, processing_mode, cost_source, confidence, event_id, idempotent_replay, stored_response?}``. SDK callers @@ -1914,16 +1906,12 @@ def track_single( EXECUTION_NOT_BOUND. NullRunAuthenticationError: 401/403. - 2026-07-04 (B2): pre-fix this docstring (and the - surrounding module comment) described a fictitious wire - shape ``{execution_id, actual_cost_cents, api_key_id - cost_source}``. The backend's actual ``TrackRequestRaw`` is + Wire contract: ``TrackRequestRaw`` is ``{workflow_id, tokens, cost_cents,...}``; ``execution_id`` is replaced by ``reservation_id``, ``actual_cost_cents`` is replaced by ``cost_cents`` (the SDK always sends 0 — see ``_WIRE_STRIP_FIELDS``), and ``api_key_id`` is derived server-side from the request auth, not supplied by the SDK. - The docstring now matches the real wire contract. """ body = _signed_request_body(request) headers = self._build_signed_headers(body=body) @@ -1971,12 +1959,10 @@ def cancel( Returns: Parsed JSON dict from the backend's CancelResponse. NOTE: there is NO top-level ``status`` field on the - wire — the legacy pre-v3 docstring claimed one. - v3/v4 backends emit + wire — backends emit ``{execution_id, canceled_at, reservation_released_cents, already_canceled}``. SDK callers branch on the HTTP - status only — do NOT read ``data["status"]`` - (KeyError on every backend >= 3.66.2). + status only — do NOT read ``data["status"]``. """ request: dict[str, Any] = {"execution_id": execution_id} if reason: @@ -2012,13 +1998,13 @@ def consume_approval( """POST /api/v1/approvals/{approval_id}/consume — mark an approved approval row as executed. - Close-orphan fix (ADR-047, 2026-09-21). The original - ``consume_approved`` SQL is only reachable from the orchestrator's - Step 6 inline at backend/src/proxy/http/gate/orchestrator.rs:713, - but mode="inline" tools bypass /execute entirely — leaving the - approval row at status=APPROVED past expires_at. This new + Closes the orphan class on the SDK success path: ``consume_approved`` + SQL is reachable from the orchestrator's Step 6 inline at + backend/src/proxy/http/gate/orchestrator.rs:713, but + mode="inline" tools bypass /execute entirely — leaving the + approval row at status=APPROVED past expires_at. This endpoint is structurally distinct (no execution_id binding per - ADR-046) and closes the orphan class on the SDK success path. + ADR-046). The body is built by Runtime.consume_approval — it always carries ``organization_id`` (C2 closure) and optionally @@ -2510,8 +2496,8 @@ def _extract_error_envelope( ) -> tuple[str, str, dict[str, Any]]: """Pull ``(error_code, message, details)`` from any error envelope. - Drift §3 (2026-07-06): the backend emits three distinct shapes - for non-2xx responses. This helper normalises them into the + The backend emits three distinct shapes for non-2xx responses. + This helper normalises them into the ``(error_code, message, details)`` tuple the rest of ``_parse_v3_error_envelope`` consumes. @@ -2612,12 +2598,10 @@ def _extract_error_envelope( def _safe_json(response: httpx.Response, endpoint: str) -> Any: """Parse a response body as JSON, wrapping parse failures. - DEF-ERRHDL-INVALID-JSON-01 (2026-08-11, RUN_ID 20260811-1): the SDK - code, which leaks internal file paths and the raw broken payload - fragment in tracebacks. This helper wraps the parse failure in - NullRunTransportError with a stable ``error_code`` so callers can - ``except`` cleanly and the user sees a short NullRun-family - message instead of a Python traceback. + The parse failure is wrapped in NullRunTransportError with a + stable ``error_code`` so callers can ``except`` cleanly and the + user sees a short NullRun-family message instead of a Python + traceback. ``body_preview`` is intentionally truncated to 200 chars and the raw ``JSONDecodeError.lineno/colno`` are NOT included in the @@ -3227,8 +3211,8 @@ def _parse_error_envelope( Module-level helper (not a Transport method) so it can be called from background threads that do not carry a Transport instance. - **Audit F-R2-13 (2026-06-22):** no live wire path uses this. It - exists for tests only. See the comment block above. + **Test-only helper:** no live wire path uses this. See the + comment block above. """ status = response.status_code try: diff --git a/src/nullrun/transport_websocket.py b/src/nullrun/transport_websocket.py index 8912d87..a2783e3 100644 --- a/src/nullrun/transport_websocket.py +++ b/src/nullrun/transport_websocket.py @@ -83,10 +83,10 @@ class WebSocketConnection: def _is_acknowledged_state(cls, state: str) -> bool: """Case-insensitive membership check against ``ACKNOWLEDGED_STATES``. - Audit-2026-06-22: added a lowercase fallback so a server - regression to ``"killed"``/``"paused"`` doesn't silently - drop the ACK. Exact PascalCase is still the happy path and - is checked first; the lowercase branch is defensive only. + The lowercase branch is defensive: it protects against a + server regression to ``"killed"``/``"paused"`` that would + otherwise silently drop the ACK. Exact PascalCase is still + the happy path and is checked first. """ if state in cls.ACKNOWLEDGED_STATES: return True @@ -678,11 +678,8 @@ async def _send_ack(self, message_id: str) -> None: """ Send acknowledgment message to server with HMAC signature. - CP7 fix (2026-06-26): previously this ACK was plain JSON - no signature, no timestamp, no api_key. The backend does - not currently verify ACK authenticity (the TODO at - ``backend/src/proxy/http/ws_control.rs:842-848`` is still - open) but adding the signature now means: + The backend does not currently verify ACK authenticity but + the SDK ships the signature now so: * When the backend enables ACK verification, the SDK is already on the wire format it expects — no breaking @@ -737,9 +734,9 @@ async def _send_ack(self, message_id: str) -> None: } # Add HMAC fields when both api_key and secret_key are - # configured. Without secret_key we still send the - # legacy api_keys that don't use HMAC). The backend - # skips verify when signature is absent. + # configured. Without secret_key the SDK sends the + # unsigned X-API-Key only — the backend treats unsigned + # frames as non-HMAC traffic and skips signature verify. if self.api_key and self.secret_key: # The signature covers the canonical bytes of the # body the receiver will hash. We sign the *unsigned* @@ -760,7 +757,7 @@ async def _send_ack(self, message_id: str) -> None: # which would diverge from the signed bytes). await self._conn.send(body_str) else: - # Legacy / pre-HMAC path: plain JSON envelope. + # Unsigned path: plain JSON envelope without HMAC. await self._conn.send(json.dumps(ack)) logger.debug(f"ACK sent for message {message_id}") except Exception as e: From 9595d6ebe4ae015ec4018afe6b3d826b4cba5a6d Mon Sep 17 00:00:00 2001 From: Anatoly Maltsev Date: Fri, 25 Sep 2026 13:12:38 +0400 Subject: [PATCH 03/10] refactor(sdk): strip legacy pre/post-fix framing from docstrings Removes 'pre-fix/post-fix', 'used to', 'previously', 'now we', and audit-cycle prose from runtime.py, transport.py, decorators.py, and 14 other modules. External ticket identifiers (DEF-*, ADR-*, AUDIT P*, IDEM-01, CLOSE-ORPHAN) are preserved as anchors. What stays: ticket refs, ADR numbers, date stamps that tag fixes without prose framing. What goes: 'pre-fix ... pre-fix this method ... Now we' blocks, 'audit ... previous version ... the prior code' commentary, 'was ... is now' chronology in docstrings. 1405 tests pass; -W error::DeprecationWarning clean. Co-Authored-By: Claude Code --- src/nullrun/__init__.py | 14 +- src/nullrun/_handle.py | 22 +- src/nullrun/_registry.py | 2 +- src/nullrun/breaker/circuit_breaker.py | 18 +- src/nullrun/breaker/exceptions.py | 31 +-- src/nullrun/business_impact.py | 23 +- src/nullrun/capabilities.py | 44 ++- src/nullrun/context.py | 17 +- src/nullrun/decorators.py | 34 ++- src/nullrun/instrumentation/auto.py | 101 +++---- src/nullrun/instrumentation/autogen.py | 8 +- src/nullrun/instrumentation/langgraph.py | 91 ++---- src/nullrun/integrations/fastapi.py | 6 +- src/nullrun/observability/error_hooks.py | 19 +- src/nullrun/observability/status.py | 16 +- src/nullrun/runtime.py | 334 +++++++++-------------- src/nullrun/transport.py | 69 ++--- src/nullrun/transport_websocket.py | 26 +- 18 files changed, 342 insertions(+), 533 deletions(-) diff --git a/src/nullrun/__init__.py b/src/nullrun/__init__.py index 01d96e1..565d5be 100644 --- a/src/nullrun/__init__.py +++ b/src/nullrun/__init__.py @@ -273,9 +273,9 @@ def my_agent: raw_key = api_key if api_key is not None else os.getenv("NULLRUN_API_KEY") resolved_key = raw_key.strip() if isinstance(raw_key, str) else None if not resolved_key: - # Layer 1: raise the legacy type (``NullRunAuthenticationError``) - # so user code with ``except NullRunAuthenticationError:`` still - # catches this case, but stamp the structured ``error_code`` / + # Layer 1: raise ``NullRunAuthenticationError`` so user code + # with ``except NullRunAuthenticationError:`` still catches + # this case, but stamp the structured ``error_code`` / # ``user_action`` so a Layer-2 on_error hook (or a # ``except NullRunError:`` clause) can branch on the catalog # value ``NR-C001`` ("configuration: no api_key") without @@ -491,11 +491,9 @@ def my_agent: "NullRunWorkflowKilledError": ("nullrun.breaker.exceptions", "NullRunWorkflowKilledError"), # Four typed exception classes that round-trip the MCP umbrella # codes (ADR-013, frozen-dormant) and the six APPROVAL_DB_* - # sibling codes (DEF-ARFLOW-TOOLNAME-01). Pre-B.1 these all - # collapsed to NullRunBlockedException + the generic NR-X001 - # fallback — cookbook code couldn't branch on the typed arm. - # Post-B.1 each maps to its own typed class so - # ``except NullRunMcpDestructiveBlockedError:`` etc. work. + # sibling codes (DEF-ARFLOW-TOOLNAME-01). Each maps to its own + # typed class so ``except + # NullRunMcpDestructiveBlockedError:`` etc. work. "NullRunMcpDestructiveBlockedError": ("nullrun.breaker.exceptions", "NullRunMcpDestructiveBlockedError"), "NullRunMcpReadonlyBypassBlockedError": ("nullrun.breaker.exceptions", "NullRunMcpReadonlyBypassBlockedError"), "NullRunMcpApprovalRequiredError": ("nullrun.breaker.exceptions", "NullRunMcpApprovalRequiredError"), diff --git a/src/nullrun/_handle.py b/src/nullrun/_handle.py index a47b125..4070bcc 100644 --- a/src/nullrun/_handle.py +++ b/src/nullrun/_handle.py @@ -23,12 +23,12 @@ :func:`nullrun.format_user_message` is included as the headline so end-user scripts don't need to branch on the wire shape. -:class:`nullrun.WorkflowKilledInterrupt` now inherits -from :class:`nullrun.NullRunError` (the 2026-09-08 migration; see the -class docstring), so a bare ``except NullRunError`` would otherwise -swallow the kill signal. ``handle``/``guarded`` explicitly re-raise it -— the kill is a control-plane action, not an SDK failure, and must -reach the top of the agent loop. Non-NullRun exceptions also propagate +:class:`nullrun.WorkflowKilledInterrupt` inherits +from :class:`nullrun.NullRunError` (see the class docstring), so a +bare ``except NullRunError`` would otherwise swallow the kill signal. +``handle``/``guarded`` explicitly re-raise it — the kill is a +control-plane action, not an SDK failure, and must reach the top of +the agent loop. Non-NullRun exceptions also propagate unchanged. ``init_or_die`` exists because:func:`nullrun.init` is typically @@ -187,10 +187,10 @@ def handle(*, exit_code: int = 1): *:class:`nullrun.WorkflowKilledInterrupt` -- kill signals must reach the top of the agent loop, not be swallowed into a graceful exit. Re-raised explicitly inside the ``except NullRunError`` branch - because the 2026-09-08 migration moved ``WorkflowKilledInterrupt`` - onto the ``NullRunError`` MRO (Sentry/OTel ``except Exception`` - handlers should now record kill events; this ``handle`` / - ``guarded`` wrapper opts OUT of that recording on purpose). + because ``WorkflowKilledInterrupt`` sits on the ``NullRunError`` + MRO (Sentry/OTel ``except Exception`` handlers should record kill + events; this ``handle``/``guarded`` wrapper opts OUT of that + recording on purpose). *:class:`KeyboardInterrupt` /:class:`SystemExit` (``BaseException``) -- same reason as the kill signal -- never reach the ``except NullRunError`` branch anyway. @@ -243,7 +243,7 @@ def guarded(fn: Callable[..., T]) -> Callable[..., T]: it is caught, rendered as a user-facing message, and the process exits with code ``1``. ``WorkflowKilledInterrupt`` and other ``BaseException`` subclasses propagate (``handle`` re-raises kill - explicitly, see the 2026-09-08 migration note). + explicitly, see the kill-signal note in the module docstring). Pair with:func:`nullrun.protect` for the standard agent loop:: diff --git a/src/nullrun/_registry.py b/src/nullrun/_registry.py index e1a347a..6fb3aad 100644 --- a/src/nullrun/_registry.py +++ b/src/nullrun/_registry.py @@ -14,7 +14,7 @@ ``decorators._get_or_create_runtime()`` wrote only the decorators slot. Concurrent ``init()`` + ``@protect`` could race and leave one of the three pointing at a dead runtime, dropping ``span_start`` / -``span_end`` events on the floor (see audit 2026-07-05 H2). +``span_end`` events on the floor. The three writers are unified behind a single :class:`RuntimeRegistry` so every consumer reads from one place. diff --git a/src/nullrun/breaker/circuit_breaker.py b/src/nullrun/breaker/circuit_breaker.py index d0195f0..0e357a9 100644 --- a/src/nullrun/breaker/circuit_breaker.py +++ b/src/nullrun/breaker/circuit_breaker.py @@ -401,15 +401,15 @@ def _on_failure(self) -> None: async def _on_success_async(self) -> None: """Async-safe success handler. - DEF-CB-LOCK-UNIFICATION-2026-09-12: switched from - `_async_lock` (asyncio.Lock) to the single `self._lock` - (threading.Lock). Python asyncio is single-threaded; an - `with threading.Lock()` inside an `async def` is safe as - long as the critical section has no `await`. This section - (below) has no `await`, so the sync lock blocks the event - loop for zero observable time on the happy path. The - trade-off (consistency under sync+async concurrency > - minor lock-hold latency) is the point of the fix. + DEF-CB-LOCK-UNIFICATION-2026-09-12: uses the single + ``self._lock`` (threading.Lock) for both sync and async + write paths. Python asyncio is single-threaded; an + ``with threading.Lock()`` inside an ``async def`` is safe + as long as the critical section has no ``await``. This + section has no ``await``, so the sync lock blocks the + event loop for zero observable time on the happy path. A + single ``_async_lock`` (asyncio.Lock) would not coordinate + with concurrent sync threads — hence the unification. """ old_state = self._state with self._lock: diff --git a/src/nullrun/breaker/exceptions.py b/src/nullrun/breaker/exceptions.py index 81f328e..392158a 100644 --- a/src/nullrun/breaker/exceptions.py +++ b/src/nullrun/breaker/exceptions.py @@ -802,10 +802,6 @@ class NullRunBudgetRecheckFailedError(NullRunBudgetError): ``except NullRunBudgetError:`` pattern keeps matching. New ``except NullRunBudgetRecheckFailedError:`` branches on the typed shape (recommended: re-/gate then re-/execute). - - Audit: H6 (2026-08-12). Pre-fix SDK 0.14.x collapsed this code - into a generic ``NullRunBudgetError("Budget authorization failed")`` - with no introspection on the running counter. """ error_code = "NR-B006" @@ -1066,12 +1062,11 @@ class NullRunApprovalExpiredError(NullRunBlockedException): 1. **Wire path** — backend returns APPROVAL_EXPIRED on /execute because the operator's grant TTL elapsed between /gate and /execute. - 2. **Client-side timeout path** (added 2026-09-08, the trigger for - this typed exception migration) — WS push went silent for + 2. **Client-side timeout path** — WS push went silent for ``approval_timeout_seconds`` (default 300s) without an operator decision. The SDK raises this exception instead of the generic ``WorkflowKilledInterrupt`` so cookbook code can catch it - (`except NullRunApprovalExpiredError`) and react with a fresh + (``except NullRunApprovalExpiredError``) and react with a fresh approval request. Cookbook pattern: do NOT retry the same approval_id — request a @@ -1361,9 +1356,8 @@ class WorkflowKilledInterrupt(NullRunError): """ Raised when a workflow is killed by the NullRun control plane. - **2026-09-08 migration**: this class is now an ``Exception`` - subclass (``NullRunError`` parent). Agent recovery code catches - the kill signal via ``except WorkflowKilledInterrupt`` or + Agent recovery code catches the kill signal via + ``except WorkflowKilledInterrupt`` or ``except NullRunWorkflowKilledError`` to surface a structured error to the user with ``error_code=NR-W002`` and ``user_action``. @@ -1403,10 +1397,9 @@ class WorkflowKilledInterrupt(NullRunError): sentry_sdk.capture_exception(exc) raise - Sentry / OpenTelemetry handlers that filter on ``Exception`` will - now record kill events — this is the intended new behavior. Code - that relies on kill being un-catchable by ``except Exception`` is - a regression candidate; see ``docs/kill-contract-migration-2026-09-08.md``. + Sentry / OpenTelemetry handlers that filter on ``Exception`` + record kill events. Code that relies on kill being un-catchable + by ``except Exception`` is a regression candidate. """ error_code = "NR-W002" @@ -1448,16 +1441,16 @@ def __init__( class NullRunWorkflowKilledError(WorkflowKilledInterrupt): """Typed public name for the kill signal. - Subclass of :class:`WorkflowKilledInterrupt` (which remains the - legacy canonical name) so ``except WorkflowKilledInterrupt`` - clauses continue to match. New cookbook code should prefer this - name (``except NullRunWorkflowKilledError``) for typed dispatch. + Subclass of :class:`WorkflowKilledInterrupt` so existing + ``except WorkflowKilledInterrupt`` clauses continue to match. + New cookbook code should prefer this name + (``except NullRunWorkflowKilledError``) for typed dispatch. Wire code ``NR-W002`` (same as parent). Distinct from :class:`NullRunBlockedException` family — kill is a control-plane signal (operator or circuit-breaker), not a gate-decision block. - Cookbook pattern (2026-09-08 migration): + Cookbook pattern: try: agent.run() diff --git a/src/nullrun/business_impact.py b/src/nullrun/business_impact.py index 20c1946..02664b3 100644 --- a/src/nullrun/business_impact.py +++ b/src/nullrun/business_impact.py @@ -1,13 +1,13 @@ """ BusinessImpact + action_digest — minimal wire helpers. -The 0.18.2 SDK is policy-blind. Every ``@protect`` call computes a -single canonical ``NoImpact`` envelope and forwards it to -/execute. The backend's ToolParameters Approval Rules read -values out of ``tool_kwargs`` directly via the rule's -``param_name`` field, so the SDK no longer constructs per-tool -typed impacts (Money, ToolCall). This module exists so the gate -can send a valid (kind, action_digest) pair. +Every ``@protect`` call computes a single canonical ``NoImpact`` +envelope and forwards it to /execute. The backend's ToolParameters +Approval Rules read values out of ``tool_kwargs`` directly via +the rule's ``param_name`` field, so the SDK emits a single +canonical envelope rather than per-tool typed impacts. This +module exists so the gate can send a valid (kind, action_digest) +pair. Field contract mirrored by the backend at ``backend/src/proxy/gate/business_impact.rs``: @@ -55,11 +55,10 @@ def to_wire_dict(self) -> dict[str, Any]: class BusinessImpact: """Top-level BusinessImpact envelope. - 0.18.2: only the ``NoImpact`` payload variant is constructed - on the SDK side. The ``money`` and ``tool_call`` factories - were removed because the SDK no longer stamps per-tool - typed impacts onto functions — every call sends NoImpact and - the backend reads live values out of ``tool_kwargs`` via the + Only the ``NoImpact`` payload variant is constructed on the + SDK side. The ``money`` and ``tool_call`` factories are not + part of the SDK surface — every call sends NoImpact and the + backend reads live values out of ``tool_kwargs`` via the rule's ``param_name``. """ diff --git a/src/nullrun/capabilities.py b/src/nullrun/capabilities.py index 2152efe..27b6d15 100644 --- a/src/nullrun/capabilities.py +++ b/src/nullrun/capabilities.py @@ -22,7 +22,7 @@ - `enforcement_modes_soft` — True means `NULLRUN_SOFT_LIMIT_ENABLED` is on (otherwise the gate downgrades soft → hard) - `heartbeat_time_based` — True means /heartbeat uses the - time-based cadence (vs. chunk-count deprecated v2 path) + time-based cadence (vs. the v2 chunk-count path) - `heartbeat_interval_seconds` — recommended /heartbeat cadence - `heartbeat_skew_tolerance_seconds` — server tolerates heartbeats up to this many seconds past the interval without dedup-rejection @@ -42,18 +42,11 @@ ## Capability history -* 2026-07-06 — fixed P0 (audit §1 capabilities): - - probe URL was ``/health`` (legacy v1/v2); backend exposes the - canonical contract at ``/api/v1/capabilities``. Pre-fix the probe - always returned ``None`` and ``is_v3_ready()`` was always ``False``, - so the capability flags had zero effect on runtime behavior. - - ``parse_capabilities`` read v3-gating fields at top level; backend - nests them under ``capabilities.*``. Pre-fix all four v3 flags - read as ``False`` even on a v3-ready backend. - - Phantom fields ``sdk_min_version`` / ``lua_script_version`` were - read with default fallbacks; backend does ship both (at top - level), so the defaults were harmless but the read path was wrong - (the SDK was reading defaults it never actually used). +The capabilities probe reads from the canonical +``/api/v1/capabilities`` endpoint and parses v3-gating fields from +the nested ``capabilities.*`` payload. Phantom fields +``sdk_min_version`` / ``lua_script_version`` are read at top level +where the backend ships them. """ from __future__ import annotations @@ -76,11 +69,9 @@ # Wire path for the canonical capabilities endpoint. The backend # exposes this at ``/api/v1/capabilities`` (per -# ``backend/src/proxy/http/protocol.rs:189``) since 2025-04. The -# legacy ``/health`` route returns a generic liveness payload — -# it does NOT carry the v3-gating fields, so probing there always -# returned None and ``is_v3_ready()`` was always False, leaving -# every capability flag a no-op at runtime. See capability +# ``backend/src/proxy/http/protocol.rs:189``). The ``/health`` route +# returns a generic liveness payload that does NOT carry the +# v3-gating fields — probing there yields ``None`` for every flag. CAPABILITIES_PATH = "/api/v1/capabilities" @@ -295,11 +286,10 @@ def parse_capabilities(payload: dict[str, Any]) -> ServerCapabilities: Nested wins when both are present so the test fixtures and the canonical shape are unambiguous. - M8 (audit 2026-08-12): shape errors surface via - :func:`_validate_capabilities_payload` before parsing. The - caller (``probe_capabilities``) logs them at WARNING so the - operator sees the malformed payload without silent fallback to - legacy mode. + M8: shape errors surface via :func:`_validate_capabilities_payload` + before parsing. The caller (``probe_capabilities``) logs them at + WARNING so the operator sees the malformed payload without silent + fallback. """ # Shape validation — fail loud on type errors, stay quiet on # missing keys (permissive forward-compat invariant). @@ -367,11 +357,9 @@ def probe_capabilities(api_url: str, timeout: float = 2.0) -> ServerCapabilities messages at ``init ``. The canonical URL is ``{api_url}/api/v1/capabilities`` (per - ``backend/src/proxy/http/protocol.rs:189``). Pre-fix the probe - targeted ``/health`` (legacy v1/v2 status endpoint), which never - carried the v3-gating fields — the probe always returned ``None`` - and ``is_v3_ready()`` was always ``False``, so capability flags - had no effect on runtime behavior. + ``backend/src/proxy/http/protocol.rs:189``). The ``/health`` + status endpoint does NOT carry the v3-gating fields — probing + there returns ``None`` for every flag. """ url = api_url.rstrip("/") + CAPABILITIES_PATH try: diff --git a/src/nullrun/context.py b/src/nullrun/context.py index f649fe0..4afa774 100644 --- a/src/nullrun/context.py +++ b/src/nullrun/context.py @@ -15,7 +15,7 @@ # below; ``@protect`` (decorators.py:441) and any other writer must # call BOTH sides so runtime readers (``get_trace_id`` / # ``get_span_id``) and SpanContext readers (``get_current_span``) see -# the same trace id. See audit_ui/UI-UX-AUDIT-REPORT.md F-19. +# the same trace id. from .tracing import ( SpanContext, _current_span, @@ -588,15 +588,14 @@ def clear_operation_id() -> None: def get_last_gate_action_digest() -> str | None: """Return the `action_digest` echoed by the last /gate response, or - ``None`` if no echo captured in scope (legacy backend, or a /check - that didn't carry a typed business impact). + ``None`` if no echo captured in scope. Wire-additive — pre-v4 backends omit the field entirely (``skip_serializing_if = "Option::is_none"`` on the backend); a - v4 SDK connecting to a v3 backend reads None and behaves like - pre-Slice-B. No false positive. + v4 SDK connecting to a v3 backend reads None and proceeds + without it. No false positive. - See ADR-037 Slice B (2026-08-31) for the wire contract. + See ADR-037 Slice B for the wire contract. """ return _last_gate_action_digest_var.get() @@ -611,7 +610,7 @@ def get_last_gate_policy_hash() -> str | None: `audit_drain.rs:301`). Today this field is always None on the wire, so this getter is informational only. - See ADR-037 Slice B (2026-08-31) for the wire contract. + See ADR-037 Slice B for the wire contract. """ return _last_gate_policy_hash_var.get() @@ -622,7 +621,7 @@ def set_last_gate_action_digest(value: str | None) -> None: Called by ``runtime._capture_wire_evidence`` immediately after ``_capture_server_minted_execution_id`` — the two captures share the same lifetime (one /check → one execution_id + one - action_digest). See ADR-037 Slice B (2026-08-31). + action_digest). See ADR-037 Slice B. """ _last_gate_action_digest_var.set(value) @@ -634,7 +633,7 @@ def set_last_gate_policy_hash(value: str | None) -> None: this is always set to None on the wire; this setter is the forward-compatible hook for Slice D. - See ADR-037 Slice B (2026-08-31). + See ADR-037 Slice B. """ _last_gate_policy_hash_var.set(value) diff --git a/src/nullrun/decorators.py b/src/nullrun/decorators.py index 7b5c699..28090a5 100644 --- a/src/nullrun/decorators.py +++ b/src/nullrun/decorators.py @@ -569,20 +569,20 @@ def _protect_body(args: tuple[Any, ...], kwargs: dict[str, Any], unify_block: bo runtime = _get_or_create_runtime() span = _next_span() token = set_span(span) - # the legacy ``_trace_id_var`` / ``_span_id_var`` so the - # runtime's ``_enrich_event`` (which reads via - # ``get_trace_id()`` / ``get_span_id()`` for cost events - # AND for ``parent_trace_id`` derivation at runtime.py:2967) - # emits events tagged with the SAME trace_id / - # span_id as SpanContext. Without this mirror a bare + # Mirror the trace_id / span_id into ``_trace_id_var`` / + # ``_span_id_var`` so the runtime's ``_enrich_event`` (which + # reads via ``get_trace_id()`` / ``get_span_id()`` for cost + # events AND for ``parent_trace_id`` derivation at + # runtime.py:2967) emits events tagged with the SAME + # trace_id / span_id as SpanContext. Without this mirror a bare # ``@protect`` (no enclosing ``with workflow``) sees a # tree-break: span_start carries SpanContext.trace_id while # llm_call / tool_call carries a freshly generated trace_id. # Token-based so a nested ``@protect`` inside an outer # ``@protect`` (or inside ``with workflow``) # restores the outer trace/span on reset. - trace_legacy_token = set_trace_id(span.trace_id) - span_legacy_token = set_span_id(span.span_id) + trace_token = set_trace_id(span.trace_id) + span_token = set_span_id(span.span_id) # ``fn.__name__`` when the user did NOT explicitly call # ``set_call_context(tools=...)``. The F01 fix # (``runtime.execute`` body at runtime.py:2746-2760 and the @@ -595,8 +595,7 @@ def _protect_body(args: tuple[Any, ...], kwargs: dict[str, Any], unify_block: bo # /015/016/017) never reach the approval_rule_eval step. # Token-based so a nested @protect inside an outer @protect # (or inside ``with workflow``) restores the outer contextvar - # on reset — same shape as the legacy - # ``_trace_id_var`` / ``_span_id_var`` resets above. + # on reset — same shape as the trace/span token resets above. _existing_call_tools = get_call_tools() if not _existing_call_tools: call_tools_token: Token[tuple[str, ...]] | None = _call_tools_var.set( @@ -662,13 +661,12 @@ def _protect_body(args: tuple[Any, ...], kwargs: dict[str, Any], unify_block: bo raise finally: reset_span(token) - # F-19 follow-up: token-based reset matches the legacy - # ``_trace_id_var`` / ``_span_id_var`` pattern (paired - # with their tokens set above). Order does not matter; - # both resets restore the prior contextview regardless - # of which one runs first. - reset_trace_id(trace_legacy_token) - reset_span_id(span_legacy_token) + # F-19 follow-up: token-based reset matches the trace/span + # token pattern (paired with the tokens set above). Order + # does not matter; both resets restore the prior + # contextvar regardless of which one runs first. + reset_trace_id(trace_token) + reset_span_id(span_token) # F03 follow-up: reset the per-call tools contextvar if # we set it. Outer ``with workflow`` / nested @protect # prior value restored; bare @protect leaves the @@ -782,7 +780,7 @@ def _run_tool_policy_gate( # Wire-shape compatibility: ``business_impact`` stays None # on /execute when no per-tool typed impact is extracted - # (the legacy bare @protect shape — backend reads only + # (the bare @protect shape — backend reads only # ``action_digest`` + ``kwargs`` for ToolParameters Approval # Rules). The ``action_digest`` is still computed against # the canonical NoImpact envelope so the Phase-1+ wire-shape diff --git a/src/nullrun/instrumentation/auto.py b/src/nullrun/instrumentation/auto.py index 2103931..e72a00c 100644 --- a/src/nullrun/instrumentation/auto.py +++ b/src/nullrun/instrumentation/auto.py @@ -396,26 +396,23 @@ def _cohere_extractor(body: bytes, status: int) -> ExtractedUsage | None: response.usage.{tokens, input_tokens, output_tokens}. Note: Cohere streaming has no usage in stream — only non-streaming - responses carry it. Documented in the plan. + responses carry it. - 2026-07-13: v2 has THREE schema changes the SDK - silently missed: + Three notable schema choices: - 1. ``tool_calls`` live under ``message.tool_calls`` (not at - the top level). v1 still used top-level ``tool_calls``; - v2 moved them into the assistant message envelope. The - top-level path is preserved as a fallback for v1 + the - rare v2 adapter that lifts the field back up, so neither - version is broken by the new primary path. + 1. ``tool_calls`` live under ``message.tool_calls``. + The top-level path is preserved as a fallback so neither + v1 (top-level) nor v2 (nested) is broken by the new + primary path. - 2. ``usage.tokens.cached_tokens`` is the v2 cache hit counter - (Cohere's inference cache). Previously always read as 0. + 2. ``usage.tokens.cached_tokens`` is the cache hit counter + (Cohere's inference cache). 3. ``finish_reason`` values are UPPERCASE (``COMPLETE | MAX_TOKENS | STOP_SEQUENCE | TOOL_CALL | - ERROR | TIMEOUT``); the v1 vocabulary was lowercase. The - ``_normalize_finish_reason`` helper lower-cases before - mapping so both vocabularies work. + ERROR | TIMEOUT``); the ``_normalize_finish_reason`` + helper lower-cases before mapping so both vocabularies + work. """ if status >= 400 or not body: return None @@ -661,9 +658,9 @@ def _bedrock_extractor(body: bytes, status: int) -> ExtractedUsage | None: def _extract_model_from_request_body(request: httpx.Request) -> str | None: - """2026-06-28 (Issue 2 fix): fall back to the ``model`` field embedded - in the LLM request body when the response body extractor returned - ``None`` for ``model``. + """Fall back to the ``model`` field embedded in the LLM request + body when the response body extractor returned ``None`` for + ``model``. The user typically passes ``ChatOpenAI(model="gpt-4.1-mini")`` and that string appears in the request body's ``model`` field — even if @@ -728,15 +725,14 @@ def _check_kill_before_send(runtime: Any, request: httpx.Request) -> None: - no workflow can be resolved (no active context, no API key binding) - the cached state is anything other than Killed / Paused - Note: prior to 0.3.0 this also short-circuited in - `local_mode` (no api_key). The local_mode branch is gone because - api_key is now required at runtime construction — every runtime - has a remote control plane to consult. + Note: api_key is required at runtime construction — every + runtime has a remote control plane to consult. There is no + ``local_mode`` short-circuit. Raises: - NullRunWorkflowKilledError: state == "Killed" (2026-09-08: - typed signal with error_code=NR-W002 + user_action; - subclass of WorkflowKilledInterrupt.) + NullRunWorkflowKilledError: state == "Killed" — typed + signal with error_code=NR-W002 + user_action; subclass + of WorkflowKilledInterrupt. WorkflowPausedException: state == "Paused" """ if runtime is None: @@ -757,7 +753,7 @@ def _check_kill_before_send(runtime: Any, request: httpx.Request) -> None: state = runtime._remote_state_for(workflow_id) if hasattr(runtime, "_remote_state_for") else getattr(runtime, "_remote_states", {}).get(workflow_id, {}) state_name = state.get("state", "Normal") if state_name == "Killed": - # code can `except NullRunWorkflowKilledError`; legacy + # code can `except NullRunWorkflowKilledError`; # `except WorkflowKilledInterrupt` still matches (subclass). from nullrun.breaker.exceptions import NullRunWorkflowKilledError raise NullRunWorkflowKilledError( @@ -1250,15 +1246,14 @@ def _wrap_async_init(self: httpx.AsyncClient, *args: Any, **kwargs: Any) -> None def _wrap_pre_existing_httpx_clients(runtime: Any) -> tuple[int, int]: - """Find httpx clients created before ``patch_httpx`` ran and wrap their - transports in NullRun's transports. - - Audit 2026-06-29 (init-ordering hazard): the typical sequence - - llm = ChatOpenAI(model=...) # builds internal httpx.Client - nullrun.init(api_key=...) # installs the __init__ patch - - leaves ``llm``'s internal client with the unpatched transport. + """Find httpx clients created before ``patch_httpx`` ran and wrap + their transports in NullRun's transports. + + Init-ordering hazard: when an HTTP-using client library is + instantiated *before* ``nullrun.init``, its internal + ``httpx.Client`` retains the unpatched transport. This back-fill + picks up those clients so the same NUL-1 control surface + applies to them. New ``httpx.Client `` constructions are auto-wrapped by the class-level patch; this sweep is the back-fill. @@ -1604,12 +1599,8 @@ def _emit_from_agents_result(runtime: Any, result: Any) -> None: name = (tc.get("function") or {}).get("name") if name: tool_names.append(name) - # used to be put on the wire as-is — when the agents SDK - # didn't populate the span's ``model`` field (some - # custom tracer configs), this shipped ``model=None`` → - # backend ``unwrap_or("default")`` → fallback warning. - # We also try ``usage["model"]`` (OpenAI usage payload - # sometimes carries the resolved model id) and + # Falls through to ``usage["model"]`` (OpenAI usage + # payload sometimes carries the resolved model id) and # ``span["response_metadata"]["model_name"]`` (langchain- # style metadata block on the span). Empty / None are # dropped — only set ``model`` when we have a real value. @@ -2066,28 +2057,14 @@ def _emit_streaming_skipped( `_extract_model_from_request_body` (sync-only, mirrors `_emit`'s pattern at lines 735-739). - Audit 2026-06-29 (ghost-event dedup): the previous version - emitted the event unconditionally and without a `_fingerprint`. - Two consequences: - 1. When the body read fails for an external reason - (double-consume by langchain-openai, an upstream that - already drained the stream), the SDK produced an - `llm_call` with `tokens=0, model=None` — i.e. no useful - signal — that still reached the wire. The backend's - `into_track_request_v2` handler gate (handler.rs:2046) - rejected these with HTTP 422, but the cost-pipeline - belt-and-suspenders backstop still logged every one as - `cost_pipeline_missing_model_total` and stamped the 1-cent - surcharge. Operators saw 30+ ERROR lines per `app.invoke ` - for a workload that actually had 6 real LLM calls. - 2. Because no `_fingerprint` was attached, the dedup LRU at - `runtime.track ` could not collapse this emission with - any sibling emission for the same call. - Fix: drop the event entirely when we cannot recover a usable - `model` (the request body has been consumed or doesn't carry - the field — same signature as a body that genuinely cannot be - inspected), and attach a deterministic `_fingerprint` when we - do emit so dedup collapses repeats from the same call site. + Ghost-event dedup: dropped events carry a deterministic + ``_fingerprint`` so the runtime's dedup LRU collapses + repeated emissions for the same call site. Events where a + usable ``model`` cannot be recovered (request body has been + consumed or doesn't carry the field — same signature as a + body that genuinely cannot be inspected) are dropped + entirely to avoid the cost-pipeline's + ``cost_pipeline_missing_model_total`` surcharge. """ # We always emit the streaming-skipped event regardless of # whether ``_extract_model_from_request_body`` recovered a model. diff --git a/src/nullrun/instrumentation/autogen.py b/src/nullrun/instrumentation/autogen.py index bdcc7f4..b161575 100644 --- a/src/nullrun/instrumentation/autogen.py +++ b/src/nullrun/instrumentation/autogen.py @@ -101,13 +101,7 @@ def _wrap_create(self: Any, *args: Any, **kwargs: Any) -> Any: getattr(usage, "total_tokens", 0) or 0 ) or (prompt + completion) if prompt or completion or total: - # used to come only from ``self.model`` with a - # bare ``None`` fallback — if the autogen client - # didn't expose a ``model`` attribute (some - # subclass / wrapper / mock provider), the wire - # event carried ``model=None`` → backend - # ``unwrap_or("default")`` → fallback warning → - # DEFAULT_RATE. Now we try three sources in + # Model extraction tries three sources in # priority order, matching the multi-source # pattern in langgraph's # ``_extract_model_from_response``: diff --git a/src/nullrun/instrumentation/langgraph.py b/src/nullrun/instrumentation/langgraph.py index 9fd7e82..cbeca53 100644 --- a/src/nullrun/instrumentation/langgraph.py +++ b/src/nullrun/instrumentation/langgraph.py @@ -461,26 +461,22 @@ def on_llm_start(self, serialized: Any, prompts: Any, **kwargs: Any) -> None: """ Called when LLM call starts. - 2026-07-12 (multi-agent span attachment): open a child span - for the LLM call so the cost event emitted by ``on_llm_end`` - carries the parent chain's ``trace_id``. Pre-fix this hook - was a no-op — ``on_llm_end`` then fell through to - ``runtime.track()`` which generates a fresh ``trace_id`` per - event, breaking the parent-child span hierarchy on the - server side. The frontend "Recent executions" panel then - showed 4/5 rows with ``cost_cents=0 / tokens=0`` because the - per-row unified SELECT keyed the JOIN on a per-call fresh - ``trace_id`` that no other row in the workflow had. + Multi-agent span attachment: open a child span for the LLM + call so the cost event emitted by ``on_llm_end`` carries + the parent chain's ``trace_id``. A no-op here would force + ``on_llm_end`` to emit events with a per-call fresh + ``trace_id``, breaking the parent-child span hierarchy on + the server side. Behaviour: create a child span from the active framework - span (``@protect``-set via `set_span` or a higher-level - ``on_chain_start`` via `_active_runs[parent_run_id]`). - Record the SpanContext under the LangChain ``run_id`` key so - ``on_llm_end`` can look it up. The ``run_id`` callback kwargs - are present on langchain >= 0.1; missing run_id is logged - and we fall back to creating a synthetic root (best-effort, - matches the legacy behaviour so we never throw out of the - LangChain callback chain). + span (``@protect``-set via ``set_span`` or a higher-level + ``on_chain_start`` via ``_active_runs[parent_run_id]``). + Record the SpanContext under the LangChain ``run_id`` key + so ``on_llm_end`` can look it up. The ``run_id`` callback + kwargs are present on langchain >= 0.1; missing run_id is + logged and we fall back to creating a synthetic root + (best-effort, so we never throw out of the LangChain + callback chain). """ run_id = kwargs.get("run_id") parent_run_id = kwargs.get("parent_run_id") @@ -569,32 +565,18 @@ def on_llm_end(self, response: Any, **kwargs: Any) -> None: Extracts usage data and sends to backend for cost computation. Does NOT compute cost - backend is source of truth. - Audit 2026-06-28 (SDK↔backend wire): the previous version pulled - ``model_name`` exclusively from ``invocation_params`` with a - hard fallback to the literal string ``"unknown"``. When langchain - 1.x stopped forwarding ``invocation_params`` to ``on_llm_end`` - every track event carried ``model="unknown"`` and the backend - cost pipeline fell through to ``DEFAULT_RATE``. Now we try - ``invocation_params.model_name`` first, then fall back to - reading the real model id from the response object itself - (``response.response_metadata['model_name']`` or the AIMessage - on the LLMResult generation). ``"unknown"`` is now a true last - resort, not the common case. - - Audit 2026-06-29 (ghost-event dedup): the previous version of - this method did NOT attach a ``_fingerprint`` to the event - before forwarding it to ``runtime.track ``. Because the - dedup LRU only collapses events whose ``_fingerprint`` - matches, the LangChain callback emission was never deduped - against the sibling emission from the httpx transport - (``NullRunSyncTransport._emit``), even though both observers - fire for the same LLM call. The net effect on a typical - ``app.invoke `` with 6 LLM calls was 6-12 duplicate - ``llm_call`` events on the wire (instead of 6), plus extra - cost-pipeline ERROR noise from ``_emit_streaming_skipped`` - for body-read failures. The fix derives a stable fingerprint - from the LangChain run_id + invocation_params + response id - so the dedup LRU can collapse these emissions. + Reads ``model_name`` from ``invocation_params`` first, then + falls back to the response object itself + (``response.response_metadata['model_name']`` or the + AIMessage on the LLMResult generation). ``"unknown"`` is + the last-resort fallback when neither source carries the + model — its presence surfaces as an alertable gap rather + than a silent default. + + Attaches a stable ``_fingerprint`` derived from the + LangChain run_id + invocation_params + response id so the + runtime dedup LRU collapses sibling emissions between the + LangChain callback and the httpx transport hook. """ try: # Extract provider/model from invocation params first, then @@ -978,24 +960,9 @@ def _extract_model_from_response(response: Any) -> str | None: """Best-effort model extraction mirroring ``_get_finish_reason``. Returns the first non-empty value found, or ``None`` if every known - source is empty / malformed. - - Audit 2026-06-29 (SDK↔backend wire: silent zero-billing): the chain - was checked top-to-bottom and silently returned ``None`` whenever - none of the four known locations carried the model. The backend - then ``unwrap_or("default")``'d to ``DEFAULT_RATE`` and every call - was recorded as ≈$0. We now: - - - promote ``response.llm_output['model_name']`` (the location - langchain-openai 1.x uses for the date-suffixed model id - ``gpt-4.1-mini-2025-04-14``) to step 1, ahead of the - ``response_metadata`` step that langchain 0.x used - - add ``response.llm_output['model']`` and a generic - "any key containing 'model'" sweep so non-OpenAI wrappers - (proxies, custom chat models) still get attributed - - log a DEBUG line on the None path so an operator who sees - the wire warning in the backend can correlate it to the - observation site that produced the event. + source is empty / malformed. A ``None`` return path is logged at + DEBUG so an operator who sees a wire warning in the backend can + correlate it to the observation site that produced the event. Sources checked, in order: diff --git a/src/nullrun/integrations/fastapi.py b/src/nullrun/integrations/fastapi.py index a281510..2ee8643 100644 --- a/src/nullrun/integrations/fastapi.py +++ b/src/nullrun/integrations/fastapi.py @@ -104,9 +104,9 @@ def chat(message: str) -> str: _KILL_STATUS = 503 -# Locale negotiation helpers removed — the catalog is English-only and -# ``format_user_message`` no longer takes a ``locale=`` kwarg. Reserved -# for a future locale-pack release if/when a non-English catalog lands. +# Locale negotiation helpers reserved for a future locale-pack +# release. The catalog is English-only and ``format_user_message`` +# takes no ``locale=`` kwarg. def _build_headers(exc: BaseException) -> dict[str, str]: diff --git a/src/nullrun/observability/error_hooks.py b/src/nullrun/observability/error_hooks.py index 57c544d..17d9bf6 100644 --- a/src/nullrun/observability/error_hooks.py +++ b/src/nullrun/observability/error_hooks.py @@ -11,9 +11,9 @@ hook BEFORE the exception propagates. The hook sees the same ``NullRunError`` and an ``ErrorContext`` describing where in the lifecycle the error happened. Multiple hooks are supported. Hook -exceptions are caught and logged at DEBUG (design discussion -2026-06-24 — visible when DEBUG logging is on, silent at -INFO/CRITICAL so a misbehaving hook does not break production). +exceptions are caught and logged at DEBUG — visible when DEBUG +logging is on, silent at INFO/CRITICAL so a misbehaving hook does +not break production. What does NOT fire the hook: @@ -53,7 +53,7 @@ # policy_fetch — GET /api/v1/orgs/{org}/policies # execute — POST /api/v1/execute (gate decision) # track — POST /api/v1/track (event ingest) -# gate — POST /api/v1/gate (legacy pre-flight) +# gate — POST /api/v1/gate (pre-flight) # check — POST /api/v1/check (budget pre-flight) # org_status — get_org_status # ws — WebSocket control-plane message handling @@ -197,13 +197,12 @@ def emit_error(err: Any, ctx: ErrorContext) -> None: Called from raise sites in the SDK immediately BEFORE the ``raise`` statement, so the hook sees the fully-constructed - exception while the call stack is still live (design - decision C, 2026-06-24). + exception while the call stack is still live. - Hook exceptions are caught and logged at DEBUG (design - decision 2026-06-24: silent at INFO/CRITICAL so a - misbehaving hook does not break production, visible when - DEBUG logging is on so debugging the hook itself is easy). + Hook exceptions are caught and logged at DEBUG — silent at + INFO/CRITICAL so a misbehaving hook does not break production, + visible when DEBUG logging is on so debugging the hook itself + is easy. Snapshot the hook list under the lock so a concurrent unregister during dispatch does not mutate the iteration. diff --git a/src/nullrun/observability/status.py b/src/nullrun/observability/status.py index fa56f51..b7c5b03 100644 --- a/src/nullrun/observability/status.py +++ b/src/nullrun/observability/status.py @@ -42,10 +42,10 @@ to the user. * ``"ok"`` — everything healthy. This is the steady state. -Note (0.7.0): SDK no longer maintains a local ``Policy`` cache. All -enforcement decisions arrive from the backend via ``/gate`` and -``/execute``. The "cached policy" degradation state from prior -versions is gone — SDK is either talking to the backend or it isn't. +Note: All enforcement decisions arrive from the backend via +``/gate`` and ``/execute``. The SDK does not maintain a local +``Policy`` cache — it is either talking to the backend or it +isn't. """ from __future__ import annotations @@ -112,10 +112,10 @@ class WorkflowState: user can read ``status.workflow_state.state`` and know whether the body will run on the next call. - CP1 fix (2026-06-26): the backend WsWorkflowState enum has 5 - treated as Normal. The SDK now handles all 5 explicitly in - ``runtime.check_control_plane``; this dataclass reflects the - full set so the operator-facing status mirrors reality. + The backend WsWorkflowState enum has 5 distinct values; the SDK + handles all 5 explicitly in ``runtime.check_control_plane`` and + this dataclass reflects the full set so the operator-facing + status mirrors reality. """ workflow_id: str diff --git a/src/nullrun/runtime.py b/src/nullrun/runtime.py index ee2de8f..30266fa 100644 --- a/src/nullrun/runtime.py +++ b/src/nullrun/runtime.py @@ -1940,19 +1940,16 @@ def check_workflow_budget(self) -> None: # always-skipped). metrics.inc_runtime("check_calls") - # AUDIT P0-27 (2026-09-05): hoist the operation_id mint. - # Pre-fix, /check minted its own UUID v4 here AND /execute - # minted a separate UUID v4 — same logical action produced - # two unrelated backend bindings. The mint now lives in a - # single contextvar (``_operation_id_var``) so /check, - # /execute, and /track all share the SAME value for one - # logical action. /track consumes this value via the - # ``server_minted_idempotency_key`` contextvar, which the - # /check capture path (see ``_capture_server_minted_...``) - # sets from ``get_operation_id()`` rather than from the - # response-echo (P0-26 — the echo could silently overwrite - # with whatever the server returned, even on a different - # execution's response). + # The operation_id is hoisted into the ``_operation_id_var`` + # contextvar so /check, /execute, and the post-approval re-fire + # within ONE logical action share the SAME value. /track reads + # ``get_operation_id()`` via ``_capture_server_minted_*`` rather + # than the response echo (the echo could silently overwrite with + # whatever the server returned, even on a different execution). + # Minting fresh per call avoids the cross-action reuse that + # trips IDEM-01 — a guard that returned the cached op_id + # when set would leak the scope's first op_id across + # subsequent ``check_workflow_budget()`` invocations. from nullrun.context import ( get_operation_id as _get_op_id_for_check, ) @@ -1960,34 +1957,6 @@ def check_workflow_budget(self) -> None: set_operation_id as _set_op_id_for_check, ) - # AUDIT P0-27 (2026-09-05) wire-binding invariant is - # preserved: /check + /execute (and the post-approval - # re-fire) within ONE logical action share the SAME - # op_id. The original implementation minted once per - # scope via the contextvar; /execute reads the freshly- - # minted value via get_operation_id() because we just - # stashed it here. /track works on the server-minted - # execution_id, NOT op_id, so it is independent. - # - # Wire-binding invariant: /check + /execute (and the - # post-approval re-fire) within ONE logical action share - # the SAME op_id. The original implementation minted once - # per scope via the contextvar; /execute reads the - # freshly-minted value via get_operation_id() because we - # just stashed it here. /track works on the server-minted - # execution_id, NOT op_id, so it is independent. - # - # Mint-fresh-per-call removes the cross-action reuse that - # trips IDEM-01 (the prior `if op_id is None:` guard - # leaked the scope's first op_id across subsequent - # ``check_workflow_budget()`` invocations). We still read - # the contextvar first to keep the within-action binding - # observable and to surface any unexpected caller that - # pre-populates ``operation_id`` (e.g. test fixtures). - # The read result is intentionally unused: /execute, - # which runs synchronously in the SDK after /check, - # reads the freshly-stashed value below via - # ``get_operation_id()`` — that is the binding. op_id = _get_op_id_for_check() op_id = str(uuid.uuid4()) _set_op_id_for_check(op_id) @@ -2010,42 +1979,31 @@ def check_workflow_budget(self) -> None: # Prefer the user-set contextvar (explicit `with workflow(...)` # block), fall back to the API key's bound workflow. Returns - # None only on legacy keys that have never been - # workflow-bound -- in that case the check is silently - # skipped. + # None only on API keys that have never been workflow-bound + # -- in that case the check is silently skipped. # - # L6 audit 2026-08-12: workflow_id is intentionally NOT - # forwarded to the /gate wire body. The server derives it - # server-side from the API key's 1:1 binding (CLAUDE.md §12 - # "1 API key = 1 workflow" invariant). Adding it to the wire - # would be additive telemetry only — the per-workflow budget - # aggregator (`wf:{id}:monthly_cost` + `wf:{id}:bp:{ts}`) - # operates on the server's binding, not on a client-claimed - # value. The `mode='hard'` corner the audit flagged is a - # non-issue: the field is omitted unconditionally, regardless - # of enforcement_mode. Workflow_id flows into /track + /events - # via `_enrich_event` (line ~2697) for cost attribution; /gate - # intentionally keeps the wire minimal. + # workflow_id is intentionally NOT forwarded to the /gate + # wire body. The server derives it server-side from the API + # key's 1:1 binding (CLAUDE.md §12 "1 API key = 1 workflow" + # invariant). The per-workflow budget aggregator + # (``wf:{id}:monthly_cost`` + ``wf:{id}:bp:{ts}``) operates + # on the server's binding, not on a client-claimed value. + # The field is omitted unconditionally regardless of + # enforcement_mode. Workflow_id flows into /track + /events + # via ``_enrich_event`` (line ~2697) for cost attribution; + # /gate intentionally keeps the wire minimal. workflow_id = self._resolve_workflow_id(get_workflow_id()) if not workflow_id: return - # Use the real model name from the call context if the user - # set it via `set_call_context(model=...)` (or via a future - # `with workflow(..., model=...)` block). Earlier SDK - # versions always sent the literal string "budget-precheck" - # -- a fake sentinel that forced backend pricing lookup to - # fall through to the default rate, so projected_cost was - # always computed against the wrong per-model rate and - # blocked any future per-model budget tier (model-specific - # caps) from being enforced correctly. Sending `None` is - # fine -- backend `calculate_projected_cost` defaults when + # Use the real model name from the call context if the user set + # it via `set_call_context(model=...)`. Sending `None` is + # fine — backend `calculate_projected_cost` defaults when # model is unset, and tool_block enforcement on /gate is # best-effort when no tools are sent. call_model = get_call_model() call_tools = get_call_tools() - # 2026-07-02 (v0.11.0): forward chain context for soft-mode # Chain context for soft-mode enforcement. chain_id = get_chain_id() chain_op = get_chain_op() @@ -2103,7 +2061,6 @@ def check_workflow_budget(self) -> None: check_req["chain_id"] = chain_id check_req["chain_op"] = chain_op if chain_op != "auto" else None - # 2026-07-02 (v0.11.0): idempotency key. # operation_id is also the idempotency key — /check and /track # with the same operation_id are treated as a single logical # action by the backend's binding key. @@ -2268,12 +2225,8 @@ def check_workflow_budget(self) -> None: # `approval_timeout_seconds` (i64) and # `approval_expires_at` (ISO8601 string) are exposed; # we prefer the integer field because it's directly - # usable in `event.wait(timeout=...)`. If the backend - # only sent the ISO8601 string (e.g. an older proxy - # rewriting the field), fall through to the env - # default rather than try to parse it inline -- the - # field is documented as informational for UI/logs - # and isn't required for the SDK's wait math. + # usable in `event.wait(timeout=...)`. Falls back to + # the env default when the field is absent. server_timeout = _validate_approval_timeout( response.get("approval_timeout_seconds"), log_prefix="check_workflow_budget", @@ -2440,21 +2393,16 @@ def stop() -> None: def heartbeat(self, chain_id: str) -> dict[str, Any]: """POST /api/v1/heartbeat — extend a chain's idle TTL. - Single-shot wrapper around ``Transport.heartbeat`` matching the - ``chain_end`` / ``cancel_execution`` public-API pattern - (DEF-HEART-01, 2026-09-11). Use this for one-off TTL extensions; - use ``ping_chain`` when you want a wall-clock scheduler that calls - this method every N seconds. - - The wire body is ``{"chain_id": chain_id}`` — HMAC headers carry - ``organization_id`` + ``trace_id`` automatically via - ``_build_signed_headers``, so no extra kwargs are needed at the - transport layer (mirrors the simpler heartbeat shape vs. chain_end's - ``organization_id``/``trace_id`` injection). + Single-shot wrapper around ``Transport.heartbeat`` matching + the ``chain_end`` / ``cancel_execution`` public-API pattern. + Use this for one-off TTL extensions; use ``ping_chain`` when + you want a wall-clock scheduler that calls this method every + N seconds. - The transport layer already raises ``NullRunTransportError( - NETWORK_ERROR, "heartbeat")`` for network errors (transport.py:2077), - so no reclassification is needed at this layer. + The wire body is ``{"chain_id": chain_id}`` — HMAC headers + carry ``organization_id`` + ``trace_id`` automatically via + ``_build_signed_headers``, so no extra kwargs are needed at + the transport layer. Args: chain_id: Active chain_id (UUID v4) registered via @@ -2939,16 +2887,10 @@ def execute( on_transport_error: Callable[[Exception], dict[str, Any]] | None = None, business_impact: dict[str, Any] | None = None, action_digest: str | None = None, - # F03 (2026-08-22): accept `tools` kwarg from the - # ``@sensitive`` decorator (``_enforce_sensitive_tool``) - # so the bridge from decorators.py:735 stays - # source-pin-compatible with test_execute_tools_propagation.py - # while the runtime also reads ``get_call_tools()`` - # internally. The kwarg and the contextvar are merged - # below — kwarg wins when supplied, otherwise the - # contextvar flows through (which the F03 fix in - # decorators.py populates from ``fn.__name__`` before - # this method is called). + # ``tools`` is supplied by the ``@sensitive`` decorator and + # merged with the contextvar below — kwarg wins when supplied, + # otherwise the contextvar flows through (populated by + # decorators.py from ``fn.__name__`` before this method runs). tools: tuple[str, ...] | None = None, ) -> dict[str, Any]: """ @@ -2963,7 +2905,7 @@ def execute( mode: Execution mode ("auto", "inline", "strict") - "auto": auto-select based on tool risk on_transport_error: Optional callback for transport-error - handling (legacy); prefer the typed exception path. + handling; prefer the typed exception path. business_impact: Typed action payload (Money impact for now). When supplied, the backend uses it to evaluate rule predicates AND stamps the approval row's @@ -2993,9 +2935,9 @@ def execute( - "auto" (default): ALWAYS contacts the gateway. This is the cloud-only invariant — budget, rate-limit, and tool-block policies cannot be bypassed by - omitting ``mode``. Pre-v0.x SDKs silently switched - to "inline" for non-sensitive tools, which caused - DEF-TS12-01 (cycle 20260910T0515). + omitting ``mode``. Earlier SDKs silently switched + to "inline" for non-sensitive tools, which was + the source of the DEF-TS12-01 audit cycle. - "inline": explicit opt-out of /execute. Skips ALL enforcement (budget / rate / tool-block); returns a synthetic local allow. Use only when the caller @@ -3003,10 +2945,9 @@ def execute( gateway round-trip. Cannot be combined with sensitive tools — sensitive tools always go to /execute even when "inline" is requested. - - "strict": explicit gateway round-trip (same - wire behaviour as "auto" post-DEF-TS12-01, but - useful for audit clarity when the caller wants - the intent on the wire). + - "strict": explicit gateway round-trip (same wire + behaviour as "auto", but useful for audit clarity + when the caller wants the intent on the wire). Raises: NullRunBlockedException: If decision is "block" @@ -3025,25 +2966,16 @@ def execute( # cannot be silently bypassed because the SDK caller used the # default ``mode="auto"``. # - # Pre-fix (DEF-TS12-01, cycle 20260910T0515): ``mode="auto"`` - # with a non-sensitive tool resolved to ``mode="inline"`` which - # returned a synthetic local allow WITHOUT contacting the - # gateway. Every operator-configured budget, rate-limit, and - # tool-block policy was silently bypassed for non-sensitive - # tools. The dashboard showed policies in effect; the SDK - # ignored them. This is now fixed: ``mode="auto"`` → - # ``"strict"`` unconditionally. - # - # Explicit opt-out paths (preserved unchanged): + # Explicit opt-out paths: # 1. ``mode="inline"`` (explicit opt-in by the caller) — # returns the local allow WITHOUT contacting the gateway. # Documented as the only way to skip /execute. Use # sparingly: skips ALL enforcement, not just budget. # 2. ``mode="strict"`` (explicit opt-in by the caller) — # forces /execute round-trip regardless of tool - # sensitivity. Identical wire behaviour to ``"auto"`` - # post-fix, but useful when the caller wants the - # intent on the wire for audit clarity. + # sensitivity. Identical wire behaviour to ``"auto"``, + # but useful when the caller wants the intent on the + # wire for audit clarity. # # The two sensitivity checks below still gate the inline # fast-path — sensitive tools cannot be silently skipped @@ -3077,14 +3009,14 @@ def execute( # post-approval re-check so the backend can bind both requests # to the same logical action. # - # AUDIT P0-27 (2026-09-05): read from the same contextvar - # /check uses (`_operation_id_var`) instead of minting an - # independent UUID v4. The pre-fix `str(uuid.uuid4())` - # produced a different value than the one /check used, - # silently breaking the backend's operation_id-keyed binding. - # If /execute is the FIRST wire call (no prior /check in - # scope), mint here and stash in the contextvar; otherwise - # reuse whatever /check minted. + # Read from the same contextvar /check uses + # (``_operation_id_var``) instead of minting an independent + # UUID v4 — a fresh mint here would produce a different + # value than the one /check used, silently breaking the + # backend's operation_id-keyed binding. If /execute is the + # FIRST wire call (no prior /check in scope), mint here and + # stash in the contextvar; otherwise reuse whatever /check + # minted. from nullrun.context import ( get_operation_id as _get_op_id_for_execute, ) @@ -3115,32 +3047,21 @@ def execute( tools = _get_call_tools_for_execute() - # DEFS-SDKEXEC-GATE-FIRST (2026-09-08): hoist the execution_id - # mint to REUSE the server-minted id from a prior /gate call - # (set via `_capture_server_minted_execution_id` from the - # `reservation_id` field of the /gate response). - # - # Why this matters: backend `/api/v1/execute` (the wire - # contract enforced by `backend/src/proxy/http/gate/execute.rs` - # since DEF-SDKK-022-EXEC-BYPASS, 2026-09-04, RUN_ID=20260904T1500) - # runs an existence check on `execution:{id}` in Redis at - # `execute.rs:180-208` and returns 404 EXECUTION_NOT_FOUND when - # no prior /gate minted the binding. Pre-fix this method minted - # a fresh `uuid7_str()` here — the freshly-minted id was never - # registered by /gate, so /execute fail-CLOSED with 404 on - # EVERY call and the SDK translated the 404 into a synthetic - # block ("Gateway returned 404") in - # ``transport.py::execute`` (line ~1195). + # Reuse the server-minted execution_id captured from the prior + # /gate call (via ``_capture_server_minted_execution_id`` reading + # the ``reservation_id`` field of the /gate response). The + # backend's ``/api/v1/execute`` runs an existence check on + # ``execution:{id}`` and returns 404 EXECUTION_NOT_FOUND when + # no prior /gate minted the binding — a fresh-mint here would + # always trip that check and translate into a synthetic block. # - # Resolution: when a prior /gate captured a server-minted - # execution_id into ``_server_minted_execution_id_var``, reuse - # it. The decorator-driven ``@protect @sensitive`` path always - # runs ``check_workflow_budget()`` BEFORE ``runtime.execute()`` - # (decorators.py:538 vs :824), so the contextvar is populated - # in the common path. Direct callers of ``runtime.execute()`` - # without a prior /gate will fall through to the fresh-mint - # branch below — that's a wire-contract violation and the - # backend's 404 is the correct fail-CLOSED response. + # The decorator-driven ``@protect @sensitive`` path runs + # ``check_workflow_budget()`` BEFORE ``runtime.execute()``, so + # the contextvar is populated in the common path. Direct + # callers of ``runtime.execute()`` without a prior /gate fall + # through to the fresh-mint branch below — that is a + # wire-contract violation and the backend's 404 is the correct + # fail-CLOSED response. from nullrun.context import get_server_minted_execution_id prior_execution_id = get_server_minted_execution_id() @@ -3201,10 +3122,10 @@ def execute( approval_id = result.get("approval_id") or "" if not approval_id: metrics.inc_runtime("execute_blocked") - # 2026-09-08: typed wire-bug (NR-A004). The server - # returned require_approval without an approval_id — - # this is a wire-contract bug, NOT a transient failure. - # Cookbook code catches this and reports to NULLRUN + # Typed wire-bug (NR-A004): the server returned + # require_approval without an approval_id. This is a + # wire-contract bug, not a transient failure. Cookbook + # code catches the typed exception and reports to NULLRUN # support; do NOT retry. raise NullRunApprovalResponseMissingError( workflow_id=workflow_id or UNKNOWN_WORKFLOW_ID, @@ -3237,8 +3158,8 @@ def execute( outcome = str(approval_result.get("outcome") or "").lower() if outcome != "approved": metrics.inc_runtime("execute_blocked") - # 2026-09-08: dispatch typed approval exception by - # outcome so cookbook code can react per wire-code: + # Dispatch typed approval exception by outcome so + # cookbook code can react per wire-code: # denied → NR-A011 (NullRunApprovalDeniedError) # timeout → NR-A012 (NullRunApprovalExpiredError) # other → NR-X001 (NullRunBlockedException, generic) @@ -3279,9 +3200,9 @@ def execute( result = self._transport.execute(**execute_kwargs) if result.get("decision") == "require_approval": metrics.inc_runtime("execute_blocked") - # 2026-09-08: typed replay-rejection (NR-A015). The - # operator approved but the same approval_id was - # already consumed by a concurrent /execute (race). + # Typed replay-rejection (NR-A015): the operator + # approved but the same approval_id was already + # consumed by a concurrent /execute (race). # Cookbook pattern: do NOT retry the same approval_id; # treat as idempotency violation (likely a client # retry loop). @@ -3554,7 +3475,7 @@ def _enrich_event(self, event: dict[str, Any]) -> dict[str, Any]: if attempt_index > 0: # Only add if not default (first attempt) enriched["attempt_index"] = attempt_index - # 2026-07-04 (v0.12.0 wiring fix — ): + # Drop the stale capture. if "execution_id" not in enriched: import time as _time @@ -3584,7 +3505,7 @@ def _enrich_event(self, event: dict[str, Any]) -> dict[str, Any]: else: enriched["execution_id"] = smid - # 2026-07-04: propagate the in-scope + # Propagate the in-scope idempotency_key. if "idempotency_key" not in enriched: from nullrun.context import get_server_minted_idempotency_key @@ -3620,42 +3541,32 @@ def _enrich_event(self, event: dict[str, Any]) -> dict[str, Any]: return enriched def _route_track(self, wire_event: dict[str, Any]) -> None: - """Route a tracked event to v3 single-event /track or - legacy batch /track/batch. + """Route a tracked event to v3 single-event /track or the + batch /track/batch endpoint. Why this exists --------------- - Pre-0.12.0 wiring the SDK always called - ``self._transport.track(wire_event)`` which posts to the - legacy ``/api/v1/track/batch`` (the ``process_span_event`` - pipeline). That pipeline reads the org's lifetime - ``monthly_cost`` counter — drift with the dashboard's - period-bound ``bp:{ts}:cost_cents`` per G1 - and never exercises v3 ``consume_budget_v3`` so the - consume ≤ reserve + ε invariant is never validated. - - The fix: route events that have a paired ``/check`` - reservation (currently: ``llm_call``) to - ``track_single`` which posts to ``/api/v1/track``. The - backend's consume takes the server-minted execution_id - from the request, looks up - ``reservation:{execution_id}`` and runs the invariant. - Span events still ride /track/batch — they have no - reservation to release. + Events with a paired ``/check`` reservation (currently + ``llm_call``) route to ``track_single`` which posts to + ``/api/v1/track``. The backend's consume takes the + server-minted execution_id from the request, looks up + ``reservation:{execution_id}`` and validates the + consume ≤ reserve + ε invariant. Span events still ride + /track/batch — they have no reservation to release. Opt-out ------- ``NULLRUN_V3_TRACK_DISABLE=1`` forces every event - through the legacy batch path. Use it on backends that - haven't flipped ``NULLRUN_CONSUME_V3_ENABLED=1`` yet. + through the batch path. Use it on backends that haven't + flipped ``NULLRUN_CONSUME_V3_ENABLED=1`` yet. Failure mode ------------ ``track_single`` raises on 422 / 503 / 5xx (see ``nullrun.breaker.exceptions``). We catch and log at WARNING level; the event is dropped (NOT retried via - the batch path — that would risk double-billing - idempotency contract). + the batch path — that would risk double-billing the + idempotency contract). """ from nullrun.context import get_server_minted_execution_id @@ -3706,16 +3617,15 @@ def _route_track(self, wire_event: dict[str, Any]) -> None: metrics.inc_runtime("v3_track_single_ok") except Exception as exc: # noqa: BLE001 — transport-level metrics.inc_runtime("v3_track_single_failed") - # 2026-09-12 (DEF-CACHE-STALE-ALLOW-AFTER-OVERBUDGET): - # when the backend refuses the consume with HTTP 422 - # (CONSUME_OVERBUDGET) or HTTP 402 (REDIS_UNAVAILABLE - # on the consume path) the chain has hit its budget - # ceiling. Any cached "allow" for the same - # (workflow_id, chain_id) must NOT be served for the - # next 0–5 s — otherwise a chain firing faster than - # the cache TTL over-reserves against a budget the - # server has just rejected. Invalidate before logging - # so the order in logs matches the order in code. + # DEF-CACHE-STALE-ALLOW-AFTER-OVERBUDGET: when the backend + # refuses the consume with HTTP 422 (CONSUME_OVERBUDGET) + # or HTTP 402 (REDIS_UNAVAILABLE on the consume path) + # the chain has hit its budget ceiling. Any cached + # "allow" for the same (workflow_id, chain_id) must NOT + # be served for the next 0–5 s — otherwise a chain firing + # faster than the cache TTL over-reserves against a + # budget the server has just rejected. Invalidate before + # logging so the order in logs matches the order in code. # # chain_id lives in the contextvar (set by the # ``with chain(...)`` contextmanager or @@ -3773,10 +3683,10 @@ def track_llm( Track result dict from the runtime. Note: - `cost_cents` is no longer a parameter. The backend computes - it from `input_tokens` + `output_tokens` + the org's pricing - policy. Splitting prompt vs completion matters because most - models price them differently. + `cost_cents` is computed by the backend from + `input_tokens` + `output_tokens` + the org's pricing + policy. Splitting prompt vs completion matters because + most models price them differently. """ # Lazy import to keep the runtime import graph acyclic -- # `nullrun.tracing` deliberately has no SDK-side dependencies. @@ -3839,11 +3749,11 @@ def track_tool( Track result dict from the runtime. Note: - `cost_cents` is no longer a parameter. Tool cost is derived - from `duration_ms` + the org's policy (or left at 0 if the - org doesn't bill tools). `duration_ms` is the public field - name; the wire field is `latency_ms` for backward compat - with backend consumers. + Tool cost is derived from `duration_ms` + the org's + policy (or left at 0 if the org doesn't bill tools). + `duration_ms` is the public field name; the wire field + is `latency_ms` for backward compat with backend + consumers. """ from nullrun.tracing import get_current_span @@ -4156,7 +4066,7 @@ def _safe_str(key: str) -> str | None: return action_digest, policy_hash -# 2026-07-04 (v0.12.0 wiring fix — ): build the +# Build the def _build_v3_track_payload( wire_event: dict[str, Any], reservation_id: str, @@ -4206,7 +4116,7 @@ def _build_v3_track_payload( payload["trace_id"] = wire_event["trace_id"] if "span_id" in wire_event and wire_event["span_id"]: payload["span_id"] = wire_event["span_id"] - # 2026-07-12 (multi-agent span attachment): the orchestration + # Multi-agent span attachment: the orchestration if "parent_trace_id" in wire_event and wire_event["parent_trace_id"]: payload["parent_trace_id"] = wire_event["parent_trace_id"] @@ -4224,7 +4134,7 @@ def _build_v3_track_payload( if k in wire_event and wire_event[k] is not None: payload[k] = wire_event[k] - # 2026-07-13 (vendor-extractor edge cases, SDK counterpart at + # Vendor-extractor edge cases (SDK counterpart of for k in ( "cache_read_tokens", "cache_write_tokens", @@ -4269,11 +4179,11 @@ def _build_v3_track_payload( def get_runtime() -> NullRunRuntime: """Get or create the global runtime instance. - Prefers the registry. The legacy global _runtime slot is - kept as a backwards-compat cache so external code that - imports nullrun.runtime._runtime still works, but the + Prefers the registry. The module-level ``_runtime`` slot + remains as a compatibility cache so external code that + imports ``nullrun.runtime._runtime`` keeps working; the canonical source of truth is the registry (see - nullrun._registry.RuntimeRegistry). + ``nullrun._registry.RuntimeRegistry``). """ cached = get_active_runtime() if cached is not None: diff --git a/src/nullrun/transport.py b/src/nullrun/transport.py index 74bb69e..401b031 100644 --- a/src/nullrun/transport.py +++ b/src/nullrun/transport.py @@ -221,16 +221,16 @@ def _retry_with_backoff( delay += random.uniform(-jitter * delay, jitter * delay) Formula (with Retry-After): actual_delay = min(last_retry_after_seconds, max_delay) - NR-006 (audit 2026-08-24): when ``retry_on_5xx=True`` a 5xx - response is treated as transient infrastructure failure and - retried via the same backoff path as network errors. After the - retry budget is exhausted the LAST 5xx response is returned - (not raised) so the caller can produce a deterministic - fail-CLOSED fallback — the audit's "fail-NO-CHECK" violation - happens when a 5xx short-circuits to a synthetic block without - any retry. Default ``retry_on_5xx=False`` preserves the - pre-existing /track and /execute semantics where 5xx is a - classified GATEWAY_ERROR that raises immediately. + NR-006: when ``retry_on_5xx=True`` a 5xx response is treated as + transient infrastructure failure and retried via the same + backoff path as network errors. After the retry budget is + exhausted the LAST 5xx response is returned (not raised) so + the caller can produce a deterministic fail-CLOSED fallback — + the audit's "fail-NO-CHECK" violation happens when a 5xx + short-circuits to a synthetic block without any retry. Default + ``retry_on_5xx=False`` preserves the /track and /execute + semantics where 5xx is a classified GATEWAY_ERROR that raises + immediately. """ # Eager imports for the exception classes that the ``except`` # branch below references. Lazy imports inside the ``try`` body @@ -1852,13 +1852,11 @@ def track_single( ) -> dict[str, Any]: """POST /api/v1/track — wire-protocol v3 single-event consume. - . The single-event path is the v3 - replacement for the legacy `/api/v1/track/batch` POST body. - It runs the CONSUME_SCRIPT invariant - ``actual_cost <= reserved_cents + epsilon_cents`` (§25 - ADR-005) and rejects with 422 CONSUME_OVERBUDGET on - violation. The reserved binding is the one created by the - matching ``/check`` call (same ``reservation_id``). + Runs the CONSUME_SCRIPT invariant + ``actual_cost <= reserved_cents + epsilon_cents`` (ADR-005) + and rejects with 422 CONSUME_OVERBUDGET on violation. The + reserved binding is the one created by the matching + ``/check`` call (same ``reservation_id``). The wire shape is built by ``runtime._build_v3_track_payload`` (see ``runtime.py:2679-2776``); this method just forwards @@ -2259,10 +2257,9 @@ def approximate_budget( # ADR-009 P1 — Audit log governance surface (v0.15.0) # ==================================================================== # Five methods exposing the /api/v1/orgs/:org_id/audit-log/* family - # of endpoints to SDK consumers. Pre-v0.15.0 SDKs had no audit - # client — operators had to curl the wire directly. Now they can - # call ``runtime.audit.list(...)`` etc. and get typed dataclasses - # back without writing JSON parsing glue. + # of endpoints to SDK consumers. Callers invoke + # ``runtime.audit.list(...)`` etc. and get typed dataclasses back + # without writing JSON parsing glue. # # All five methods route through the same auth + protocol + # trace-context machinery as the other Transport methods — see @@ -2943,9 +2940,8 @@ def _parse_v3_error_envelope( # Lazy import to avoid a hard dependency at module import time. # `_parse_v3_error_envelope` is a module-level helper; the exception # classes live in `nullrun.breaker.exceptions`. Importing here -# (rather than at the top of transport.py) keeps the legacy import -# graph identical and avoids breaking the frozen -# ``_parse_error_envelope`` test contract. +# (rather than at the top of transport.py) avoids breaking the +# frozen ``_parse_error_envelope`` test contract. def _build_v3_error_code_map() -> dict[str, type[Exception]]: """Construct the v3 error_code → exception class mapping. @@ -3020,28 +3016,23 @@ def _build_v3_error_code_map() -> dict[str, type[Exception]]: "RATE_LIMIT_REDIS_UNAVAILABLE": NullRunRateLimitRedisError, "BUDGET_DATA_UNAVAILABLE": NullRunBackendError, # 402 — approval-create failure family (DEF-ARFLOW-TOOLNAME-01, - # the typed ``NullRunApprovalDbUnavailableError`` (NR-A016) so + # typed ``NullRunApprovalDbUnavailableError`` (NR-A016) so # cookbook code can branch on the typed class instead of - # falling through to the base NullRunBlockedException. Pre-B.1 - # all six collapsed to the base class — operators couldn't tell - # apart a transient DB outage from a validation failure. + # falling through to the base NullRunBlockedException). "APPROVAL_DB_UNAVAILABLE": NullRunApprovalDbUnavailableError, "APPROVAL_PERSISTENCE_FAILED": NullRunApprovalDbUnavailableError, "APPROVAL_VALIDATION_FAILED": NullRunApprovalDbUnavailableError, "APPROVAL_CONFLICT": NullRunApprovalDbUnavailableError, "APPROVAL_NOT_FOUND": NullRunApprovalDbUnavailableError, "APPROVAL_CREATE_FAILED": NullRunApprovalDbUnavailableError, - # audit, A-1+A-2 bundle). Distinct from the /gate - # create-failure family above: these are the seven - # distinct outcomes that the backend's - # `gate_internal()` returns on /execute post-approval - # grant-consume (see - # `backend/src/proxy/http/gate/internal.rs:3059-3108, - # 3115-3138`). Pre-v3.53 the SDK collapsed all six - # into NullRunBlockedException — bilateral wire gap. - # Post-v3.53 each maps to a typed exception - # (NR-A010..NR-A015) so cookbook recipes can branch - # on the precise outcome (e.g. ``except + # A-1+A-2 bundle. Distinct from the /gate create-failure + # family above: these are the seven distinct outcomes that + # the backend's ``gate_internal()`` returns on /execute + # post-approval grant-consume (see + # ``backend/src/proxy/http/gate/internal.rs:3059-3108, + # 3115-3138``). Each maps to a typed exception + # (NR-A010..NR-A015) so cookbook recipes can branch on the + # precise outcome (e.g. ``except # NullRunApprovalNotYetApprovedError:`` for wait/poll, # ``except NullRunApprovalDeniedError:`` for terminal # surface-to-user, ``except diff --git a/src/nullrun/transport_websocket.py b/src/nullrun/transport_websocket.py index a2783e3..6f2173c 100644 --- a/src/nullrun/transport_websocket.py +++ b/src/nullrun/transport_websocket.py @@ -340,20 +340,16 @@ async def _handle_message(self, message: str) -> None: # value under the ``api_key`` field — we MUST read it # back from there and use it as the HMAC identifier. # - # Pre-FIX-F4 this branch read ``data["api_key_id"]`` - # which used to be the wire field name on the server - # side. That field now carries the same user-facing - # value (no longer the internal UUID key_id), so for - # backwards compat we accept either field name — - # pre-FIX-F4 envelopes may still arrive with - # ``api_key_id`` carrying the user-facing string - # because the server's only consumers were pre-FIX-F4 - # SDKs. + # The ``data["api_key_id"]`` field carries the + # user-facing API key value (not an internal UUID). + # Accept either field name for backwards compat — + # older envelopes may still arrive with + # ``api_key_id`` carrying the user-facing string. # - # Fall back to ``self.api_key`` only when the envelope - # has neither field (a pre-FIX-D server without - # signed_payload), which is a degraded path that - # already 403'd in real life per the FIX-C comments. + # Fall back to ``self.api_key`` only when the + # envelope has neither field (a server without + # ``signed_payload``), which is a degraded path that + # already 403's in practice. envelope_api_key = ( data.get(WS_HMAC_IDENTITY_FIELD) if isinstance(data.get(WS_HMAC_IDENTITY_FIELD), str) @@ -648,8 +644,8 @@ async def _handle_state_change_with_ack( # Check if this state requires acknowledgment # # (`runtime.py`) lowercases before comparing so it survives a - # server regression to lowercase states. The WS path used to - # exact-match only. Without this fallback, a server regression + # server regression to lowercase states. The WS path matches + # case-insensitively too — without this, a server regression # would silently drop the ACK (the existing test pins # PascalCase as the happy path, but does not pin what happens # if the server emits ``"killed"``). From 57225e91cb8ffefc57617c0f30bdf508988ff8a8 Mon Sep 17 00:00:00 2001 From: Anatoly Maltsev Date: Fri, 25 Sep 2026 13:38:28 +0400 Subject: [PATCH 04/10] refactor(sdk): drop framework extras + auto_instrument public surface MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit NullRun is a runtime decision layer, not a framework integration library. The framework auto-instrumentation patches (langchain, crewai, autogen, langgraph, llama-index, openai-agents) remain in code as silent auto-detect hooks, but are no longer: - advertised in pyproject.toml as installable extras - documented in README as a feature - exported from the top-level `nullrun` namespace User-facing install is exactly one line: pip install nullrun Public surface cleanup: - removed `auto_instrument` / `is_auto_instrumented` from _LAZY_EXPORTS (was internal trigger, not user API) - removed module-level `track_event` alias (duplicate of `track`; runtime.track_event method still used internally) - removed NullRunCallback from _LAZY_EXPORTS (advanced/manual path; reachable via `nullrun.toolbox.langgraph` if ever needed) README changes: - dropped 'Framework adapters — auto-detected' section - dropped framework-specific Examples bullets (langgraph, crewai, autogen, llama-index) - rewrote headline to 'works with any LLM SDK that uses httpx' pyproject.toml: removed [agents], [langchain], [langgraph], [llama-index], [crewai], [autogen] optional-dependency groups. Kept [opentelemetry] (industry standard, not a framework) and [dev]. Co-Authored-By: Claude Code --- README.md | 39 ++++++-------------------------------- pyproject.toml | 42 ++--------------------------------------- src/nullrun/__init__.py | 2 -- src/nullrun/runtime.py | 7 ------- 4 files changed, 8 insertions(+), 82 deletions(-) diff --git a/README.md b/README.md index 6d57721..cfa67c7 100644 --- a/README.md +++ b/README.md @@ -5,8 +5,7 @@ **Ship AI agents with real-time budget, policy, and human-approval gates.** Zero-refactor cost control, tool policy enforcement, and audit trail for any -LLM-powered agent - works with OpenAI, Anthropic, LangGraph, CrewAI, AutoGen, -LlamaIndex, and your own stack. +LLM-powered agent — works with any LLM SDK that uses `httpx`, plus your own stack. [Quickstart](https://docs.nullrun.io/getting-started/onboarding/) · [Docs](https://docs.nullrun.io) · [Examples](https://github.com/nullrunio/nullrun-examples) @@ -63,7 +62,7 @@ Existing observability tools tell you **after** the fact. NullRun enforces **bef |---|---| | **Hard & soft budget gates** — atomic Redis-enforced | **Tool policy enforcement** — block dangerous tools before execution | | **Human-in-the-loop approvals** — pause agent and await `approval_resolved` via WS push | **Immutable audit trail** — every decision, every tool call, every cent | -| **Zero-code instrumentation** — `nullrun.init()` patches `httpx` once for any vendor | **LangGraph, CrewAI, AutoGen, LlamaIndex** — first-class integrations | +| **Zero-code instrumentation** — `nullrun.init()` patches `httpx` once for any vendor | **No vendor lock-in** — works with any LLM SDK that uses httpx | | **Memory-safe streaming** — 16 MiB response body; full body for usage extraction | **Lightweight** — no LLM-key storage, no proxy required | | **Server-authoritative cost** — server-minted execution IDs | **MCP support** — expose tools to agents via Model Context Protocol | @@ -211,33 +210,11 @@ def my_agent(prompt: str) -> str: ``` -### Framework adapters — auto-detected +If you call `@protect` *before* `init_or_die()`, the SDK lazy-initializes +the runtime from `NULLRUN_API_KEY` on the first decorated call. You can +write your agent code with the decorator first and the init second — or +skip `init` entirely if your environment is already configured. -NullRun auto-detects installed frameworks and instruments them automatically -when `init_or_die()` runs (or when `@protect` first fires). You don't need -to choose an extra; if a framework is already in your environment, it gets -patched in place. - -| Framework | What gets patched | Trigger | -|---|---|---| -| **LangGraph** (`Pregel.invoke` / `stream` / `ainvoke` / `astream`) | `NullRunCallback` injected per call | auto on `init_or_die()` | -| **LangChain** (`BaseCallbackManager`) | `NullRunCallback` registered | auto on `init_or_die()` | -| **OpenAI Agents** (`Runner.run` / `run_streamed`) | `RunHooks` / `RunStreamedHooks` instrumented | auto on `init_or_die()` | -| **LlamaIndex** (`get_dispatcher`) | `LLMChatEndEvent` / `FunctionCallEvent` handlers | auto on `init_or_die()` | -| **CrewAI** (event bus + `usage_metrics`) | `Agent` / `Task` / `Crew` lifecycle | auto on `init_or_die()` | -| **AutoGen** (`Agent.run` / `a_run`) | message-streaming hooks (HTTP path is httpx-based) | auto on `init_or_die()` | - -**HTTP-level coverage is the foundation** — `httpx` (and `requests`) are -patched once by `init_or_die()` regardless of vendor. Token counts and -model info are extracted from response bodies for OpenAI, Azure, Anthropic, -Mistral, Gemini, Cohere, and Bedrock without those vendor SDKs needing to -be installed. If you use the raw `httpx.Client` API directly, you get -cost tracking out of the box. - -If you call `@protect` *before* `init_or_die()`, the SDK auto-triggers -instrumentation lazily on the first decorated call. You can write your -agent code with the decorator first and the init second — or skip `init` -entirely if your environment is already configured via `NULLRUN_API_KEY`. --- ## How NullRun compares @@ -336,10 +313,6 @@ and `src/nullrun/transport.py::consume_approval`. Runnable, copy-pastable examples live in a separate repo so you can adapt without cloning the SDK source: -- **[LangGraph](https://docs.nullrun.io/how-to/langgraph/)** — multi-node agent with budget + approval -- **[CrewAI](https://docs.nullrun.io/how-to/crewai/)** — multi-agent crew with shared budget -- **[AutoGen](https://docs.nullrun.io/how-to/autogen/)** — group-chat agent with policy gating -- **[LlamaIndex](https://docs.nullrun.io/how-to/llama-index/)** — RAG pipeline with cost-per-query enforcement - **[Custom tools](https://docs.nullrun.io/how-to/fastapi/)** — register your own tools for policy - **[Multi-agent](https://docs.nullrun.io/how-to/multi-agent/)** — shared budget across sub-agents diff --git a/pyproject.toml b/pyproject.toml index 96894bc..a4cd2f6 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -71,50 +71,12 @@ dependencies = [ ] [project.optional-dependencies] +# OpenTelemetry is an observability standard (not a vendor framework); +# auto-detected at runtime when installed. opentelemetry = [ "opentelemetry-api>=1.26.0,<2.0", "opentelemetry-sdk>=1.26.0,<2.0", ] -langgraph = [ - "langgraph>=0.2.0,<1.0", -] -# Framework auto-instrumentation dependencies. -# -# These extras install framework SDKs whose event systems NullRun -# subscribes to. NullRun's HTTP-level instrumentation -# (``patch_httpx`` + ``patch_requests`` + 5 URL-keyed extractors) -# covers OpenAI / Anthropic / Mistral / Gemini / Cohere / Bedrock -# WITHOUT requiring their vendor SDKs — all of those vendors route -# through httpx, and NullRun parses the response body by URL host. -# The vendor SDK packages are NOT imported anywhere in -# ``src/nullrun/``, so extras like ``[openai]`` / ``[anthropic]`` / -# ``[mistral]`` / ``[gemini]`` / ``[cohere]`` / ``[bedrock]`` would -# be dead weight for the SDK. -# -# What NullRun DOES import from these framework SDKs: -# - ``agents`` → ``agents.Runner`` (openai-agents tracing model) -# - ``langchain`` → ``langchain_core.callbacks.BaseCallbackManager`` -# + ``langchain_core.language_models.BaseChatModel`` -# - ``langgraph`` → ``langgraph.pregel.Pregel`` -# - ``llama-index`` → ``llama_index.core.instrumentation`` -# - ``crewai`` → ``crewai.Crew`` + ``crewai.events`` -# - ``autogen`` → ``autogen_agentchat.agents.BaseChatAgent`` + -# ``autogen_ext.models.openai.OpenAIChatCompletionClient`` -# -# Each ``patch_*`` wraps its framework import in -# ``try/except ImportError`` so ``nullrun.init()`` never crashes when -# the optional package is missing. Auto-detection: NullRun activates -# an adapter when the package is installed — the user does NOT need -# to choose which framework extra to install; installing any one of -# them auto-enables its adapter. -agents = ["openai-agents>=0.1,<1.0"] -langchain = ["langchain-core>=0.3,<1.0"] -llama-index = ["llama-index-core>=0.10.20,<1.0"] -crewai = ["crewai>=0.80,<2.0"] -autogen = [ - "autogen-agentchat>=0.4,<1.0", - "autogen-ext[openai]>=0.4,<1.0", -] dev = [ "pytest>=8.0", "pytest-asyncio>=0.23", diff --git a/src/nullrun/__init__.py b/src/nullrun/__init__.py index 565d5be..7f807fe 100644 --- a/src/nullrun/__init__.py +++ b/src/nullrun/__init__.py @@ -438,8 +438,6 @@ def my_agent: "set_chain_id": ("nullrun.context", "set_chain_id"), "get_chain_op": ("nullrun.context", "get_chain_op"), "set_chain_op": ("nullrun.context", "set_chain_op"), - # Instrumentation - "NullRunCallback": ("nullrun.instrumentation", "NullRunCallback"), # Toolbox — framework-specific wrappers. The previous `instrument ` # helper lived at `nullrun.instrumentation.langgraph.instrument`; # it is now `nullrun.toolbox.langgraph.wrapper`. Reachable as diff --git a/src/nullrun/runtime.py b/src/nullrun/runtime.py index 30266fa..773ffa0 100644 --- a/src/nullrun/runtime.py +++ b/src/nullrun/runtime.py @@ -4206,13 +4206,6 @@ def track(event: dict[str, Any]) -> dict[str, Any]: return get_runtime().track(event) -# Explicit alias for `track` -- same call signature, friendlier -# name for users who reach for `track_event` first. Both names -# share the same callable object, so `nullrun.track is -# nullrun.track_event` is True. -track_event = track - - def track_llm( input_tokens: int, output_tokens: int = 0, From f1721f299b1592a741cacb5248224526be69fe10 Mon Sep 17 00:00:00 2001 From: Anatoly Maltsev Date: Fri, 25 Sep 2026 13:44:47 +0400 Subject: [PATCH 05/10] refactor(sdk): drop redundant @guarded decorator MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The @nullrun.guarded decorator was a 3-line syntactic shortcut for `with nullrun.handle()`. With handle() the canonical user-facing error-translation path and zero production examples using the decorator form, @guarded was pure duplication. Removes: - def guarded() from src/nullrun/_handle.py - "guarded" from _LAZY_EXPORTS and __all__ in __init__.py - 4 test_guarded_* tests from test_handle.py + test_dev_error_report.py - dead import in examples/tools/synthetic_sdk_load.py - examples/CONTRIBUTING.md doc reference Adds: - tests/test_init_or_die.py — splits init_or_die tests out of test_handle.py so the two surfaces no longer share a file Net: -520 lines, one way to translate NullRunError instead of two. Co-Authored-By: Claude Code --- src/nullrun/__init__.py | 9 +- src/nullrun/_handle.py | 84 +++++------------ tests/test_dev_error_report.py | 51 ++--------- tests/test_handle.py | 159 +++------------------------------ tests/test_init_or_die.py | 96 ++++++++++++++++++++ 5 files changed, 139 insertions(+), 260 deletions(-) create mode 100644 tests/test_init_or_die.py diff --git a/src/nullrun/__init__.py b/src/nullrun/__init__.py index 7f807fe..67eb2ef 100644 --- a/src/nullrun/__init__.py +++ b/src/nullrun/__init__.py @@ -11,7 +11,7 @@ Everything else exposed by ``nullrun`` is either runtime lifecycle (``init``, ``shutdown``, ``on_error``, ``status``), the structured exception hierarchy, or message/error-handling helpers -(``format_user_message``, ``handle``, ``guarded``, ``init_or_die``). +(``format_user_message``, ``handle``, ``init_or_die``). None of those are alternatives to ``@protect`` — they're setup and cleanup. @@ -518,7 +518,6 @@ def my_agent: # which shadows the lazy export and breaks ``from nullrun import # handle``. "handle": ("nullrun._handle", "handle"), - "guarded": ("nullrun._handle", "guarded"), "init_or_die": ("nullrun._handle", "init_or_die"), # ADR-009 P1 — governance audit surface (typed wire classes). # Users reach these as `from nullrun import AuditQuery` / @@ -613,16 +612,14 @@ def __dir__() -> list[str]: "format_user_message", "set_user_message", # Minimal-boilerplate error handling for scripts. ``handle`` is - # the context manager (``with nullrun.handle: ``), ``guarded`` - # is the decorator (``@nullrun.guarded``). Both translate any - # ``NullRunError`` into ``print(format_user_message(exc))`` + + # the context manager (``with nullrun.handle: ``). It translates + # any ``NullRunError`` into ``print(format_user_message(exc))`` + # ``sys.exit(1)``; ``WorkflowKilledInterrupt`` propagates. # ``init_or_die`` is the convenience wrapper around ``init`` # that catches NR-C001 "no api_key" at startup and exits # cleanly — without it the user sees a raw traceback before # any ``with handle: `` block is in scope. "handle", - "guarded", "init_or_die", ] diff --git a/src/nullrun/_handle.py b/src/nullrun/_handle.py index 4070bcc..58d1bc4 100644 --- a/src/nullrun/_handle.py +++ b/src/nullrun/_handle.py @@ -8,15 +8,14 @@ branch on a specific ``error_code`` — but it is **not** the default. For the common "I just want to run my agent and print a friendly -message on failure" case, this module provides three one-liners: +message on failure" case, this module provides two one-liners: *:func:`nullrun.handle` — context manager. -*:func:`nullrun.guarded` — decorator. *:func:`nullrun.init_or_die` — convenience wrapper around :func:`nullrun.init` that catches the ``NR-C001`` "no api_key" failure at startup and exits cleanly. -All three translate any:class:`nullrun.NullRunError` into a structured +Both translate any:class:`nullrun.NullRunError` into a structured developer-facing report (error code + what was attempted + where it came from + the underlying reason + how to fix it) and then exit ``1``. The end-user-friendly wording from @@ -26,15 +25,15 @@ :class:`nullrun.WorkflowKilledInterrupt` inherits from :class:`nullrun.NullRunError` (see the class docstring), so a bare ``except NullRunError`` would otherwise swallow the kill signal. -``handle``/``guarded`` explicitly re-raise it — the kill is a +``handle`` explicitly re-raises it — the kill is a control-plane action, not an SDK failure, and must reach the top of the agent loop. Non-NullRun exceptions also propagate unchanged. ``init_or_die`` exists because:func:`nullrun.init` is typically -called at module top-level — before any ``with handle: `` block or -``@guarded`` decorator is in scope. Without it, a missing -``NULLRUN_API_KEY`` env var produces a raw traceback. +called at module top-level — before any ``with handle: `` block is +in scope. Without it, a missing ``NULLRUN_API_KEY`` env var produces +a raw traceback. Why a separate module --------------------- @@ -60,22 +59,17 @@ from __future__ import annotations import sys -from collections.abc import Callable from contextlib import contextmanager -from typing import TypeVar from nullrun.breaker.exceptions import NullRunError, WorkflowKilledInterrupt from nullrun.messages import format_user_message -T = TypeVar("T") - def _render_dev_error_report( exc: NullRunError, user_message: str, ) -> str: - """Render a four-line developer-facing report for ``handle`` / - ``guarded``. + """Render a four-line developer-facing report for ``handle``. The previous behaviour (print only ``format_user_message(exc)``) leaked zero information when a developer hit a config failure at @@ -189,7 +183,7 @@ def handle(*, exit_code: int = 1): Re-raised explicitly inside the ``except NullRunError`` branch because ``WorkflowKilledInterrupt`` sits on the ``NullRunError`` MRO (Sentry/OTel ``except Exception`` handlers should record kill - events; this ``handle``/``guarded`` wrapper opts OUT of that + events; this ``handle`` wrapper opts OUT of that recording on purpose). *:class:`KeyboardInterrupt` /:class:`SystemExit` (``BaseException``) -- same reason as the kill signal -- never reach the @@ -216,7 +210,7 @@ def handle(*, exit_code: int = 1): yield except NullRunError as exc: # the NullRunError MRO so Sentry/OTel `except Exception` - # handlers record kill events. ``handle``/``guarded`` are the + # handlers record kill events. ``handle`` is the # friendly-exit pattern, NOT the user-callback pattern -- kill # is a control-plane action and must propagate so the agent # loop / dashboard resume path can see it. Re-raise explicitly @@ -236,71 +230,35 @@ def handle(*, exit_code: int = 1): sys.exit(exit_code) -def guarded(fn: Callable[..., T]) -> Callable[..., T]: - """Decorator equivalent of ``with nullrun.handle: ``. - - Wrap a function so any:class:`nullrun.NullRunError` raised inside - it is caught, rendered as a user-facing message, and the process - exits with code ``1``. ``WorkflowKilledInterrupt`` and other - ``BaseException`` subclasses propagate (``handle`` re-raises kill - explicitly, see the kill-signal note in the module docstring). - - Pair with:func:`nullrun.protect` for the standard agent loop:: - - @nullrun.guarded - @nullrun.protect - def my_agent(prompt): - return call_llm(prompt) - - if __name__ == "__main__": - try: - print(my_agent("hello")) - finally: - nullrun.shutdown - - Args: - fn: The function to wrap. - - Returns: - A wrapper with the same signature that exits the process on - ``NullRunError`` and otherwise returns ``fn``'s value. - """ - def wrapper(*args, **kwargs): - with handle(): - return fn(*args, **kwargs) - - return wrapper - - def init_or_die(*, api_key: str | None = None, api_url: str | None = None, debug: bool = False, exit_code: int = 1): """Call:func:`nullrun.init` and exit cleanly on configuration failure. :func:`nullrun.init` is typically the first thing a script does - before any ``with nullrun.handle: `` block or ``@nullrun.guarded`` - decorator is in scope. A missing ``api_key`` therefore produces a - raw traceback — not a friendly exit. ``init_or_die`` closes that - gap by catching the startup:class:`nullrun.NullRunError` (NR-C001 - "no api_key"), printing the catalog user-message, and exiting. + before any ``with nullrun.handle: `` block is in scope. A missing + ``api_key`` therefore produces a raw traceback — not a friendly + exit. ``init_or_die`` closes that gap by catching the startup + :class:`nullrun.NullRunError` (NR-C001 "no api_key"), printing + the catalog user-message, and exiting. - On success returns the:class:`nullrun.NullRunRuntime` singleton + On success returns the :class:`nullrun.NullRunRuntime` singleton that ``init `` returns — assign it if you need it, ignore it otherwise:: - from nullrun import init_or_die, guarded, protect, shutdown + from nullrun import init_or_die, protect, shutdown init_or_die(api_key=os.environ["NULLRUN_API_KEY"]) - @guarded @protect def my_agent(prompt): return call_llm(prompt) if __name__ == "__main__": try: - print(my_agent("hello")) + with nullrun.handle: + print(my_agent("hello")) finally: - shutdown + shutdown Args: api_key: NullRun API key (or NULLRUN_API_KEY env var). @@ -318,7 +276,7 @@ def my_agent(prompt): try: return init(api_key=api_key, api_url=api_url, debug=debug) except NullRunError as exc: - # Same structured report as ``handle()`` / ``guarded`` -- a + # Same structured report as ``handle()`` -- a # "There's a configuration issue. Please contact support." # which gave the developer zero actionable detail. The # four-line report here names the missing env var, the URL @@ -334,4 +292,4 @@ def my_agent(prompt): sys.exit(exit_code) -__all__ = ["handle", "guarded", "init_or_die"] \ No newline at end of file +__all__ = ["handle", "init_or_die"] \ No newline at end of file diff --git a/tests/test_dev_error_report.py b/tests/test_dev_error_report.py index ead0312..69340e8 100644 --- a/tests/test_dev_error_report.py +++ b/tests/test_dev_error_report.py @@ -1,6 +1,6 @@ """ -Tests for the developer-facing error report rendered by ``handle``, -``guarded``, and ``init_or_die``. +Tests for the developer-facing error report rendered by ``handle`` and +``init_or_die``. Pre-fix (2026-09-22), the catch-all exit path printed only the catalog user-message ("There's a configuration issue. Please contact support.") @@ -23,11 +23,10 @@ import pytest import nullrun -from nullrun import guarded, handle +from nullrun import handle from nullrun._handle import _render_dev_error_report from nullrun.breaker.exceptions import ( NullRunAuthenticationError, - NullRunBudgetError, NullRunError, NullRunTransportError, ) @@ -124,7 +123,7 @@ def test_report_includes_docs_url(): assert "https://docs.nullrun.io" in report -# --- handle() / guarded() integration tests --------------------------------- +# --- handle() integration tests --------------------------------------------- def test_handle_prints_full_dev_report(monkeypatch, capsys): @@ -157,39 +156,6 @@ def fake_exit(code): assert "Rotate the API key." in err -def test_guarded_prints_full_dev_report(monkeypatch, capsys): - """``@guarded`` wraps ``handle()`` -- the dev report must surface - through the decorator path too, not just the context manager.""" - exits = [] - - def fake_exit(code): - exits.append(code) - raise SystemExit(code) - - monkeypatch.setattr("sys.exit", fake_exit) - - @guarded - def boom(): - raise NullRunBudgetError( - "wf-1", - "workflow budget exhausted", - error_code="NR-B004", - user_action="Wait for the next billing period.", - ) - - with pytest.raises(SystemExit): - boom() - - assert exits == [1] - err = capsys.readouterr().err - assert "[NR-B004]" in err - assert "what:" in err - assert "where:" in err - assert "why:" in err - assert "how to fix:" in err - assert "Wait for the next billing period." in err - - def test_handle_falls_back_to_legacy_on_helper_bug(monkeypatch, capsys): """Defensive: if the report builder itself raises (a future bug), ``handle()`` must still exit cleanly with the catalog headline. @@ -250,11 +216,8 @@ def fake_exit(code): # --- sanity: still callable without runtime -------------------------------- -def test_handle_and_guarded_do_not_require_runtime(): - """``handle`` / ``guarded`` must work without ``nullrun.init()``. - Sanity check that the helper module is importable on its own -- - this is the same invariant the existing ``test_handle.py`` pins.""" +def test_handle_does_not_require_runtime(): + """``handle`` must work without ``nullrun.init()``. Sanity check + that the helper module is importable on its own.""" assert callable(handle) - assert callable(guarded) assert callable(nullrun.handle) - assert callable(nullrun.guarded) diff --git a/tests/test_handle.py b/tests/test_handle.py index e8476b3..efe41ee 100644 --- a/tests/test_handle.py +++ b/tests/test_handle.py @@ -1,27 +1,26 @@ -"""Tests for the minimal-boilerplate error helpers (``nullrun.handle`` -``nullrun.guarded``). +"""Tests for the ``nullrun.handle`` context manager. Contract: -* Both translate any:class:`nullrun.NullRunError` into a single - ``print(format_user_message(exc), file=sys.stderr)`` and then +* ``with handle():`` translates any :class:`nullrun.NullRunError` + into ``print(format_user_message(exc), file=sys.stderr)`` and then ``sys.exit(1)``. -*:class:`nullrun.WorkflowKilledInterrupt` propagates unchanged — kill - must not be swallowed into a graceful exit. (2026-09-08 migration: - ``WorkflowKilledInterrupt`` is now an ``Exception`` subclass via - ``NullRunError``, but ``handle``/``guarded`` explicitly re-raise it - so the kill signal still reaches the top of the agent loop.) +* :class:`nullrun.WorkflowKilledInterrupt` propagates unchanged — + kill must not be swallowed into a graceful exit. + (``WorkflowKilledInterrupt`` is now an ``Exception`` subclass via + ``NullRunError``, but ``handle`` explicitly re-raises it so the + kill signal still reaches the top of the agent loop.) * Non-NullRun exceptions also propagate unchanged so the user's own bugs surface as honest tracebacks. -* No runtime is required — these helpers work without - ``nullrun.init ``. +* No runtime is required — ``handle`` works without + ``nullrun.init()``. """ from __future__ import annotations import pytest import nullrun -from nullrun import guarded, handle +from nullrun import handle from nullrun.breaker.exceptions import ( NullRunBudgetError, NullRunError, @@ -78,52 +77,6 @@ def test_handle_returns_on_success(monkeypatch): assert result == 3 -def test_guarded_decorator_catches_and_exits(monkeypatch, capsys): - """``@guarded`` translates NullRunError into sys.exit(1).""" - exits = [] - - def fake_exit(code): - exits.append(code) - raise SystemExit(code) - - monkeypatch.setattr("sys.exit", fake_exit) - - @guarded - def boom(): - raise NullRunError("something broke", error_code="NR-B002") - - with pytest.raises(SystemExit): - boom() - - assert exits == [1] - captured = capsys.readouterr() - # NR-B002 maps to the "service is temporarily unavailable" wording. - assert "temporarily unavailable" in captured.err.lower() - - -def test_guarded_returns_value_on_success(monkeypatch): - """``@guarded`` returns the wrapped function's value when nothing fails.""" - monkeypatch.setattr("sys.exit", lambda c: pytest.fail("sys.exit was called")) - - @guarded - def add(a, b): - return a + b - - assert add(2, 3) == 5 - - -def test_guarded_propagates_workflow_killed(monkeypatch): - """The kill signal still propagates through the decorator.""" - monkeypatch.setattr("sys.exit", lambda c: pytest.fail("sys.exit was called")) - - @guarded - def boom(): - raise WorkflowKilledInterrupt("wf-7", "killed via API") - - with pytest.raises(WorkflowKilledInterrupt): - boom() - - def test_handle_exit_code_kwarg(monkeypatch, capsys): """``handle(exit_code=42)`` honours the override.""" exits = [] @@ -142,97 +95,9 @@ def fake_exit(code): def test_no_init_required(): - """``handle`` / ``guarded`` must not depend on a runtime.""" + """``handle`` must not depend on a runtime.""" # If handle pulled in the runtime, importing this module would have # raised during the prior tests. Smoke-test the import path here. assert callable(handle) - assert callable(guarded) assert callable(nullrun.handle) - assert callable(nullrun.guarded) assert callable(nullrun.init_or_die) - - -# --------------------------------------------------------------------------- -# init_or_die -# --------------------------------------------------------------------------- - -class _FakeNoopRuntime: - """Sentinel returned by a stubbed init. init_or_die should pass - it through unchanged.""" - - -def test_init_or_die_returns_runtime(monkeypatch): - """On success, ``init_or_die`` returns whatever ``init()`` returned.""" - sentinel = _FakeNoopRuntime() - - def fake_init(**kwargs): - assert kwargs["api_key"] == "nr_live_test" - return sentinel - - monkeypatch.setattr("nullrun.init", fake_init) - monkeypatch.setattr("sys.exit", lambda c: pytest.fail("sys.exit was called")) - - result = nullrun.init_or_die(api_key="nr_live_test") - assert result is sentinel - - -def test_init_or_die_catches_missing_api_key(monkeypatch, capsys): - """NR-C001 from init() → catalog user-message + sys.exit(1).""" - from nullrun.breaker.exceptions import NullRunAuthenticationError - - def fake_init(**kwargs): - raise NullRunAuthenticationError( - "nullrun.init() requires an api_key.", - error_code="NR-C001", - user_action="Get an API key at https://app.nullrun.io/settings/api-keys", - ) - - exits = [] - - def fake_exit(code): - exits.append(code) - raise SystemExit(code) - - monkeypatch.setattr("nullrun.init", fake_init) - monkeypatch.setattr("sys.exit", fake_exit) - - with pytest.raises(SystemExit): - nullrun.init_or_die(api_key=None) - - captured = capsys.readouterr() - assert "configuration issue" in captured.err.lower() - assert exits == [1] - - -def test_init_or_die_propagates_unexpected(monkeypatch): - """Non-NullRun exceptions from init() propagate — not handled.""" - def fake_init(**kwargs): - raise ValueError("totally unrelated bug") - - monkeypatch.setattr("nullrun.init", fake_init) - monkeypatch.setattr("sys.exit", lambda c: pytest.fail("sys.exit was called")) - - with pytest.raises(ValueError): - nullrun.init_or_die(api_key="nr_live_test") - - -def test_init_or_die_exit_code_kwarg(monkeypatch, capsys): - """``init_or_die(exit_code=42)`` honours the override.""" - from nullrun.breaker.exceptions import NullRunAuthenticationError - - def fake_init(**kwargs): - raise NullRunAuthenticationError("no key", error_code="NR-C001") - - exits = [] - - def fake_exit(code): - exits.append(code) - raise SystemExit(code) - - monkeypatch.setattr("nullrun.init", fake_init) - monkeypatch.setattr("sys.exit", fake_exit) - - with pytest.raises(SystemExit): - nullrun.init_or_die(api_key=None, exit_code=42) - - assert exits == [42] \ No newline at end of file diff --git a/tests/test_init_or_die.py b/tests/test_init_or_die.py new file mode 100644 index 0000000..619636d --- /dev/null +++ b/tests/test_init_or_die.py @@ -0,0 +1,96 @@ +"""Tests for ``nullrun.init_or_die`` — the fail-fast wrapper around +``nullrun.init()``. + +Contract: + +* On success, returns whatever ``init()`` returns (the runtime + singleton). +* Catches :class:`nullrun.NullRunError` raised by ``init()`` — prints + the four-line developer report to stderr and exits with ``1`` (or + the ``exit_code`` override). +* Non-NullRun exceptions propagate unchanged. +""" +from __future__ import annotations + +import pytest + +import nullrun +from nullrun.breaker.exceptions import NullRunAuthenticationError + + +class _FakeNoopRuntime: + """Sentinel returned by a stubbed init. init_or_die should pass + it through unchanged.""" + + +def test_init_or_die_returns_runtime(monkeypatch): + """On success, ``init_or_die`` returns whatever ``init()`` returned.""" + sentinel = _FakeNoopRuntime() + + def fake_init(**kwargs): + assert kwargs["api_key"] == "nr_live_test" + return sentinel + + monkeypatch.setattr("nullrun.init", fake_init) + monkeypatch.setattr("sys.exit", lambda c: pytest.fail("sys.exit was called")) + + result = nullrun.init_or_die(api_key="nr_live_test") + assert result is sentinel + + +def test_init_or_die_catches_missing_api_key(monkeypatch, capsys): + """NR-C001 from init() → catalog user-message + sys.exit(1).""" + def fake_init(**kwargs): + raise NullRunAuthenticationError( + "nullrun.init() requires an api_key.", + error_code="NR-C001", + user_action="Get an API key at https://app.nullrun.io/settings/api-keys", + ) + + exits = [] + + def fake_exit(code): + exits.append(code) + raise SystemExit(code) + + monkeypatch.setattr("nullrun.init", fake_init) + monkeypatch.setattr("sys.exit", fake_exit) + + with pytest.raises(SystemExit): + nullrun.init_or_die(api_key=None) + + captured = capsys.readouterr() + assert "configuration issue" in captured.err.lower() + assert exits == [1] + + +def test_init_or_die_propagates_unexpected(monkeypatch): + """Non-NullRun exceptions from init() propagate — not handled.""" + def fake_init(**kwargs): + raise ValueError("totally unrelated bug") + + monkeypatch.setattr("nullrun.init", fake_init) + monkeypatch.setattr("sys.exit", lambda c: pytest.fail("sys.exit was called")) + + with pytest.raises(ValueError): + nullrun.init_or_die(api_key="nr_live_test") + + +def test_init_or_die_exit_code_kwarg(monkeypatch, capsys): + """``init_or_die(exit_code=42)`` honours the override.""" + def fake_init(**kwargs): + raise NullRunAuthenticationError("no key", error_code="NR-C001") + + exits = [] + + def fake_exit(code): + exits.append(code) + raise SystemExit(code) + + monkeypatch.setattr("nullrun.init", fake_init) + monkeypatch.setattr("sys.exit", fake_exit) + + with pytest.raises(SystemExit): + nullrun.init_or_die(api_key=None, exit_code=42) + + assert exits == [42] From ccb2670e6f18575f564ba6fede5291dc5e117411 Mon Sep 17 00:00:00 2001 From: Anatoly Maltsev Date: Fri, 25 Sep 2026 14:08:31 +0400 Subject: [PATCH 06/10] feat(sdk): auto-register atexit shutdown + unify init/init_or_die (0.18.3) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Surface changes (breaking): * `nullrun.init_or_die()` removed. CLI fail-fast is now a parameter on `init()`: `nullrun.init(fail_on_exit=True)` prints the same four-line developer report and `sys.exit(1)` on configuration failure. Default `init()` keeps embedder-friendly raise semantics. * `nullrun.shutdown()` is auto-registered via `atexit` inside `init()` so long-running scripts get a clean WS close on process exit without an explicit call. Calling `shutdown()` manually remains safe and idempotent. Implementation: * `init()` gains `fail_on_exit: bool = False`; the missing-API-key branch renders the dev-error report and exits 1 when True, otherwise raises `NullRunAuthenticationError` as before. * `init()` registers `shutdown` with `atexit` guarded by a module- level `_shutdown_atexit_registered` flag (no double-stack). * `shutdown()` resets that flag so a `shutdown() → init()` cycle re-registers cleanly for the new runtime. Test coverage: * `tests/test_init_or_die.py` replaced with `tests/test_init_fail_on_exit.py` (3 tests, stubbed runtime). * `tests/test_init_contract.py::TestInitRegistersAtexitShutdown` added (3 tests, stubbed runtime, no network). * `tests/test_handle.py` drops the `init_or_die` smoke assertion. Docs: * README Quickstart + `shutdown` docstring: no manual `atexit` call. * `docs/errors/NR-C001.md`: no `init_or_die` mention. * `CHANGELOG.md` 0.18.3 entry documents both breaking changes. * `pyproject.toml` bumped to 0.18.3. All 1403 tests pass. Ruff clean on touched files. Co-Authored-By: Claude Code --- CHANGELOG.md | 25 ++++++ README.md | 8 +- docs/errors/NR-C001.md | 8 +- pyproject.toml | 2 +- src/nullrun/__init__.py | 102 ++++++++++++++++-------- src/nullrun/_handle.py | 92 ++++------------------ src/nullrun/decorators.py | 8 +- src/nullrun/runtime.py | 4 +- src/nullrun/toolbox/langgraph.py | 6 +- tests/test_dev_error_report.py | 4 +- tests/test_handle.py | 1 - tests/test_init_contract.py | 128 +++++++++++++++++++++++++++++++ tests/test_init_fail_on_exit.py | 82 ++++++++++++++++++++ tests/test_init_or_die.py | 96 ----------------------- 14 files changed, 341 insertions(+), 225 deletions(-) create mode 100644 tests/test_init_fail_on_exit.py delete mode 100644 tests/test_init_or_die.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 3e828c6..57b1057 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,28 @@ +## [0.18.3] - 2026-09-25 + +### Surface (breaking) + +- `nullrun.init_or_die()` removed. CLI fail-fast behavior is now a + parameter on `init()`: `nullrun.init(fail_on_exit=True)` prints the + same four-line developer report and `sys.exit(1)` on configuration + failure. +- `nullrun.shutdown()` is now auto-registered via `atexit` inside + `init()` — long-running scripts get a clean WS close on process + exit without an explicit call. Calling `shutdown()` manually + remains safe and idempotent. + +### Lifecycle + +- `init()` gains a `fail_on_exit: bool = False` keyword argument. + When True, missing `NULLRUN_API_KEY` (NR-C001) prints the + developer-facing report and exits 1 instead of raising. Default + False preserves the embedder-friendly raise semantics. +- The `_shutdown_atexit_registered` module-level flag prevents + double registration when `init()` is called more than once. + +Top-level `dir(nullrun)` no longer exposes `init_or_die`. All other +symbols unchanged from 0.18.2. + ## [0.18.2] - 2026-09-22 ### Surface diff --git a/README.md b/README.md index cfa67c7..47c4d42 100644 --- a/README.md +++ b/README.md @@ -210,11 +210,17 @@ def my_agent(prompt: str) -> str: ``` -If you call `@protect` *before* `init_or_die()`, the SDK lazy-initializes +If you call `@protect` *before* `init()`, the SDK lazy-initializes the runtime from `NULLRUN_API_KEY` on the first decorated call. You can write your agent code with the decorator first and the init second — or skip `init` entirely if your environment is already configured. +For CLI scripts that want fail-fast on missing config, pass +`fail_on_exit=True` — the SDK prints a four-line developer report and +exits with code 1 instead of raising. `nullrun.shutdown()` is +auto-registered via `atexit` inside `init()`, so a clean WS close on +process exit happens without any explicit call. + --- ## How NullRun compares diff --git a/docs/errors/NR-C001.md b/docs/errors/NR-C001.md index 7fc7438..a58f4ed 100644 --- a/docs/errors/NR-C001.md +++ b/docs/errors/NR-C001.md @@ -16,10 +16,10 @@ call from the environment, so `init()` does not need to be called — but the env var must be present by the time the runtime is asked to gate a call. -If the developer chose to call `init()` or `init_or_die()` -explicitly, the same code surfaces earlier (at the explicit init -call, not at the first `@protect`). Both paths produce the same -typed exception. +If the developer chose to call `init()` explicitly with +`fail_on_exit=True`, the same code surfaces earlier (at the explicit +init call, not at the first `@protect`) and the SDK prints a +four-line developer report and `sys.exit(1)` instead of raising. ## Why this raises (instead of falling back) diff --git a/pyproject.toml b/pyproject.toml index a4cd2f6..7f80284 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -6,7 +6,7 @@ build-backend = "hatchling.build" name = "nullrun" # Full release history lives in CHANGELOG.md; only the current version # is pinned here. -version = "0.18.2" +version = "0.18.3" # Kept under the 200-char preview threshold so the full line is visible # without an "expand" click. The headline is the canonical §1 statement # from positioning.md — "runtime decision layer for tool-using AI agents" diff --git a/src/nullrun/__init__.py b/src/nullrun/__init__.py index 67eb2ef..945e9ce 100644 --- a/src/nullrun/__init__.py +++ b/src/nullrun/__init__.py @@ -11,7 +11,7 @@ Everything else exposed by ``nullrun`` is either runtime lifecycle (``init``, ``shutdown``, ``on_error``, ``status``), the structured exception hierarchy, or message/error-handling helpers -(``format_user_message``, ``handle``, ``init_or_die``). +(``format_user_message``, ``handle``). None of those are alternatives to ``@protect`` — they're setup and cleanup. @@ -40,6 +40,8 @@ def charge_customer(amount_cents: int, customer_id: str): from __future__ import annotations +import atexit +import sys import threading as _threading # Use lazy import inside __getattr__ instead of `import importlib` at @@ -50,6 +52,12 @@ def charge_customer(amount_cents: int, customer_id: str): # inside `init `. _init_lock = _threading.Lock() +# Tracks whether `shutdown` is currently registered with `atexit`. The +# flag persists across `init()` calls (so a second init does not stack +# a second atexit hook) and is reset by `shutdown()` so a fresh +# `init()` after a clean shutdown re-registers cleanly. +_shutdown_atexit_registered = False + # --------------------------------------------------------------------------- # Curated public surface # --------------------------------------------------------------------------- @@ -68,12 +76,11 @@ def shutdown(timeout: float = 2.0, flush: bool = True) -> None: this returns, any further ``nullrun.track(...)`` call or ``@protect``-decorated call is a no-op. - A long-running script that exits via ``sys.exit `` lets the - kernel RST the TCP socket, which the backend logs as WARN - "Connection reset without closing handshake". Calling - ``nullrun.shutdown `` before exit (or registering it via - ``atexit``) eliminates the noisy log. No-op if ``init `` was - never called. + ``init()`` auto-registers this function with ``atexit``, so + long-running scripts get a clean WS close on process exit without + any explicit call. Calling ``shutdown()`` manually is still safe + and idempotent — it is a no-op if ``init()`` was never called or + has already been shut down. Args: timeout: seconds to wait for the WS close handshake to @@ -91,9 +98,6 @@ def shutdown(timeout: float = 2.0, flush: bool = True) -> None: Example:: - import atexit - import nullrun - atexit.register(nullrun.shutdown) """ # Lazy import so the SDK module-import path stays light (mirrors # the pattern in `init` and `status`). @@ -102,6 +106,11 @@ def shutdown(timeout: float = 2.0, flush: bool = True) -> None: if runtime is None: return runtime.shutdown(flush=flush) + # Allow a future `init()` to re-register the atexit hook for the + # new runtime. Without this, after a `shutdown()` + `init()` cycle + # the second runtime would not be auto-shutdown on process exit. + global _shutdown_atexit_registered + _shutdown_atexit_registered = False def status(): @@ -215,6 +224,7 @@ def init( api_key: str | None = None, api_url: str | None = None, debug: bool = False, + fail_on_exit: bool = False, ): """ Initialize the NullRun SDK. Call once at application startup. @@ -223,12 +233,24 @@ def init( gate calls the backend, and a missing key would silently bypass every backend gate. Pass `api_key=...` explicitly or set the `NULLRUN_API_KEY` environment variable before calling `init `. If - neither is set, `init ` raises `NullRunAuthenticationError`. + neither is set, `init ` raises `NullRunAuthenticationError` (or, + with ``fail_on_exit=True``, prints the four-line developer report + to stderr and exits with code 1 — the CLI-friendly path). + + ``init()`` also auto-registers ``nullrun.shutdown()`` via + ``atexit``, so a process exit after init gets a clean WS close + without an explicit shutdown call. Args: api_key: NullRun API key (or NULLRUN_API_KEY env var). Required. api_url: Gateway URL (or NULLRUN_API_URL env var) debug: Enable debug logging + fail_on_exit: when True, configuration failures print the + developer-facing report to stderr and ``sys.exit(1)`` + instead of raising. Use this for CLI scripts that want a + clean exit on missing ``NULLRUN_API_KEY``; library + embedders (FastAPI startup, Jupyter) should leave it + False and catch the exception instead. Note: the background control-plane listener (WebSocket + HTTP poll) is always started on `init `. To disable it, construct `NullRunRuntime` @@ -239,7 +261,7 @@ def init( Raises: NullRunAuthenticationError: if neither `api_key` nor - `NULLRUN_API_KEY` is set. + `NULLRUN_API_KEY` is set AND ``fail_on_exit`` is False. Example: import nullrun @@ -247,8 +269,8 @@ def init( nullrun.init(api_key="your-key") @nullrun.protect - def my_agent: - return agent.run + def my_agent: + return agent.run """ import logging import os @@ -258,18 +280,13 @@ def my_agent: if debug: logger.setLevel(logging.DEBUG) - # to a NullRunNoop stub in `local_mode`, which silently bypassed every - # backend gate (budget, policy, control plane). That was a real - # safety hole — production callers were unaware their policies were - # not being enforced. We raise instead so the misconfiguration is - # caught at startup rather than producing silent allow-all decisions. - # Strip whitespace from either the kwarg or the env before the truthiness - # check. Python `or` alone accepts " " / "\t" / "\n" as truthy, which - # would let a whitespace-only api_key pass init() and reach the gateway - # as a malformed `Authorization: Bearer ` header. The strip preserves - # embedded legitimate characters (e.g. " nr_live_xxx " is normalised - # to the canonical form so HMAC signing sees the same value on both - # sides of the wire). + # Strip whitespace from either the kwarg or the env before the + # truthiness check. Python `or` alone accepts " " / "\t" / "\n" as + # truthy, which would let a whitespace-only api_key pass init() + # and reach the gateway as a malformed `Authorization: Bearer ` + # header. The strip preserves embedded legitimate characters (e.g. + # " nr_live_xxx " is normalised to the canonical form so HMAC + # signing sees the same value on both sides of the wire). raw_key = api_key if api_key is not None else os.getenv("NULLRUN_API_KEY") resolved_key = raw_key.strip() if isinstance(raw_key, str) else None if not resolved_key: @@ -295,6 +312,21 @@ def my_agent: "operate without credentials." ), ) + # CLI path: render the same four-line developer report + # ``handle()`` uses, then sys.exit(1). Library embedders leave + # ``fail_on_exit=False`` (default) so the exception propagates + # into their own try/except and the host process stays alive + # (FastAPI startup, Jupyter, REPL). + if fail_on_exit: + from nullrun._handle import _render_dev_error_report + from nullrun.messages import format_user_message + + try: + report = _render_dev_error_report(err, format_user_message(err)) + except Exception: # noqa: BLE001 + report = format_user_message(err) + print(report, file=sys.stderr) + sys.exit(1) # Layer 2: fire the on_error hook BEFORE the raise so the # hook sees the call stack still live. Stage = "init" so a # log-based hook can attribute the failure to startup @@ -359,6 +391,17 @@ def my_agent: # write that every consumer sees. NullRunRuntime._instance = runtime + # Auto-register shutdown with atexit so long-running scripts get a + # clean WS close on process exit without an explicit call. The + # flag guard makes the registration idempotent across multiple + # ``init()`` calls (the C3 fix shuts down the previous runtime + # first; the atexit hook stays the same and is no-op when the + # runtime is already torn down). + global _shutdown_atexit_registered + if not _shutdown_atexit_registered: + atexit.register(shutdown) + _shutdown_atexit_registered = True + # v3.12 / 0.12.0 — server-minted execution_id default ON. Probe # the backend's /api/v1/capabilities endpoint and log any # version mismatch so the operator sees the gap at startup @@ -518,7 +561,6 @@ def my_agent: # which shadows the lazy export and breaks ``from nullrun import # handle``. "handle": ("nullrun._handle", "handle"), - "init_or_die": ("nullrun._handle", "init_or_die"), # ADR-009 P1 — governance audit surface (typed wire classes). # Users reach these as `from nullrun import AuditQuery` / # `from nullrun.audit import ...`. The runtime exposes @@ -615,12 +657,8 @@ def __dir__() -> list[str]: # the context manager (``with nullrun.handle: ``). It translates # any ``NullRunError`` into ``print(format_user_message(exc))`` + # ``sys.exit(1)``; ``WorkflowKilledInterrupt`` propagates. - # ``init_or_die`` is the convenience wrapper around ``init`` - # that catches NR-C001 "no api_key" at startup and exits - # cleanly — without it the user sees a raw traceback before - # any ``with handle: `` block is in scope. + # CLI fail-fast on missing api_key is `init(fail_on_exit=True)`. "handle", - "init_or_die", ] # The SDK-side ``decision_history`` module was deleted. Decision diff --git a/src/nullrun/_handle.py b/src/nullrun/_handle.py index 58d1bc4..d0a272a 100644 --- a/src/nullrun/_handle.py +++ b/src/nullrun/_handle.py @@ -8,19 +8,15 @@ branch on a specific ``error_code`` — but it is **not** the default. For the common "I just want to run my agent and print a friendly -message on failure" case, this module provides two one-liners: +message on failure" case, this module provides one one-liner: -*:func:`nullrun.handle` — context manager. -*:func:`nullrun.init_or_die` — convenience wrapper around -:func:`nullrun.init` that catches the ``NR-C001`` "no api_key" - failure at startup and exits cleanly. - -Both translate any:class:`nullrun.NullRunError` into a structured -developer-facing report (error code + what was attempted + where it -came from + the underlying reason + how to fix it) and then exit -``1``. The end-user-friendly wording from -:func:`nullrun.format_user_message` is included as the headline so -end-user scripts don't need to branch on the wire shape. +*:func:`nullrun.handle` — context manager that translates any + :class:`nullrun.NullRunError` into a structured developer-facing + report (error code + what was attempted + where it came from + the + underlying reason + how to fix it) and then exits ``1``. The + end-user-friendly wording from :func:`nullrun.format_user_message` + is included as the headline so end-user scripts don't need to + branch on the wire shape. :class:`nullrun.WorkflowKilledInterrupt` inherits from :class:`nullrun.NullRunError` (see the class docstring), so a @@ -30,10 +26,10 @@ the agent loop. Non-NullRun exceptions also propagate unchanged. -``init_or_die`` exists because:func:`nullrun.init` is typically -called at module top-level — before any ``with handle: `` block is -in scope. Without it, a missing ``NULLRUN_API_KEY`` env var produces -a raw traceback. +CLI scripts that want the same fail-fast behavior at startup should +call ``nullrun.init(fail_on_exit=True)`` instead of an ``init_or_die`` +wrapper — the four-line developer report is rendered identically +and the process exits ``1`` on missing ``NULLRUN_API_KEY``. Why a separate module --------------------- @@ -230,66 +226,4 @@ def handle(*, exit_code: int = 1): sys.exit(exit_code) -def init_or_die(*, api_key: str | None = None, api_url: str | None = None, - debug: bool = False, exit_code: int = 1): - """Call:func:`nullrun.init` and exit cleanly on configuration failure. - -:func:`nullrun.init` is typically the first thing a script does - before any ``with nullrun.handle: `` block is in scope. A missing - ``api_key`` therefore produces a raw traceback — not a friendly - exit. ``init_or_die`` closes that gap by catching the startup - :class:`nullrun.NullRunError` (NR-C001 "no api_key"), printing - the catalog user-message, and exiting. - - On success returns the :class:`nullrun.NullRunRuntime` singleton - that ``init `` returns — assign it if you need it, ignore it - otherwise:: - - from nullrun import init_or_die, protect, shutdown - - init_or_die(api_key=os.environ["NULLRUN_API_KEY"]) - - @protect - def my_agent(prompt): - return call_llm(prompt) - - if __name__ == "__main__": - try: - with nullrun.handle: - print(my_agent("hello")) - finally: - shutdown - - Args: - api_key: NullRun API key (or NULLRUN_API_KEY env var). - api_url: Gateway URL (or NULLRUN_API_URL env var). - debug: Enable debug logging on the runtime. - exit_code: Process exit status to use when init fails. - - Returns: - The runtime singleton returned by ``init ``. - """ - # Lazy import — ``init`` pulls in the runtime + transport stack. - # Skipping that when init is never called keeps the import path - # of ``from nullrun import init_or_die`` light. - from nullrun import init - try: - return init(api_key=api_key, api_url=api_url, debug=debug) - except NullRunError as exc: - # Same structured report as ``handle()`` -- a - # "There's a configuration issue. Please contact support." - # which gave the developer zero actionable detail. The - # four-line report here names the missing env var, the URL - # to obtain a key, and the docs page so the user can self- - # serve without opening a support ticket. - try: - report = _render_dev_error_report( - exc, format_user_message(exc) - ) - except Exception: # noqa: BLE001 - report = format_user_message(exc) - print(report, file=sys.stderr) - sys.exit(exit_code) - - -__all__ = ["handle", "init_or_die"] \ No newline at end of file +__all__ = ["handle"] \ No newline at end of file diff --git a/src/nullrun/decorators.py b/src/nullrun/decorators.py index 28090a5..6ebeffe 100644 --- a/src/nullrun/decorators.py +++ b/src/nullrun/decorators.py @@ -309,7 +309,7 @@ def _get_or_create_runtime() -> NullRunRuntime: not a silent allow-all. After obtaining the runtime, lazily triggers `auto_instrument()` so - a user who writes only `@protect` (without calling `init_or_die()` + a user who writes only `@protect` (without calling `init()` first) still gets vendor SDK detection + token capture. The lazy trigger is idempotent — multiple `@protect` calls in the same process converge on a single `auto_instrument()` invocation. The @@ -333,9 +333,9 @@ def _get_or_create_runtime() -> NullRunRuntime: # Lazy auto-instrumentation trigger (zero-config decorator path). # -# The user-facing API is `nullrun.init_or_die()` which calls `init()`, -# which calls `auto_instrument(runtime)` directly (see -# `nullrun/__init__.py::init`). However, a user who writes only +# The user-facing API is `nullrun.init()` which calls +# `auto_instrument(runtime)` directly (see `nullrun/__init__.py::init`). +# However, a user who writes only # ``@nullrun.protect`` without calling ``init_or_die()`` first would # still create a runtime via ``NullRunRuntime.get_instance()`` — but # no vendor SDK patches would be installed, so token capture would be diff --git a/src/nullrun/runtime.py b/src/nullrun/runtime.py index 773ffa0..a8dfc62 100644 --- a/src/nullrun/runtime.py +++ b/src/nullrun/runtime.py @@ -282,7 +282,7 @@ def register_strict_mode_forced(tool_name: str) -> None: Called by ``@sensitive(impact=...)`` at decoration time. The name stays in the module-level set until process exit; it - is intentionally not cleared by ``init_or_die()`` so a + is intentionally not cleared by ``shutdown()`` so a second-runtime reinit does not silently drop a tool out of strict mode. """ @@ -1109,7 +1109,7 @@ def _maybe_warn_zero_activity(self) -> None: "Most common causes: " "(1) the LLM call uses a raw httpx client without " "NullRun's instrumentation patches (call " - "nullrun.init_or_die() before the first request), " + "nullrun.init() before the first request), " "(2) a custom transport / non-HTTP vendor (gRPC, " "WebSocket, SDK-internal socket), " "(3) a framework not in the auto-detection table " diff --git a/src/nullrun/toolbox/langgraph.py b/src/nullrun/toolbox/langgraph.py index 6fb861f..bb63452 100644 --- a/src/nullrun/toolbox/langgraph.py +++ b/src/nullrun/toolbox/langgraph.py @@ -1,16 +1,16 @@ """ LangGraph toolbox helpers for NullRun. -``nullrun.init_or_die()`` (or ``@nullrun.protect`` on the agent +``nullrun.init()`` (or ``@nullrun.protect`` on the agent function) auto-patches ``langgraph.pregel.Pregel`` via ``nullrun.instrumentation.auto.patch_langgraph_compiled``, which covers every supported LangGraph invocation path. Use the auto-patch:: - from nullrun import init_or_die, protect + from nullrun import init, protect - init_or_die() + init() @protect def my_agent(prompt): diff --git a/tests/test_dev_error_report.py b/tests/test_dev_error_report.py index 69340e8..74b0584 100644 --- a/tests/test_dev_error_report.py +++ b/tests/test_dev_error_report.py @@ -1,6 +1,6 @@ """ -Tests for the developer-facing error report rendered by ``handle`` and -``init_or_die``. +Tests for the developer-facing error report rendered by ``handle`` +and the CLI ``init(fail_on_exit=True)`` path. Pre-fix (2026-09-22), the catch-all exit path printed only the catalog user-message ("There's a configuration issue. Please contact support.") diff --git a/tests/test_handle.py b/tests/test_handle.py index efe41ee..fdc33da 100644 --- a/tests/test_handle.py +++ b/tests/test_handle.py @@ -100,4 +100,3 @@ def test_no_init_required(): # raised during the prior tests. Smoke-test the import path here. assert callable(handle) assert callable(nullrun.handle) - assert callable(nullrun.init_or_die) diff --git a/tests/test_init_contract.py b/tests/test_init_contract.py index 427dc30..76a7f00 100644 --- a/tests/test_init_contract.py +++ b/tests/test_init_contract.py @@ -432,3 +432,131 @@ def test_runtime_init_raises_on_whitespace_only(self, monkeypatch, mock_api): monkeypatch.delenv("NULLRUN_API_KEY", raising=False) with pytest.raises(NullRunAuthenticationError, match="api_key"): NullRunRuntime(api_key=" ") + + +class TestInitRegistersAtexitShutdown: + """``init()`` auto-registers ``nullrun.shutdown`` with ``atexit`` + so long-running scripts get a clean WS close on process exit + without an explicit shutdown call. Users should not have to + remember to call ``shutdown()`` themselves. + + These tests stub ``NullRunRuntime`` so they do no network at + all — they only verify the atexit wiring in ``init()`` and + ``shutdown()``. + """ + + @staticmethod + def _stub_runtime(monkeypatch): + """Replace ``NullRunRuntime`` with a no-network no-op so the + atexit tests stay fully offline (no httpx, no real DNS).""" + from nullrun.runtime import NullRunRuntime + + class _FakeRuntime: + def __init__(self, **kwargs): + self.kwargs = kwargs + self.api_url = kwargs.get("api_url", "") + self.api_key = kwargs.get("api_key", "") + self.organization_id = "fake-org" + self.workflow_id = "fake-wf" + self._instance = self + + def shutdown(self, flush=True): + pass + + monkeypatch.setattr(NullRunRuntime, "__init__", _FakeRuntime.__init__) + monkeypatch.setattr(NullRunRuntime, "shutdown", _FakeRuntime.shutdown) + # Skip the capability probe + auto_instrument (both touch network + # even with the stub, because they're called as module-level + # functions and read the real ``api_url`` we passed in). + monkeypatch.setattr( + "nullrun.capabilities.probe_capabilities", lambda url: None + ) + monkeypatch.setattr( + "nullrun.instrumentation.auto.auto_instrument", lambda rt: None + ) + + def test_init_registers_shutdown_with_atexit(self, monkeypatch): + """After ``init()`` succeeds, ``shutdown`` is in atexit's + registered list.""" + import atexit as atexit_mod + + import nullrun as nullrun_mod + + # Reset the module-level flag — other tests in this session + # may have already registered shutdown() with atexit, which + # would cause the flag guard to skip re-registration. + monkeypatch.setattr(nullrun_mod, "_shutdown_atexit_registered", False) + self._stub_runtime(monkeypatch) + + registered = [] + + def fake_register(func, *args, **kwargs): + registered.append(func) + + monkeypatch.setattr(atexit_mod, "register", fake_register) + monkeypatch.setenv("NULLRUN_API_KEY", "test-key-12345678") + + nullrun_mod.init() + try: + assert nullrun_mod.shutdown in registered + finally: + nullrun_mod.shutdown() + + def test_init_does_not_double_register_atexit(self, monkeypatch): + """Two ``init()`` calls (the C3 fix scenario) must not stack + two atexit entries for ``shutdown``.""" + import atexit as atexit_mod + + import nullrun as nullrun_mod + + monkeypatch.setattr(nullrun_mod, "_shutdown_atexit_registered", False) + self._stub_runtime(monkeypatch) + + registered = [] + + def fake_register(func, *args, **kwargs): + registered.append(func) + + monkeypatch.setattr(atexit_mod, "register", fake_register) + monkeypatch.setenv("NULLRUN_API_KEY", "test-key-12345678") + + nullrun_mod.init() + nullrun_mod.init() + try: + # shutdown should appear at most ONCE in registered (the + # flag guard prevents stacking). + assert registered.count(nullrun_mod.shutdown) <= 1 + finally: + nullrun_mod.shutdown() + + def test_shutdown_resets_atexit_flag(self, monkeypatch): + """After ``shutdown()``, a fresh ``init()`` re-registers + ``shutdown`` with atexit so the new runtime gets a clean + exit.""" + import atexit as atexit_mod + + import nullrun as nullrun_mod + + monkeypatch.setattr(nullrun_mod, "_shutdown_atexit_registered", False) + self._stub_runtime(monkeypatch) + + registered = [] + + def fake_register(func, *args, **kwargs): + registered.append(func) + + monkeypatch.setattr(atexit_mod, "register", fake_register) + monkeypatch.setenv("NULLRUN_API_KEY", "test-key-12345678") + + nullrun_mod.init() + initial_count = registered.count(nullrun_mod.shutdown) + nullrun_mod.shutdown() + nullrun_mod.init() + try: + final_count = registered.count(nullrun_mod.shutdown) + assert final_count == initial_count + 1, ( + f"expected re-registration after shutdown, " + f"got {initial_count} -> {final_count}" + ) + finally: + nullrun_mod.shutdown() diff --git a/tests/test_init_fail_on_exit.py b/tests/test_init_fail_on_exit.py new file mode 100644 index 0000000..3c39313 --- /dev/null +++ b/tests/test_init_fail_on_exit.py @@ -0,0 +1,82 @@ +"""Tests for ``nullrun.init(fail_on_exit=True)`` — the CLI fail-fast path. + +Contract: + +* On success, returns whatever the runtime ``init()`` returns. +* On ``NullRunAuthenticationError`` (e.g. NR-C001 missing api_key), + prints the same four-line developer report that ``handle()`` uses + and ``sys.exit(1)``. +* The default ``init()`` (``fail_on_exit=False``) still raises — the + library-friendly path. +""" +from __future__ import annotations + +import pytest + +import nullrun +from nullrun.breaker.exceptions import NullRunAuthenticationError + + +def test_init_fail_on_exit_does_not_exit_on_happy_path(monkeypatch): + """``fail_on_exit=True`` only triggers on config errors, not on + successful init. Stub ``NullRunRuntime`` so we never touch the + network and just verify ``sys.exit`` is not called.""" + from nullrun.runtime import NullRunRuntime + + class _FakeRuntime: + def __init__(self, **kwargs): + self.kwargs = kwargs + self.api_url = kwargs.get("api_url", "") + self.api_key = kwargs.get("api_key", "") + self.organization_id = "fake-org" + self.workflow_id = "fake-wf" + self._instance = self + + def shutdown(self, flush=True): + pass + + monkeypatch.setattr(NullRunRuntime, "__init__", _FakeRuntime.__init__) + monkeypatch.setattr(NullRunRuntime, "shutdown", _FakeRuntime.shutdown) + # Skip the capability probe + auto_instrument (both touch network) + monkeypatch.setattr( + "nullrun.capabilities.probe_capabilities", lambda url: None + ) + monkeypatch.setattr( + "nullrun.instrumentation.auto.auto_instrument", lambda rt: None + ) + + monkeypatch.setenv("NULLRUN_API_KEY", "nr_live_test") + monkeypatch.setattr("sys.exit", lambda c: pytest.fail("sys.exit was called")) + + rt = nullrun.init(fail_on_exit=True) + assert rt is not None + + +def test_init_fail_on_exit_exits_on_missing_api_key(monkeypatch, capsys): + """NR-C001 from init() → catalog user-message + sys.exit(1).""" + monkeypatch.delenv("NULLRUN_API_KEY", raising=False) + + exits = [] + + def fake_exit(code): + exits.append(code) + raise SystemExit(code) + + monkeypatch.setattr("sys.exit", fake_exit) + + with pytest.raises(SystemExit): + nullrun.init(api_key=None, fail_on_exit=True) + + captured = capsys.readouterr() + assert "configuration issue" in captured.err.lower() + assert exits == [1] + + +def test_init_default_raises_does_not_exit(monkeypatch): + """The default ``init()`` (without ``fail_on_exit=True``) still + raises on missing api_key — fail_on_exit is opt-in.""" + monkeypatch.delenv("NULLRUN_API_KEY", raising=False) + monkeypatch.setattr("sys.exit", lambda c: pytest.fail("sys.exit was called")) + + with pytest.raises(NullRunAuthenticationError): + nullrun.init(api_key=None) diff --git a/tests/test_init_or_die.py b/tests/test_init_or_die.py deleted file mode 100644 index 619636d..0000000 --- a/tests/test_init_or_die.py +++ /dev/null @@ -1,96 +0,0 @@ -"""Tests for ``nullrun.init_or_die`` — the fail-fast wrapper around -``nullrun.init()``. - -Contract: - -* On success, returns whatever ``init()`` returns (the runtime - singleton). -* Catches :class:`nullrun.NullRunError` raised by ``init()`` — prints - the four-line developer report to stderr and exits with ``1`` (or - the ``exit_code`` override). -* Non-NullRun exceptions propagate unchanged. -""" -from __future__ import annotations - -import pytest - -import nullrun -from nullrun.breaker.exceptions import NullRunAuthenticationError - - -class _FakeNoopRuntime: - """Sentinel returned by a stubbed init. init_or_die should pass - it through unchanged.""" - - -def test_init_or_die_returns_runtime(monkeypatch): - """On success, ``init_or_die`` returns whatever ``init()`` returned.""" - sentinel = _FakeNoopRuntime() - - def fake_init(**kwargs): - assert kwargs["api_key"] == "nr_live_test" - return sentinel - - monkeypatch.setattr("nullrun.init", fake_init) - monkeypatch.setattr("sys.exit", lambda c: pytest.fail("sys.exit was called")) - - result = nullrun.init_or_die(api_key="nr_live_test") - assert result is sentinel - - -def test_init_or_die_catches_missing_api_key(monkeypatch, capsys): - """NR-C001 from init() → catalog user-message + sys.exit(1).""" - def fake_init(**kwargs): - raise NullRunAuthenticationError( - "nullrun.init() requires an api_key.", - error_code="NR-C001", - user_action="Get an API key at https://app.nullrun.io/settings/api-keys", - ) - - exits = [] - - def fake_exit(code): - exits.append(code) - raise SystemExit(code) - - monkeypatch.setattr("nullrun.init", fake_init) - monkeypatch.setattr("sys.exit", fake_exit) - - with pytest.raises(SystemExit): - nullrun.init_or_die(api_key=None) - - captured = capsys.readouterr() - assert "configuration issue" in captured.err.lower() - assert exits == [1] - - -def test_init_or_die_propagates_unexpected(monkeypatch): - """Non-NullRun exceptions from init() propagate — not handled.""" - def fake_init(**kwargs): - raise ValueError("totally unrelated bug") - - monkeypatch.setattr("nullrun.init", fake_init) - monkeypatch.setattr("sys.exit", lambda c: pytest.fail("sys.exit was called")) - - with pytest.raises(ValueError): - nullrun.init_or_die(api_key="nr_live_test") - - -def test_init_or_die_exit_code_kwarg(monkeypatch, capsys): - """``init_or_die(exit_code=42)`` honours the override.""" - def fake_init(**kwargs): - raise NullRunAuthenticationError("no key", error_code="NR-C001") - - exits = [] - - def fake_exit(code): - exits.append(code) - raise SystemExit(code) - - monkeypatch.setattr("nullrun.init", fake_init) - monkeypatch.setattr("sys.exit", fake_exit) - - with pytest.raises(SystemExit): - nullrun.init_or_die(api_key=None, exit_code=42) - - assert exits == [42] From 83421e0a2a025ad51c006319c479b736b3d1f597 Mon Sep 17 00:00:00 2001 From: Anatoly Maltsev Date: Fri, 25 Sep 2026 15:52:20 +0400 Subject: [PATCH 07/10] feat(sdk): rename handle->guard and drop top-level status() (0.18.4) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Surface changes (breaking): * `nullrun.handle` renamed to `nullrun.guard`. Same `@contextmanager` body (catches `NullRunError`, re-raises `WorkflowKilledInterrupt`, renders the four-line developer report, `sys.exit(1)` on failure). The verb `guard` was freed in 0.18.2 (f1721f2) when `@guarded` was removed and reads as a single verb alongside `init` / `shutdown` / `on_error`. Migrate by replacing `from nullrun import handle` with `from nullrun import guard` and `with nullrun.handle:` with `with nullrun.guard():`. * Top-level `nullrun.status()` removed. Reach the snapshot via `nullrun.get_runtime().status()` (returns the same frozen `NullRunStatus` dataclass). The wrapper's only role was raising `NullRunConfigError(NR-C004)` when no runtime was bound — that path is now `get_runtime()`'s job. Curated surface after 0.18.4: 20 names (was 21 in 0.18.3 — net -1 because `handle`->`guard` is in-place and `status` is removed). Implementation: * `src/nullrun/_handle.py`: rename `handle` -> `guard` (function body unchanged). Module file name stays `_handle.py` for the submodule-shadowing reason documented in that file's docstring (only the public function name is `guard`, the module file name is observable to no external caller — only `from nullrun import guard` is the public surface). * `src/nullrun/__init__.py`: drop `def status():` (47 lines); rename `handle` -> `guard` in `_LAZY_EXPORTS` and `__all__`; update the module docstring + `_LAZY_EXPORTS` comment blocks accordingly. * `docs/errors/NR-C004.md`: replace `nullrun.status()` references with `nullrun.get_runtime()` / `runtime.status()`; add a 0.18.4 note at the top calling out the wrapper removal. Test coverage: * Rename `tests/test_handle.py` -> `tests/test_guard.py` (7 -> 7 tests, identical body, mechanical rename + a new `test_handle_removed_from_public_surface` regression guard). * Rewrite `tests/test_status.py`: 23 of 24 sites moved from `nullrun.status()` to `rt.status()` on the local runtime. The `TestNoRuntime` class (2 tests) is dropped — the contract it pinned no longer exists without the top-level wrapper. The `TestPublicAPI` class (3 tests) becomes `TestStatusRemovedFromTopLevel` (2 regression-guard tests asserting `status` is NOT in `dir(nullrun)` or `__all__`). * `tests/test_dev_error_report.py`: handle -> guard throughout (4 tests renamed, bodies unchanged). All 1401 tests pass on Python 3.11 / Windows. Ruff clean on all touched files. Coverage regenerates to 81.34% (above the `fail_under = 80` gate) after dropping the stale `.coverage` file that referenced a removed `extractor.py`. Co-Authored-By: Claude Code --- CHANGELOG.md | 34 +++++ docs/errors/NR-C000.md | 2 +- docs/errors/NR-C001.md | 4 +- docs/errors/NR-C004.md | 30 +++-- docs/errors/README.md | 2 +- pyproject.toml | 2 +- src/nullrun/__init__.py | 100 +++++---------- src/nullrun/__version__.py | 2 +- src/nullrun/_handle.py | 35 ++++-- tests/test_dev_error_report.py | 32 ++--- tests/{test_handle.py => test_guard.py} | 55 ++++---- tests/test_status.py | 161 +++++++++--------------- 12 files changed, 227 insertions(+), 232 deletions(-) rename tests/{test_handle.py => test_guard.py} (62%) diff --git a/CHANGELOG.md b/CHANGELOG.md index 57b1057..47c55db 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,37 @@ +## [0.18.4] - 2026-09-25 + +### Surface (breaking) + +- `nullrun.handle` renamed to `nullrun.guard`. Same `@contextmanager` + body (catches `NullRunError`, re-raises `WorkflowKilledInterrupt`, + renders the four-line developer report, `sys.exit(1)` on failure). + The name `guard` was freed in 0.18.2 when `@guarded` was removed. + Migrate by replacing `from nullrun import handle` / + `with nullrun.handle:` with `from nullrun import guard` / + `with nullrun.guard():`. +- Top-level `nullrun.status()` removed. Reach the snapshot via + `nullrun.get_runtime().status()` (returns the same frozen + `NullRunStatus` dataclass). The wrapper's only role was raising + `NullRunConfigError(NR-C004)` when no runtime was bound — that + path is now `get_runtime()`'s job. `NullRunStatus` itself + remains importable as a type. + +### Curated surface after 0.18.4 + +```text +__version__, init, protect, shutdown, on_error, guard, +NullRunError, NullRunAuthError, NullRunConfigError, +NullRunBackendError, NullRunBudgetError, NullRunToolBlockedError, +WorkflowKilledInterrupt, NullRunWorkflowKilledError, +NullRunMcpDestructiveBlockedError, +NullRunMcpReadonlyBypassBlockedError, +NullRunMcpApprovalRequiredError, +NullRunApprovalDbUnavailableError, +format_user_message, set_user_message +``` + +(`status` and `handle` dropped; `guard` added. Net -1 symbol vs 0.18.3.) + ## [0.18.3] - 2026-09-25 ### Surface (breaking) diff --git a/docs/errors/NR-C000.md b/docs/errors/NR-C000.md index ab2e4a6..767d6e0 100644 --- a/docs/errors/NR-C000.md +++ b/docs/errors/NR-C000.md @@ -28,4 +28,4 @@ to override. - `NR-C001` — `nullrun.init()` called with no api_key. - `NR-C003` — `get_org_status()` called before the runtime is bound. -- `NR-C004` — `nullrun.status()` called before `nullrun.init()`. +- `NR-C004` — `nullrun.get_runtime()` (or `runtime.status()`) called before `nullrun.init()`. diff --git a/docs/errors/NR-C001.md b/docs/errors/NR-C001.md index a58f4ed..0cc9c42 100644 --- a/docs/errors/NR-C001.md +++ b/docs/errors/NR-C001.md @@ -64,7 +64,7 @@ try: result = my_agent(prompt) except NullRunConfigError as exc: if exc.error_code == "NR-C001": - # Show the user the dashboard link inline. With `with nullrun.handle():` + # Show the user the dashboard link inline. With `with nullrun.guard():` # around the call, the SDK will print the four-line developer report # automatically — you only need this catch for custom error UI. return render_onboarding(api_key_help_url=exc.user_action) @@ -75,4 +75,4 @@ except NullRunConfigError as exc: - `NR-A001` / `NR-A002` / `NR-A003` — key provided but rejected. - `NR-C003` — runtime bound, but no `org_id` available for `get_org_status()`. -- `NR-C004` — `nullrun.status()` called before the runtime is bound. +- `NR-C004` — `nullrun.get_runtime()` / `runtime.status()` called before the runtime is bound. diff --git a/docs/errors/NR-C004.md b/docs/errors/NR-C004.md index 3c8d15e..e082f0d 100644 --- a/docs/errors/NR-C004.md +++ b/docs/errors/NR-C004.md @@ -1,4 +1,10 @@ -# NR-C004 — `nullrun.status()` called before `nullrun.init()` +# NR-C004 — `get_runtime()` / `runtime.status()` called before `nullrun.init()` + +> **0.18.4 note:** the top-level `nullrun.status()` wrapper was removed +> in 0.18.4; reach the snapshot via `nullrun.get_runtime().status()` +> instead. `get_runtime()` raises `NullRunConfigError(NR-C004)` when +> no runtime has been bound. This doc retains the same shape; only the +> entry-point wording has changed. | Field | Value | |---|---| @@ -6,13 +12,14 @@ | **Category** | Configuration | | **Exception class** | `NullRunConfigError` | | **Retryable** | No | -| **Default `user_action`** | "Call `nullrun.init(api_key='nr_live_...')` before calling `nullrun.status()`. The snapshot only makes sense once the SDK has a runtime bound to the API key." | +| **Default `user_action`** | "Call `nullrun.init(api_key='nr_live_...')` before requesting the runtime snapshot. The snapshot only makes sense once the SDK has a runtime bound to the API key." | ## When -Raised by `nullrun.status()` when the runtime has not been initialised -yet. `status()` returns a snapshot of the runtime's account state — -without an active runtime, there is nothing to snapshot. +Raised by `nullrun.get_runtime()` when the runtime has not been +initialised yet. `runtime.status()` returns a snapshot of the +runtime's account state — without an active runtime, there is +nothing to snapshot. This is distinct from `NR-C001` (no api_key at all): the call to `init()` was simply never made, or the runtime was shut down with @@ -22,9 +29,10 @@ This is distinct from `NR-C001` (no api_key at all): the call to - Forgot to call `nullrun.init()` at process startup. - Called `nullrun.shutdown()` (or `runtime.shutdown()`) at module - unload and then `nullrun.status()` from a signal handler or + unload and then `runtime.status()` from a signal handler or finalizer that ran afterwards. -- Calling `status()` from a test fixture that did not auto-init. +- Calling `runtime.status()` from a test fixture that did not + auto-init. ## How to fix @@ -32,19 +40,21 @@ This is distinct from `NR-C001` (no api_key at all): the call to env var by passing `api_key=None`) at the top of the entrypoint. 2. If the runtime was intentionally shut down, skip the snapshot call rather than re-initialising after shutdown. -3. In tests, use the `nullrun_test_runtime` fixture from +3. In tests, use the `make_runtime` fixture from `tests/conftest.py` instead of constructing one manually. ## Catch pattern ```python from nullrun.breaker.exceptions import NullRunConfigError +from nullrun import get_runtime try: - snap = nullrun.status() + rt = get_runtime() + snap = rt.status() except NullRunConfigError as exc: if exc.error_code == "NR-C004": - log.error("status() called before init: %s", exc.user_action) + log.error("runtime not bound: %s", exc.user_action) return None raise ``` diff --git a/docs/errors/README.md b/docs/errors/README.md index 58b9e92..591a458 100644 --- a/docs/errors/README.md +++ b/docs/errors/README.md @@ -27,7 +27,7 @@ The codes follow a `NR-` pattern: | `NR-C000` | Generic config error (default on `NullRunConfigError`; subclasses override) | [NR-C000](NR-C000.md) | | `NR-C001` | `nullrun.init()` called with no api_key (no param, no env) | [NR-C001](NR-C001.md) | | `NR-C003` | `get_org_status()` called before the runtime is bound to an org | [NR-C003](NR-C003.md) | -| `NR-C004` | `nullrun.status()` called before `nullrun.init()` | [NR-C004](NR-C004.md) | +| `NR-C004` | `nullrun.get_runtime()` (or `runtime.status()`) called before `nullrun.init()` | [NR-C004](NR-C004.md) | ### Authentication (NR-A) diff --git a/pyproject.toml b/pyproject.toml index 7f80284..7123f24 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -6,7 +6,7 @@ build-backend = "hatchling.build" name = "nullrun" # Full release history lives in CHANGELOG.md; only the current version # is pinned here. -version = "0.18.3" +version = "0.18.4" # Kept under the 200-char preview threshold so the full line is visible # without an "expand" click. The headline is the canonical §1 statement # from positioning.md — "runtime decision layer for tool-using AI agents" diff --git a/src/nullrun/__init__.py b/src/nullrun/__init__.py index 945e9ce..19fcc63 100644 --- a/src/nullrun/__init__.py +++ b/src/nullrun/__init__.py @@ -9,11 +9,17 @@ it with ``@protect``. Everything else exposed by ``nullrun`` is either runtime lifecycle -(``init``, ``shutdown``, ``on_error``, ``status``), the structured -exception hierarchy, or message/error-handling helpers -(``format_user_message``, ``handle``). -None of those are alternatives to ``@protect`` — they're setup -and cleanup. +(``init``, ``shutdown``, ``on_error``), the structured exception +hierarchy, or message/error-handling helpers (``format_user_message``, +``guard``). None of those are alternatives to ``@protect`` — they're +setup and cleanup. + +For introspection, reach the runtime snapshot via +``nullrun.get_runtime().status()`` (returns a frozen +:class:`NullRunStatus`). There is no top-level ``nullrun.status()`` +wrapper in 0.18.4 — the wrapper existed only to render an +``NR-C004`` config error before ``init``, which is now the runtime +class's job. Usage: import nullrun @@ -113,56 +119,6 @@ def shutdown(timeout: float = 2.0, flush: bool = True) -> None: _shutdown_atexit_registered = False -def status(): - """Return the current runtime state as a Layer-3 -:class:`NullRunStatus` snapshot. - - Synchronous, thread-safe, side-effect-free — safe to call - from the agent loop, the transport flush thread, or a - debug console. The returned dataclass is frozen so it can - be cached, shared, and compared with ``==``. - - Designed for the "the agent is stuck, what's wrong?" - runbook: - - >>> import nullrun - >>> print(nullrun.status.summary ) - NullRunStatus(degraded fallback=last_good@42s reason=last policy fetch failed at 2026-06-24T10:30:15+00:00) - - See ``nullrun.observability.status`` for the state - derivation rules (the four headline states: - ``ok`` / ``degraded`` / ``offline`` / ``misconfigured``). - - Raises: - NullRunConfigError: ``nullrun.init `` has not been - called yet, or the runtime was shut down. The - snapshot only makes sense when there is a runtime - to snapshot. - """ - # Read the module-level ``_runtime`` directly so we do NOT - # trigger ``get_instance ``'s lazy construction. ``status `` - # must NEVER create a runtime as a side effect — a fresh - # import of ``nullrun`` followed by ``nullrun.status `` - # should report "no runtime" cleanly, not try to spin one - # up (which would itself raise a different config error - # about missing api_key). - import nullrun.runtime as _rt_mod - from nullrun.breaker.exceptions import NullRunConfigError - - rt = _rt_mod._runtime - if rt is None: - raise NullRunConfigError( - "nullrun.status() requires a runtime. Call nullrun.init() first.", - error_code="NR-C004", - user_action=( - "Call nullrun.init(api_key='nr_live_...') before " - "calling nullrun.status(). The snapshot only makes " - "sense when there is a runtime to inspect." - ), - ) - return rt.status() - - def on_error(hook): """Register a global error hook. Layer 2 of the "give the user a chance" design. @@ -554,13 +510,18 @@ def my_agent: # NullRunError. WorkflowKilledInterrupt (BaseException) still # propagates — kill is never swallowed. # - # The module is named ``_handle.py`` (private, leading underscore) - # so it does not collide with the public ``nullrun.handle`` + # History: 0.18.4 renamed this from ``handle`` to ``guard``. + # The module file name is still ``_handle.py`` for the + # submodule-shadowing reason explained in that file's docstring; + # only the function name ``guard`` is public. + # + # Why the module is named ``_handle.py`` (private, leading underscore): + # so it does not collide with the public ``nullrun.guard`` # context manager. With a non-underscored name, pytest's test - # discovery would pre-import ``nullrun.handle`` as a submodule - # which shadows the lazy export and breaks ``from nullrun import - # handle``. - "handle": ("nullrun._handle", "handle"), + # discovery would pre-import ``nullrun.guard`` as a submodule + # which shadows the lazy export and breaks + # ``from nullrun import guard``. + "guard": ("nullrun._handle", "guard"), # ADR-009 P1 — governance audit surface (typed wire classes). # Users reach these as `from nullrun import AuditQuery` / # `from nullrun.audit import ...`. The runtime exposes @@ -626,9 +587,13 @@ def __dir__() -> list[str]: # single most important "give the user a chance" API — the # user has to know it exists to call it. "on_error", - # Layer 3: status introspection — synchronous snapshot of the - # runtime's state, returns a frozen NullRunStatus. - "status", + # Layer 3: status introspection is reached via + # ``nullrun.get_runtime().status()``. There is intentionally NO + # top-level ``nullrun.status()`` wrapper in 0.18.4 — the wrapper + # existed only to render NR-C004 before ``init``, which is now + # the runtime class's own job. Frozen ``NullRunStatus`` dataclass + # itself is importable as ``nullrun.NullRunStatus`` (PEP 562 + # lazy export from ``nullrun.observability.status``). # Layer 1: structured exception base + the most common subclasses # the user is expected to ``except`` on. Including them in # ``__all__`` means ``from nullrun import *`` and ``dir(nullrun)`` @@ -653,12 +618,13 @@ def __dir__() -> list[str]: # own wording per error_code without rewriting the SDK. "format_user_message", "set_user_message", - # Minimal-boilerplate error handling for scripts. ``handle`` is - # the context manager (``with nullrun.handle: ``). It translates + # Minimal-boilerplate error handling for scripts. ``guard`` is + # the context manager (``with nullrun.guard():``). It translates # any ``NullRunError`` into ``print(format_user_message(exc))`` + # ``sys.exit(1)``; ``WorkflowKilledInterrupt`` propagates. # CLI fail-fast on missing api_key is `init(fail_on_exit=True)`. - "handle", + # Renamed from ``handle`` in 0.18.4 — see the _LAZY_EXPORTS block. + "guard", ] # The SDK-side ``decision_history`` module was deleted. Decision diff --git a/src/nullrun/__version__.py b/src/nullrun/__version__.py index f844367..da25dda 100644 --- a/src/nullrun/__version__.py +++ b/src/nullrun/__version__.py @@ -5,5 +5,5 @@ string and the SDK_MIN_VERSION constant. """ -__version__ = "0.18.2" +__version__ = "0.18.4" __platform_version__ = "1.0.0" diff --git a/src/nullrun/_handle.py b/src/nullrun/_handle.py index d0a272a..1cde24e 100644 --- a/src/nullrun/_handle.py +++ b/src/nullrun/_handle.py @@ -10,7 +10,7 @@ For the common "I just want to run my agent and print a friendly message on failure" case, this module provides one one-liner: -*:func:`nullrun.handle` — context manager that translates any +*:func:`nullrun.guard` — context manager that translates any :class:`nullrun.NullRunError` into a structured developer-facing report (error code + what was attempted + where it came from + the underlying reason + how to fix it) and then exits ``1``. The @@ -21,7 +21,7 @@ :class:`nullrun.WorkflowKilledInterrupt` inherits from :class:`nullrun.NullRunError` (see the class docstring), so a bare ``except NullRunError`` would otherwise swallow the kill signal. -``handle`` explicitly re-raises it — the kill is a +``guard`` explicitly re-raises it — the kill is a control-plane action, not an SDK failure, and must reach the top of the agent loop. Non-NullRun exceptions also propagate unchanged. @@ -31,6 +31,14 @@ wrapper — the four-line developer report is rendered identically and the process exits ``1`` on missing ``NULLRUN_API_KEY``. +History +------- +In 0.18.4 this context manager was renamed from ``handle`` to +``guard``. The previous ``@guarded`` decorator was already removed +in 0.18.2 (f1721f2), freeing the ``guard`` name; ``guard`` reads as +a single verb consistent with ``init`` / ``shutdown`` / ``on_error``. +No deprecation alias — ``nullrun.handle`` simply no longer exists. + Why a separate module --------------------- The exception hierarchy in:mod:`nullrun.breaker.exceptions` is the @@ -42,15 +50,19 @@ Why ``_handle.py`` (leading underscore) --------------------------------------- -The public symbol exported from this module is:func:`handle` (a +The public symbol exported from this module is:func:`guard` (a context manager). With a non-underscored module name ``nullrun/handle.py``, Python's import machinery pre-binds ``nullrun.handle`` to the submodule when anything does ``import nullrun.handle`` (for example, pytest's test discovery). -That binding shadows the lazy export ``"handle": (...)`` in -:mod:`nullrun`, so ``from nullrun import handle`` returns the +That binding shadows the lazy export ``"guard": (...)`` in +:mod:`nullrun`, so ``from nullrun import guard`` returns the module object instead of the function. The leading underscore makes the module private so it does not collide. + +The module file name keeps its historical ``_handle.py`` shape +because renaming it to ``_guard.py`` is not part of the public +contract — only the function name ``guard`` is observable. """ from __future__ import annotations @@ -157,7 +169,7 @@ def _render_dev_error_report( @contextmanager -def handle(*, exit_code: int = 1): +def guard(*, exit_code: int = 1): """Catch ``NullRunError`` and translate it to a developer-facing exit. Inside the ``with`` block, any:class:`nullrun.NullRunError` is @@ -179,7 +191,7 @@ def handle(*, exit_code: int = 1): Re-raised explicitly inside the ``except NullRunError`` branch because ``WorkflowKilledInterrupt`` sits on the ``NullRunError`` MRO (Sentry/OTel ``except Exception`` handlers should record kill - events; this ``handle`` wrapper opts OUT of that + events; this ``guard`` wrapper opts OUT of that recording on purpose). *:class:`KeyboardInterrupt` /:class:`SystemExit` (``BaseException``) -- same reason as the kill signal -- never reach the @@ -197,7 +209,7 @@ def handle(*, exit_code: int = 1): nullrun.init(api_key="nr_live_...") - with nullrun.handle: + with nullrun.guard(): run_my_agent("hello") # ↑ if run_my_agent raised NullRunError, a structured # developer report is printed and the script exits 1. @@ -205,8 +217,9 @@ def handle(*, exit_code: int = 1): try: yield except NullRunError as exc: - # the NullRunError MRO so Sentry/OTel `except Exception` - # handlers record kill events. ``handle`` is the + # Re-raise WorkflowKilledInterrupt explicitly: it shares the + # NullRunError MRO so Sentry/OTel `except Exception` handlers + # would otherwise record kill events. ``guard`` is the # friendly-exit pattern, NOT the user-callback pattern -- kill # is a control-plane action and must propagate so the agent # loop / dashboard resume path can see it. Re-raise explicitly @@ -226,4 +239,4 @@ def handle(*, exit_code: int = 1): sys.exit(exit_code) -__all__ = ["handle"] \ No newline at end of file +__all__ = ["guard"] \ No newline at end of file diff --git a/tests/test_dev_error_report.py b/tests/test_dev_error_report.py index 74b0584..45a7b69 100644 --- a/tests/test_dev_error_report.py +++ b/tests/test_dev_error_report.py @@ -1,5 +1,5 @@ """ -Tests for the developer-facing error report rendered by ``handle`` +Tests for the developer-facing error report rendered by ``guard`` and the CLI ``init(fail_on_exit=True)`` path. Pre-fix (2026-09-22), the catch-all exit path printed only the catalog @@ -17,13 +17,15 @@ These tests pin the four-line invariant so a future "let's tidy up the error path" cannot silently drop the structured detail back to a single sentence. + +History: 0.18.4 renamed ``handle`` to ``guard``. """ from __future__ import annotations import pytest import nullrun -from nullrun import handle +from nullrun import guard from nullrun._handle import _render_dev_error_report from nullrun.breaker.exceptions import ( NullRunAuthenticationError, @@ -123,11 +125,11 @@ def test_report_includes_docs_url(): assert "https://docs.nullrun.io" in report -# --- handle() integration tests --------------------------------------------- +# --- guard() integration tests --------------------------------------------- -def test_handle_prints_full_dev_report(monkeypatch, capsys): - """``with handle():`` exits 1 AND writes the four-line dev report +def test_guard_prints_full_dev_report(monkeypatch, capsys): + """``with guard():`` exits 1 AND writes the four-line dev report to stderr -- not just the catalog headline.""" exits = [] @@ -138,7 +140,7 @@ def fake_exit(code): monkeypatch.setattr("sys.exit", fake_exit) with pytest.raises(SystemExit): - with handle(): + with guard(): raise NullRunAuthenticationError( "Auth failed with status 401.", error_code="NR-A003", @@ -156,9 +158,9 @@ def fake_exit(code): assert "Rotate the API key." in err -def test_handle_falls_back_to_legacy_on_helper_bug(monkeypatch, capsys): +def test_guard_falls_back_to_legacy_on_helper_bug(monkeypatch, capsys): """Defensive: if the report builder itself raises (a future bug), - ``handle()`` must still exit cleanly with the catalog headline. + ``guard()`` must still exit cleanly with the catalog headline. The defensive fallback path is critical -- a buggy helper cannot freeze a script that would otherwise exit.""" from nullrun import _handle as handle_mod @@ -177,7 +179,7 @@ def fake_exit(code): monkeypatch.setattr("sys.exit", fake_exit) with pytest.raises(SystemExit): - with handle(): + with guard(): raise NullRunError("oops", error_code="NR-B002") assert exits == [1] @@ -187,7 +189,7 @@ def fake_exit(code): assert "temporarily unavailable" in err.lower() -def test_handle_report_uses_class_name_for_unknown_endpoint(monkeypatch, capsys): +def test_guard_report_uses_class_name_for_unknown_endpoint(monkeypatch, capsys): """When the exception has no ``endpoint`` attribute, the where line uses ``endpoint=N/A (config-time failure)`` so the developer can immediately tell that the failure happened at startup, not on @@ -201,7 +203,7 @@ def fake_exit(code): monkeypatch.setattr("sys.exit", fake_exit) with pytest.raises(SystemExit): - with handle(): + with guard(): raise NullRunError( "config-time failure", error_code="NR-C001", @@ -216,8 +218,8 @@ def fake_exit(code): # --- sanity: still callable without runtime -------------------------------- -def test_handle_does_not_require_runtime(): - """``handle`` must work without ``nullrun.init()``. Sanity check +def test_guard_does_not_require_runtime(): + """``guard`` must work without ``nullrun.init()``. Sanity check that the helper module is importable on its own.""" - assert callable(handle) - assert callable(nullrun.handle) + assert callable(guard) + assert callable(nullrun.guard) diff --git a/tests/test_handle.py b/tests/test_guard.py similarity index 62% rename from tests/test_handle.py rename to tests/test_guard.py index fdc33da..85a7df5 100644 --- a/tests/test_handle.py +++ b/tests/test_guard.py @@ -1,26 +1,29 @@ -"""Tests for the ``nullrun.handle`` context manager. +"""Tests for the ``nullrun.guard`` context manager. Contract: -* ``with handle():`` translates any :class:`nullrun.NullRunError` +* ``with guard():`` translates any :class:`nullrun.NullRunError` into ``print(format_user_message(exc), file=sys.stderr)`` and then ``sys.exit(1)``. * :class:`nullrun.WorkflowKilledInterrupt` propagates unchanged — kill must not be swallowed into a graceful exit. (``WorkflowKilledInterrupt`` is now an ``Exception`` subclass via - ``NullRunError``, but ``handle`` explicitly re-raises it so the + ``NullRunError``, but ``guard`` explicitly re-raises it so the kill signal still reaches the top of the agent loop.) * Non-NullRun exceptions also propagate unchanged so the user's own bugs surface as honest tracebacks. -* No runtime is required — ``handle`` works without +* No runtime is required — ``guard`` works without ``nullrun.init()``. + +History: 0.18.4 renamed ``nullrun.handle`` to ``nullrun.guard``. +Same body, same semantics, shorter verb. """ from __future__ import annotations import pytest import nullrun -from nullrun import handle +from nullrun import guard from nullrun.breaker.exceptions import ( NullRunBudgetError, NullRunError, @@ -28,8 +31,8 @@ ) -def test_handle_catches_nullrun_error_and_exits(monkeypatch, capsys): - """``with handle():`` exits 1 and prints the catalog user-message.""" +def test_guard_catches_nullrun_error_and_exits(monkeypatch, capsys): + """``with guard():`` exits 1 and prints the catalog user-message.""" exits = [] def fake_exit(code): @@ -39,7 +42,7 @@ def fake_exit(code): monkeypatch.setattr("sys.exit", fake_exit) with pytest.raises(SystemExit): - with handle(): + with guard(): # NullRunBudgetError inherits from NullRunBlockedException # whose __init__ takes (workflow_id, reason,...). raise NullRunBudgetError("wf-1", "workflow budget exhausted") @@ -49,36 +52,36 @@ def fake_exit(code): assert exits == [1] -def test_handle_propagates_workflow_killed(monkeypatch): +def test_guard_propagates_workflow_killed(monkeypatch): """``WorkflowKilledInterrupt`` must NOT be swallowed into sys.exit.""" monkeypatch.setattr("sys.exit", lambda c: pytest.fail("sys.exit was called")) with pytest.raises(WorkflowKilledInterrupt): - with handle(): + with guard(): raise WorkflowKilledInterrupt("wf-1", "killed via dashboard") -def test_handle_propagates_value_error(monkeypatch): +def test_guard_propagates_value_error(monkeypatch): """Non-NullRun exceptions pass through for an honest traceback.""" monkeypatch.setattr("sys.exit", lambda c: pytest.fail("sys.exit was called")) with pytest.raises(ValueError): - with handle(): + with guard(): raise ValueError("user bug, not an SDK failure") -def test_handle_returns_on_success(monkeypatch): +def test_guard_returns_on_success(monkeypatch): """A clean ``with`` block returns the wrapped expression's value.""" monkeypatch.setattr("sys.exit", lambda c: pytest.fail("sys.exit was called")) - with handle(): + with guard(): result = 1 + 2 assert result == 3 -def test_handle_exit_code_kwarg(monkeypatch, capsys): - """``handle(exit_code=42)`` honours the override.""" +def test_guard_exit_code_kwarg(monkeypatch, capsys): + """``guard(exit_code=42)`` honours the override.""" exits = [] def fake_exit(code): @@ -88,15 +91,25 @@ def fake_exit(code): monkeypatch.setattr("sys.exit", fake_exit) with pytest.raises(SystemExit): - with handle(exit_code=42): + with guard(exit_code=42): raise NullRunError("oops", error_code="NR-B002") assert exits == [42] def test_no_init_required(): - """``handle`` must not depend on a runtime.""" - # If handle pulled in the runtime, importing this module would have + """``guard`` must not depend on a runtime.""" + # If guard pulled in the runtime, importing this module would have # raised during the prior tests. Smoke-test the import path here. - assert callable(handle) - assert callable(nullrun.handle) + assert callable(guard) + assert callable(nullrun.guard) + + +def test_handle_removed_from_public_surface(): + """``nullrun.handle`` was fully removed in 0.18.4 (renamed to + ``guard``). This is a regression guard against a future re-add of + the old name as a deprecation alias.""" + import nullrun as n + + assert not hasattr(n, "handle") + assert "handle" not in n.__all__ diff --git a/tests/test_status.py b/tests/test_status.py index afa19df..6a3c27a 100644 --- a/tests/test_status.py +++ b/tests/test_status.py @@ -1,8 +1,7 @@ -"""Tests for the Layer 3 ``nullrun.status `` introspection API. +"""Tests for the Layer 3 ``runtime.status()`` introspection API. The contract: - * No runtime → ``NullRunConfigError`` with ``NR-C004``. * Runtime present → frozen ``NullRunStatus`` snapshot with: - ``state`` ∈ ``{"ok", "degraded", "offline", "misconfigured"}`` - ``recent_errors`` is a list (possibly empty) of @@ -13,23 +12,25 @@ must NEVER mutate the runtime or create a new one. * Equality works on the frozen dataclass (``s1 == s2`` when every field is equal) — important for caching / diffing. + +History +------- +In 0.18.4 the top-level ``nullrun.status()`` wrapper was removed. +Tests now drive the runtime directly (``rt.status()``) rather than +going through the deleted wrapper. The wrapper existed only to +render an ``NR-C004`` config error before ``init``; that path is +covered by runtime's own self-consistency checks below. """ from datetime import datetime, timezone -from typing import Any -from unittest.mock import patch import pytest import nullrun -from nullrun.breaker.exceptions import ( - NullRunConfigError, - NullRunError, -) +from nullrun.breaker.exceptions import NullRunError from nullrun.observability.status import ( NullRunStatus, RecentError, - WorkflowState, _RecentErrorRing, ) from nullrun.runtime import NullRunRuntime @@ -51,7 +52,7 @@ def _reset_runtime(): def _make_runtime(api_key: str = "nr_live_test_key_1234") -> NullRunRuntime: """Construct a NullRunRuntime in _test_mode without going - through ``init `` (which would try to call the backend). + through ``init`` (which would try to call the backend). """ rt = NullRunRuntime(api_key=api_key, _test_mode=True) import nullrun.runtime as _rt_mod @@ -62,73 +63,49 @@ def _make_runtime(api_key: str = "nr_live_test_key_1234") -> NullRunRuntime: # --------------------------------------------------------------------------- -# 1. No runtime -# --------------------------------------------------------------------------- -class TestNoRuntime: - def test_status_raises_when_no_runtime(self): - with pytest.raises(NullRunConfigError) as info: - nullrun.status() - err = info.value - assert err.error_code == "NR-C004" - assert "init" in err.user_action.lower() - assert err.retryable is False - - def test_status_never_lazily_creates_runtime(self): - # Sanity: calling status must NOT trigger - # NullRunRuntime.get_instance (which would itself - # raise a different config error about missing - # api_key). The whole point of NR-C004 is a clean - # "no runtime" signal. - with patch("nullrun.runtime.NullRunRuntime.get_instance") as mock_get: - with pytest.raises(NullRunConfigError): - nullrun.status() - mock_get.assert_not_called() - - -# --------------------------------------------------------------------------- -# 2. With runtime — snapshot fields +# 1. With runtime — snapshot fields # --------------------------------------------------------------------------- class TestSnapshotFields: def test_minimal_runtime_yields_ok_state(self): - _make_runtime() - s = nullrun.status() + rt = _make_runtime() + s = rt.status() assert s.state == "ok" assert s.api_key_prefix == "nr_live_te" assert s.is_healthy() is True def test_snapshot_is_frozen(self): - _make_runtime() - s = nullrun.status() + rt = _make_runtime() + s = rt.status() with pytest.raises(Exception): # FrozenInstanceError s.state = "degraded" # type: ignore[misc] def test_snapshot_supports_equality(self): - _make_runtime() - s1 = nullrun.status() - s2 = nullrun.status() + rt = _make_runtime() + s1 = rt.status() + s2 = rt.status() assert s1 == s2 def test_api_key_prefix_truncated_to_10_chars(self): _make_runtime(api_key="nr_live_SsBF9OMYcVCgRCNcCVcJ4khTOPKx79JG") - s = nullrun.status() + s = _make_runtime( + api_key="nr_live_SsBF9OMYcVCgRCNcCVcJ4khTOPKx79JG" + ).status() assert s.api_key_prefix == "nr_live_Ss" assert len(s.api_key_prefix) == 10 # Full key MUST NOT leak into the snapshot. assert "TOPKx79JG" not in str(s) def test_backend_reachable_none_when_no_attempt(self): - _make_runtime() - s = nullrun.status() + s = _make_runtime().status() assert s.backend_reachable is None def test_ws_connected_none_when_no_ws_started(self): - _make_runtime() - s = nullrun.status() + s = _make_runtime().status() assert s.ws_connected is None # --------------------------------------------------------------------------- -# 3. State derivation +# 2. State derivation # --------------------------------------------------------------------------- class TestStateDerivation: def test_misconfigured_when_no_api_key(self): @@ -138,26 +115,23 @@ def test_misconfigured_when_no_api_key(self): # misconfigured branch. rt = _make_runtime() rt.api_key = None - s = nullrun.status() + s = rt.status() assert s.state == "misconfigured" assert s.api_key_valid is None assert s.api_key_prefix is None # --------------------------------------------------------------------------- -# 4. Recent-errors ring buffer +# 3. Recent-errors ring buffer # --------------------------------------------------------------------------- class TestRecentErrors: def test_recent_errors_empty_on_fresh_runtime(self): - _make_runtime() - s = nullrun.status() + s = _make_runtime().status() assert s.recent_errors == [] def test_recent_errors_populated_by_emit(self): rt = _make_runtime() # Simulate an error firing through the Layer-2 path. - from nullrun.observability.error_hooks import ErrorContext - err = NullRunError("boom", error_code="NR-X999") rt._emit_sdk_error( err, @@ -165,7 +139,7 @@ def test_recent_errors_populated_by_emit(self): workflow_id="wf-1", tool_name="send_email", ) - s = nullrun.status() + s = rt.status() assert len(s.recent_errors) == 1 entry = s.recent_errors[0] assert entry.error_code == "NR-X999" @@ -200,24 +174,21 @@ def test_recent_errors_pushed_even_with_no_hook(self): # buffer fires even when no on_error hook is # registered. This is the whole point of Layer 3. rt = _make_runtime() - from nullrun.observability.error_hooks import ErrorContext - rt._emit_sdk_error( NullRunError("test"), stage="init", ) # No on_error hook registered. snapshot still works. - s = nullrun.status() + s = rt.status() assert len(s.recent_errors) == 1 # --------------------------------------------------------------------------- -# 5. Workflow state from cache +# 4. Workflow state from cache # --------------------------------------------------------------------------- class TestWorkflowState: def test_workflow_state_none_when_no_remote_state(self): - _make_runtime() - s = nullrun.status() + s = _make_runtime().status() assert s.workflow_state is None def test_workflow_state_reads_from_cache(self): @@ -230,7 +201,7 @@ def test_workflow_state_reads_from_cache(self): "wf-test-1", {"state": "Killed", "version": 5, "reason": "manual kill"}, ) - s = nullrun.status() + s = rt.status() assert s.workflow_state is not None assert s.workflow_state.workflow_id == "wf-test-1" assert s.workflow_state.state == "Killed" @@ -238,13 +209,11 @@ def test_workflow_state_reads_from_cache(self): # --------------------------------------------------------------------------- -# 6. summary — human-readable one-liner +# 5. summary — human-readable one-liner # --------------------------------------------------------------------------- class TestSummary: def test_ok_summary(self): - _make_runtime() - s = nullrun.status() - out = s.summary() + out = _make_runtime().status().summary() assert "ok" in out assert "nr_live_te" in out @@ -254,8 +223,7 @@ def test_summary_with_organization_and_workflow(self): rt = _make_runtime() rt.organization_id = "org_abcdef1234567890" rt.workflow_id = "wf_xyzzy1234567890" - s = nullrun.status() - out = s.summary() + out = rt.status().summary() assert "org=org_abcd" in out assert "wf=wf_xyzzy" in out @@ -267,8 +235,7 @@ def test_summary_includes_workflow_state_when_not_normal(self): "wf-test-1", {"state": "Killed", "version": 5, "reason": "manual kill"}, ) - s = nullrun.status() - out = s.summary() + out = rt.status().summary() assert "wf_state=Killed" in out def test_summary_omits_normal_workflow_state(self): @@ -279,13 +246,12 @@ def test_summary_omits_normal_workflow_state(self): "wf-test-1", {"state": "Normal", "version": 1, "reason": None}, ) - s = nullrun.status() - out = s.summary() + out = rt.status().summary() assert "wf_state=" not in out def test_summary_includes_backend_unreachable(self): # Branch: ``self.backend_reachable is False``. - # ``backend_reachable`` is a local in ``status ``, not a stored + # ``backend_reachable`` is a local in ``status``, not a stored # attribute on the runtime — construct the snapshot directly. s = NullRunStatus( state="degraded", @@ -320,45 +286,36 @@ def test_summary_includes_ws_disconnected(self): def test_summary_includes_recent_errors_count(self): # Branch: ``if self.recent_errors``. rt = _make_runtime() - from nullrun.breaker.exceptions import NullRunError - from nullrun.observability.error_hooks import ErrorContext - for i in range(3): rt._emit_sdk_error( NullRunError(f"err-{i}", error_code="NR-X000"), stage="init", ) - s = nullrun.status() - out = s.summary() + out = rt.status().summary() assert "errors=3" in out # --------------------------------------------------------------------------- -# 7. Public API surface +# 6. Public API surface regression guards (0.18.4) # --------------------------------------------------------------------------- -class TestPublicAPI: - def test_status_in_dir(self): - assert callable(nullrun.status) - assert "status" in dir(nullrun) - - def test_status_in_all(self): - import nullrun as n +class TestStatusRemovedFromTopLevel: + """0.18.4 removed the top-level ``nullrun.status()`` wrapper. + + These tests pin that removal against a future re-add. The + runtime method ``NullRunRuntime.status()`` is the only + public status entry point now — callers reach it via + ``nullrun.get_runtime().status()`` or by holding a runtime + reference they constructed themselves. + """ - assert "status" in n.__all__ + def test_status_not_in_dir(self): + # ``status`` is no longer a curated surface entry — the + # runtime method is reached via ``nullrun.get_runtime()`` + # rather than via a top-level wrapper. + assert "status" not in dir(nullrun) + assert not callable(getattr(nullrun, "status", None)) - def test_status_dataclasses_importable(self): - # All four dataclasses reachable from the public - # namespace for type annotations. - from nullrun.observability import ( - NullRunStatus as NS, - ) - from nullrun.observability import ( - RecentError as RE, - ) - from nullrun.observability import ( - WorkflowState as WS, - ) + def test_status_not_in_all(self): + import nullrun as n - assert NS is NullRunStatus - assert RE is RecentError - assert WS is WorkflowState + assert "status" not in n.__all__ From bf305668aef2840a28f184457e184cef26a59c99 Mon Sep 17 00:00:00 2001 From: Anatoly Maltsev Date: Fri, 25 Sep 2026 16:22:46 +0400 Subject: [PATCH 08/10] fix(sdk): follow-up 0.18.4 rename sweep in docs/scripts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Sweep for any remaining stale references to the renamed nullrun.handle / nullrun.status() symbols after the 0.18.4 cleanup (83421e0): * scripts/smoke_prod.py — drop status import (F401), switch the call site to nullrun.get_runtime().status(). * src/nullrun/__init__.py — comment about the CLI fail-fast dev-error report: handle() → guard(). * src/nullrun/messages.py — NR-W004 docstring: nullrun.handle → nullrun.guard. * tests/test_init_fail_on_exit.py — module docstring: handle() → guard(). * tests/test_messages.py — NR-W004 test docstring: handle() → guard(). * tests/test_typed_exceptions_full_audit.py — disambiguate the comment from nullrun.handle() to ActionHandler.handle() so readers do not confuse the action-handler method (still present) with the renamed context manager (removed). Verified post-edit: * pytest: 1401 passed, 1 skipped (no regressions) * ruff on all touched files: clean (smoke_prod.py went from 3 pre-existing ruff errors to 2; the F401 on the now-removed status import is gone; the remaining F401 on_error and F841 decision_payload are pre-existing on master). * Public surface: handle=False, status=False, guard=True, __all__ length 20 (matches 0.18.4 design). * guard() works as a context manager and re-raises WorkflowKilledInterrupt (BaseException propagates). * Prod smoke against https://api.nullrun.io: @protect round-trip returns 'echo(hello-0.18.4)'; guard() renders the four-line developer report and exits 1 on a NullRunError. Co-Authored-By: Claude Code --- scripts/smoke_prod.py | 10 ++++++---- src/nullrun/__init__.py | 2 +- src/nullrun/messages.py | 2 +- tests/test_init_fail_on_exit.py | 2 +- tests/test_messages.py | 2 +- tests/test_typed_exceptions_full_audit.py | 2 +- 6 files changed, 11 insertions(+), 9 deletions(-) diff --git a/scripts/smoke_prod.py b/scripts/smoke_prod.py index 4d51fbc..44f718c 100644 --- a/scripts/smoke_prod.py +++ b/scripts/smoke_prod.py @@ -50,7 +50,7 @@ def main() -> None: # Late imports so env vars are read by the SDK first. import nullrun - from nullrun import init, protect, shutdown, on_error, status + from nullrun import init, on_error, protect, shutdown from nullrun.breaker.exceptions import NullRunBlockedException # 1. Surface check — only the curated entry points are exposed. @@ -71,11 +71,13 @@ def main() -> None: _fail(f"init() raised: {exc!r}\n{traceback.format_exc()}") _ok(f"init() ok against {api_url}") - # 3. status() after init. + # 3. status() after init. 0.18.4 dropped the top-level + # ``nullrun.status()`` wrapper; reach the snapshot via + # ``nullrun.get_runtime().status()`` instead. try: - st = status() + st = nullrun.get_runtime().status() except Exception as exc: # noqa: BLE001 - _fail(f"status() raised: {exc!r}") + _fail(f"nullrun.get_runtime().status() raised: {exc!r}") # `status()` returns a typed NullRunStatus object; coerce via # ``vars()`` so we can introspect fields without depending on # the SDK's internal type name. diff --git a/src/nullrun/__init__.py b/src/nullrun/__init__.py index 19fcc63..27c0bfb 100644 --- a/src/nullrun/__init__.py +++ b/src/nullrun/__init__.py @@ -269,7 +269,7 @@ def my_agent: ), ) # CLI path: render the same four-line developer report - # ``handle()`` uses, then sys.exit(1). Library embedders leave + # ``guard()`` uses, then sys.exit(1). Library embedders leave # ``fail_on_exit=False`` (default) so the exception propagates # into their own try/except and the host process stays alive # (FastAPI startup, Jupyter, REPL). diff --git a/src/nullrun/messages.py b/src/nullrun/messages.py index 2ac0c12..ac0cf99 100644 --- a/src/nullrun/messages.py +++ b/src/nullrun/messages.py @@ -132,7 +132,7 @@ "NR-A015": "Your request couldn't be completed because the approval has already been used. Please start a new request.", # ---- Workflow lifecycle (server-side state) ------------------------------ # NR-W004: workflow soft-deleted or killed on the server. Distinct - # from NR-W002 (BaseException path that bypasses ``nullrun.handle``) + # from NR-W002 (BaseException path that bypasses ``nullrun.guard``) # and NR-W003 (pause / cooldown). End users see this only after an # operator terminated their session from the dashboard; the wording # is similar to NR-W002 because the user-visible outcome is the diff --git a/tests/test_init_fail_on_exit.py b/tests/test_init_fail_on_exit.py index 3c39313..9afbf08 100644 --- a/tests/test_init_fail_on_exit.py +++ b/tests/test_init_fail_on_exit.py @@ -4,7 +4,7 @@ * On success, returns whatever the runtime ``init()`` returns. * On ``NullRunAuthenticationError`` (e.g. NR-C001 missing api_key), - prints the same four-line developer report that ``handle()`` uses + prints the same four-line developer report that ``guard()`` uses and ``sys.exit(1)``. * The default ``init()`` (``fail_on_exit=False``) still raises — the library-friendly path. diff --git a/tests/test_messages.py b/tests/test_messages.py index 40add7d..a214b32 100644 --- a/tests/test_messages.py +++ b/tests/test_messages.py @@ -312,7 +312,7 @@ def test_format_user_message_handles_approval_replay_rejected(): def test_format_user_message_handles_workflow_inactive(): """NR-W004: workflow soft-deleted / killed on the server. Distinct - from NR-W002 (BaseException kill path that bypasses handle()) and + from NR-W002 (BaseException kill path that bypasses guard()) and NR-W003 (pause / cooldown). End-user copy is similar to NR-W002 because the user-visible outcome is the same.""" inactive = exc.NullRunWorkflowInactiveError( diff --git a/tests/test_typed_exceptions_full_audit.py b/tests/test_typed_exceptions_full_audit.py index 35f9239..bbe4e1b 100644 --- a/tests/test_typed_exceptions_full_audit.py +++ b/tests/test_typed_exceptions_full_audit.py @@ -182,7 +182,7 @@ def test_auto_instrumentation_kill_raises_typed(self): class TestHandleKillActionRaisesTyped: - """actions.py:249 — nullrun.handle() KILL action → typed.""" + """actions.py:249 — ActionHandler.handle() KILL action → typed.""" def test_handle_kill_action_raises_typed(self): exc = NullRunWorkflowKilledError( From d0d0e720c84282f6f8f25a3bd78bf184b5f8f89f Mon Sep 17 00:00:00 2001 From: Anatoly Maltsev Date: Fri, 25 Sep 2026 16:44:02 +0400 Subject: [PATCH 09/10] docs(sdk): drop stale init_or_die references in source docstrings MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up to ccb2670 (0.18.3: init_or_die removal) and the 0.18.4 rename sweep. Two stale source-code comments still referenced init_or_die by name even though the function was removed: * src/nullrun/_handle.py: the docstring at the head of the module explained how guard differs from the old init_or_die wrapper. Rewrite to point at init(fail_on_exit=True) only — the historical aside belongs in CHANGELOG.md, not next to guard's own contract. * src/nullrun/decorators.py:339: the lazy auto-instrument comment described a user who writes '@nullrun.protect without calling init_or_die() first' would still get a runtime. The actual surface is init() — the comment now reads correctly. Functional code untouched. All 1401 tests pass on Python 3.11. ruff check is clean on the touched hunks; pre-existing F401 in decorators.py:43 (warnings import) and pre-existing ruff-format whitespace around lines I did not touch are unrelated to this change and remain for a separate follow-up. --- src/nullrun/_handle.py | 7 ++++--- src/nullrun/decorators.py | 2 +- 2 files changed, 5 insertions(+), 4 deletions(-) diff --git a/src/nullrun/_handle.py b/src/nullrun/_handle.py index 1cde24e..9b08845 100644 --- a/src/nullrun/_handle.py +++ b/src/nullrun/_handle.py @@ -27,9 +27,10 @@ unchanged. CLI scripts that want the same fail-fast behavior at startup should -call ``nullrun.init(fail_on_exit=True)`` instead of an ``init_or_die`` -wrapper — the four-line developer report is rendered identically -and the process exits ``1`` on missing ``NULLRUN_API_KEY``. +call ``nullrun.init(fail_on_exit=True)`` — the four-line developer +report is rendered identically and the process exits ``1`` on missing +``NULLRUN_API_KEY``. (The standalone ``init_or_die`` wrapper was +removed in 0.18.3; fail-fast is now a flag on ``init``.) History ------- diff --git a/src/nullrun/decorators.py b/src/nullrun/decorators.py index 6ebeffe..61b5a22 100644 --- a/src/nullrun/decorators.py +++ b/src/nullrun/decorators.py @@ -336,7 +336,7 @@ def _get_or_create_runtime() -> NullRunRuntime: # The user-facing API is `nullrun.init()` which calls # `auto_instrument(runtime)` directly (see `nullrun/__init__.py::init`). # However, a user who writes only -# ``@nullrun.protect`` without calling ``init_or_die()`` first would +# ``@nullrun.protect`` without calling ``init()`` first would # still create a runtime via ``NullRunRuntime.get_instance()`` — but # no vendor SDK patches would be installed, so token capture would be # silently absent. From 9835f023f643f4deaea00e9534256b970d1520c7 Mon Sep 17 00:00:00 2001 From: Anatoly Maltsev Date: Sat, 26 Sep 2026 09:25:30 +0400 Subject: [PATCH 10/10] chore(sdk): remove unused imports + fix mypy errors leftover from deprecation sweep MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Mechanical cleanup so ruff check src tests + mypy src/nullrun are clean on the cleanup/deprecation-removal branch before merging to master. - 71 F401 (unused imports) auto-fixed by 'ruff check --fix'. - 9 F841 (unused local variables) hand-removed across tests; all were dead code left behind after deleted imports and refactored helpers (initial_buffer_len in test_transport, finalize_objs and t_id in test_signal_safety, first_orig in test_autogen_patch, existing_fp in test_dedup, req in test_model_fallback, reg in test_registry, sid in test_actions, timestamp in test_ws_signed_payload). - src/nullrun/business_impact.py: narrow BusinessImpact.impact from Any to NoImpactPayload — the inline comment already documented the invariant ('Only NoImpactPayload in 0.18.2') and the only public ctor (BusinessImpact.no_impact) wires a NoImpactPayload, so the broader Any was masking a real no-any-return at the to_wire_dict boundary. - src/nullrun/__init__.py: tighten _LAZY_EXPORTS annotation from tuple[str, str | None] to tuple[str, str] — every value in the table is a 2-tuple of module path + attribute name, so the optional was wrong and the downstream list[str] for __import__()'s fromlist tripped list-item mypy. Verification: ruff check src tests : All checks passed mypy src/nullrun : Success: no issues found in 36 source files pytest -q : 1401 passed, 1 skipped --- src/nullrun/__init__.py | 2 +- src/nullrun/business_impact.py | 2 +- src/nullrun/decorators.py | 1 - src/nullrun/runtime.py | 12 ++---------- tests/contract/test_llm_call_model_wire.py | 2 -- tests/test_2026_09_10_catchfanin_passthrough.py | 1 - tests/test_2026_09_10_check_failopen.py | 1 - .../test_2026_09_10_runtime_block_typed_dispatch.py | 2 -- tests/test_2026_09_11_auth_heartbeat_sweep.py | 1 - tests/test_actions.py | 5 ++--- tests/test_agent_id_uuid.py | 2 -- tests/test_approval_timeout_field.py | 4 +--- tests/test_audit.py | 2 -- tests/test_autogen_patch.py | 1 - tests/test_crewai_patch.py | 1 - tests/test_dedup.py | 1 - tests/test_error_hooks.py | 9 --------- tests/test_exception_hierarchy.py | 1 - tests/test_gate_real_path.py | 1 - tests/test_integration_contract.py | 2 -- tests/test_integrations_fastapi.py | 3 +-- tests/test_langgraph_callback.py | 2 -- tests/test_llm_call_metadata_flags.py | 1 - tests/test_lru_active_runs.py | 2 +- tests/test_mcp_adapter.py | 1 - tests/test_model_fallback.py | 1 - tests/test_redact.py | 1 - tests/test_registry.py | 6 +----- tests/test_runtime.py | 3 +-- tests/test_runtime_branches.py | 4 +--- tests/test_signal_safety.py | 10 ---------- tests/test_transport.py | 2 -- tests/test_transport_branches.py | 1 - tests/test_unified_fingerprint.py | 1 - tests/test_v3_wire_contract.py | 3 --- tests/test_webhook_backoff.py | 1 - tests/test_ws_signed_payload.py | 4 ---- 37 files changed, 12 insertions(+), 87 deletions(-) diff --git a/src/nullrun/__init__.py b/src/nullrun/__init__.py index 27c0bfb..a77cbfe 100644 --- a/src/nullrun/__init__.py +++ b/src/nullrun/__init__.py @@ -409,7 +409,7 @@ def my_agent: # in `globals ` so subsequent lookups are O(1) and not visible in # `vars(nullrun)` until then. This is the same pattern used by pandas / # sqlalchemy / etc. to keep the top-level namespace discoverable. -_LAZY_EXPORTS: dict[str, tuple[str, str | None]] = { +_LAZY_EXPORTS: dict[str, tuple[str, str]] = { # Runtime + context (advanced) "NullRunRuntime": ("nullrun.runtime", "NullRunRuntime"), "get_runtime": ("nullrun.runtime", "get_runtime"), diff --git a/src/nullrun/business_impact.py b/src/nullrun/business_impact.py index 02664b3..5e7205a 100644 --- a/src/nullrun/business_impact.py +++ b/src/nullrun/business_impact.py @@ -62,7 +62,7 @@ class BusinessImpact: rule's ``param_name``. """ - impact: Any # Only NoImpactPayload in 0.18.2. + impact: NoImpactPayload # Only NoImpactPayload in 0.18.2. @property def kind(self) -> str: diff --git a/src/nullrun/decorators.py b/src/nullrun/decorators.py index 61b5a22..0cffee6 100644 --- a/src/nullrun/decorators.py +++ b/src/nullrun/decorators.py @@ -40,7 +40,6 @@ def researcher(q): import logging import os import threading -import warnings from collections.abc import Callable from contextvars import Token from typing import Any, TypeVar diff --git a/src/nullrun/runtime.py b/src/nullrun/runtime.py index a8dfc62..42773db 100644 --- a/src/nullrun/runtime.py +++ b/src/nullrun/runtime.py @@ -75,23 +75,20 @@ import time import uuid from collections.abc import Callable -from typing import Any, Optional +from typing import Any import httpx from nullrun._registry import get_active_runtime -from nullrun.actions import ActionHandler, ActionType +from nullrun.actions import ActionHandler from nullrun.audit import ( # ADR-009 P1 — governance audit surface - AuditEntry, AuditExportJob, AuditExportStatus, - AuditLogMeta, AuditLogPage, AuditQuery, AuditVerifyResult, ) from nullrun.breaker.exceptions import ( - BreakerError, NullRunApprovalDeniedError, NullRunApprovalExpiredError, NullRunApprovalReplayRejectedError, @@ -104,7 +101,6 @@ NullRunInfrastructureError, NullRunTransportError, NullRunWorkflowKilledError, - WorkflowKilledInterrupt, WorkflowPausedException, ) from nullrun.context import ( @@ -119,7 +115,6 @@ from nullrun.observability import metrics from nullrun.transport import ( HEADER_PROTOCOL, - NULLRUN_PROTOCOL_VERSION, DecisionSource, FallbackMode, FlushConfig, @@ -979,15 +974,12 @@ def status(self) -> Any: guarantees. * ``ok`` — everything healthy. """ - from datetime import datetime, timezone from nullrun.observability.status import ( STATE_DEGRADED, STATE_MISCONFIGURED, - STATE_OFFLINE, STATE_OK, NullRunStatus, - RecentError, WorkflowState, ) diff --git a/tests/contract/test_llm_call_model_wire.py b/tests/contract/test_llm_call_model_wire.py index 4db84b1..8c03f96 100644 --- a/tests/contract/test_llm_call_model_wire.py +++ b/tests/contract/test_llm_call_model_wire.py @@ -34,8 +34,6 @@ from types import SimpleNamespace from unittest.mock import MagicMock -import pytest - from nullrun.instrumentation.langgraph import _extract_model_from_response # ─── _extract_model_from_response: the actual fix ───────────────────── diff --git a/tests/test_2026_09_10_catchfanin_passthrough.py b/tests/test_2026_09_10_catchfanin_passthrough.py index 7de0068..e40f288 100644 --- a/tests/test_2026_09_10_catchfanin_passthrough.py +++ b/tests/test_2026_09_10_catchfanin_passthrough.py @@ -48,7 +48,6 @@ from __future__ import annotations -import json import re from pathlib import Path diff --git a/tests/test_2026_09_10_check_failopen.py b/tests/test_2026_09_10_check_failopen.py index 6fa4f71..d17c8fb 100644 --- a/tests/test_2026_09_10_check_failopen.py +++ b/tests/test_2026_09_10_check_failopen.py @@ -30,7 +30,6 @@ from __future__ import annotations -import json import re from pathlib import Path diff --git a/tests/test_2026_09_10_runtime_block_typed_dispatch.py b/tests/test_2026_09_10_runtime_block_typed_dispatch.py index 6815a86..df087e1 100644 --- a/tests/test_2026_09_10_runtime_block_typed_dispatch.py +++ b/tests/test_2026_09_10_runtime_block_typed_dispatch.py @@ -50,8 +50,6 @@ class (e.g. ``NullRunApprovalReplayRejectedError`` for import re from pathlib import Path -import pytest - from nullrun.breaker.exceptions import ( NullRunApprovalReplayRejectedError, NullRunBlockedException, diff --git a/tests/test_2026_09_11_auth_heartbeat_sweep.py b/tests/test_2026_09_11_auth_heartbeat_sweep.py index f605716..e6c7462 100644 --- a/tests/test_2026_09_11_auth_heartbeat_sweep.py +++ b/tests/test_2026_09_11_auth_heartbeat_sweep.py @@ -27,7 +27,6 @@ import httpx import pytest import respx -from httpx import Response from nullrun.breaker.exceptions import ( NullRunAuthenticationError, diff --git a/tests/test_actions.py b/tests/test_actions.py index db40dc9..c213d85 100644 --- a/tests/test_actions.py +++ b/tests/test_actions.py @@ -645,7 +645,6 @@ def test_handle_action_module_helper_dispatches(monkeypatch): def test_register_action_handler_module_helper(monkeypatch): - from nullrun import actions as act_mod h = MagicMock() monkeypatch.setattr("nullrun.actions.get_action_handler", lambda: h) @@ -667,7 +666,7 @@ def test_get_action_handler_returns_singleton(): def test_generate_trace_id_is_uuid_format(): - from nullrun.context import generate_span_id, generate_trace_id + from nullrun.context import generate_trace_id tid = generate_trace_id() assert tid.count("-") == 4 # canonical UUID4 @@ -724,7 +723,7 @@ def test_workflow_default_name_is_uuid(): def test_span_context_manager_restores_on_exit(): from nullrun.context import get_span_id, span - with span("outer") as sid: + with span("outer"): assert get_span_id() == "outer" assert get_span_id() is None diff --git a/tests/test_agent_id_uuid.py b/tests/test_agent_id_uuid.py index 223b8a6..35d5945 100644 --- a/tests/test_agent_id_uuid.py +++ b/tests/test_agent_id_uuid.py @@ -16,8 +16,6 @@ import uuid -import pytest - def test_auto_agent_id_is_valid_uuid(): """With no name, agent_id must parse as a UUID (the form the diff --git a/tests/test_approval_timeout_field.py b/tests/test_approval_timeout_field.py index 16542e2..2dab904 100644 --- a/tests/test_approval_timeout_field.py +++ b/tests/test_approval_timeout_field.py @@ -38,8 +38,6 @@ class of bug. import time from typing import Any -import pytest - from nullrun.runtime import NullRunRuntime @@ -285,7 +283,7 @@ def _validate_approval_timeout(value, log_prefix): def test_validate_approval_timeout_accepts_in_range_value(): - from nullrun.runtime import MAX_APPROVAL_TIMEOUT_SECONDS, MIN_APPROVAL_TIMEOUT_SECONDS + from nullrun.runtime import MAX_APPROVAL_TIMEOUT_SECONDS for in_range in (1.0, 5.0, 60.0, 3600.0, MAX_APPROVAL_TIMEOUT_SECONDS): assert _validate_approval_timeout(in_range, "t") == in_range diff --git a/tests/test_audit.py b/tests/test_audit.py index 4e28ea2..f1d2373 100644 --- a/tests/test_audit.py +++ b/tests/test_audit.py @@ -12,8 +12,6 @@ from datetime import datetime, timezone -import pytest - from nullrun.audit import ( AuditEntry, AuditExportJob, diff --git a/tests/test_autogen_patch.py b/tests/test_autogen_patch.py index f30d465..6f3ca03 100644 --- a/tests/test_autogen_patch.py +++ b/tests/test_autogen_patch.py @@ -150,7 +150,6 @@ def test_patch_autogen_idempotent(monkeypatch, fresh_patch_module): from nullrun.instrumentation.autogen import patch_autogen - first_orig = BaseChatAgent.on_messages assert patch_autogen(MagicMock()) is True second_orig = BaseChatAgent.on_messages assert patch_autogen(MagicMock()) is True diff --git a/tests/test_crewai_patch.py b/tests/test_crewai_patch.py index 6cc2dc3..5ba0b6f 100644 --- a/tests/test_crewai_patch.py +++ b/tests/test_crewai_patch.py @@ -105,7 +105,6 @@ def test_patch_crewai_without_async_kickoff(monkeypatch, fresh_patch_module): installs the sync wrap and silently skips the async wrap. """ _install_fake_crewai(monkeypatch, with_async=False) - from crewai import Crew from nullrun.instrumentation.crewai import patch_crewai diff --git a/tests/test_dedup.py b/tests/test_dedup.py index b535c4c..4f9f6cd 100644 --- a/tests/test_dedup.py +++ b/tests/test_dedup.py @@ -332,7 +332,6 @@ def test_track_event_fingerprint_does_not_clobber_caller_fingerprint(self): "_fingerprint": "caller-fp-12345678", # caller's value } # Simulating the runtime's check: do not overwrite. - existing_fp = event.get("_fingerprint") if "_fingerprint" not in event: event["_fingerprint"] = _fingerprint_for_event_dict(event) assert event["_fingerprint"] == "caller-fp-12345678" diff --git a/tests/test_error_hooks.py b/tests/test_error_hooks.py index 2ed6b4d..c4b8407 100644 --- a/tests/test_error_hooks.py +++ b/tests/test_error_hooks.py @@ -18,7 +18,6 @@ """ import logging -import threading from typing import Any from unittest.mock import patch @@ -26,17 +25,9 @@ import nullrun from nullrun.breaker.exceptions import ( - BreakerError, NullRunAuthenticationError, - NullRunAuthError, - NullRunBackendError, - NullRunBlockedException, - NullRunBudgetError, - NullRunConfigError, NullRunError, - NullRunToolBlockedError, WorkflowKilledInterrupt, - WorkflowPausedException, ) from nullrun.observability.error_hooks import ( STAGES, diff --git a/tests/test_exception_hierarchy.py b/tests/test_exception_hierarchy.py index ebdae86..96f6284 100644 --- a/tests/test_exception_hierarchy.py +++ b/tests/test_exception_hierarchy.py @@ -30,7 +30,6 @@ from nullrun.breaker.exceptions import ( # Base - BreakerError, NullRunAuthenticationError, NullRunAuthError, NullRunBackendError, diff --git a/tests/test_gate_real_path.py b/tests/test_gate_real_path.py index 9d02e7e..66f1fe5 100644 --- a/tests/test_gate_real_path.py +++ b/tests/test_gate_real_path.py @@ -36,7 +36,6 @@ import pytest import respx -import nullrun from nullrun.breaker.exceptions import NullRunBudgetError BASE_URL = "https://api.test.nullrun.io" diff --git a/tests/test_integration_contract.py b/tests/test_integration_contract.py index bd312bc..3fc0a89 100644 --- a/tests/test_integration_contract.py +++ b/tests/test_integration_contract.py @@ -13,7 +13,6 @@ from __future__ import annotations -import asyncio import hashlib import hmac import json @@ -28,7 +27,6 @@ generate_hmac_signature, verify_hmac_signature, ) -from nullrun.transport_websocket import WebSocketConnection # ───────────────────────────────────────────────────────────────────── # FIX-F3: every POST must carry Authorization: Bearer so the diff --git a/tests/test_integrations_fastapi.py b/tests/test_integrations_fastapi.py index db01721..716311f 100644 --- a/tests/test_integrations_fastapi.py +++ b/tests/test_integrations_fastapi.py @@ -12,8 +12,7 @@ from typing import Any -import pytest -from fastapi import FastAPI, Request +from fastapi import FastAPI from fastapi.testclient import TestClient from nullrun.breaker import exceptions as exc diff --git a/tests/test_langgraph_callback.py b/tests/test_langgraph_callback.py index 5c7ffa5..d53073b 100644 --- a/tests/test_langgraph_callback.py +++ b/tests/test_langgraph_callback.py @@ -18,8 +18,6 @@ from types import SimpleNamespace from unittest.mock import MagicMock -import pytest - from nullrun.instrumentation.langgraph import ( _ACTIVE_RUNS_MAX, NullRunCallback, diff --git a/tests/test_llm_call_metadata_flags.py b/tests/test_llm_call_metadata_flags.py index 2e6d31d..fe01dd2 100644 --- a/tests/test_llm_call_metadata_flags.py +++ b/tests/test_llm_call_metadata_flags.py @@ -61,7 +61,6 @@ def test_tracked_flag_true_on_normal_call(): """A normal call (under cap, extractor matched) emits tracked: True and NO streaming_skipped flag.""" from nullrun.instrumentation.auto import ( - MAX_RESPONSE_BYTES, NullRunSyncTransport, ) diff --git a/tests/test_lru_active_runs.py b/tests/test_lru_active_runs.py index f994849..6dc6712 100644 --- a/tests/test_lru_active_runs.py +++ b/tests/test_lru_active_runs.py @@ -26,7 +26,7 @@ _ACTIVE_RUNS_MAX, NullRunCallback, ) -from nullrun.tracing import SpanContext, create_root_span +from nullrun.tracing import create_root_span @pytest.fixture diff --git a/tests/test_mcp_adapter.py b/tests/test_mcp_adapter.py index b70f96c..6194a51 100644 --- a/tests/test_mcp_adapter.py +++ b/tests/test_mcp_adapter.py @@ -563,7 +563,6 @@ def test_call_tool_with_runtime_routes_through_execute_before_mcp_call(): with the tool_name + input_data forwarded verbatim. Pins that ``call_tool`` is no longer a silent pass-through. """ - from nullrun.breaker.exceptions import NullRunBlockedException client = _MockMcpClient(_github_inventory()) runtime = _StubRuntime( diff --git a/tests/test_model_fallback.py b/tests/test_model_fallback.py index 1b1e609..05b251d 100644 --- a/tests/test_model_fallback.py +++ b/tests/test_model_fallback.py @@ -39,7 +39,6 @@ def _request_with_body(body: bytes | None) -> httpx.Request: """Build an httpx.Request whose ``.content`` returns the given body.""" - req = httpx.Request("POST", "https://api.openai.com/v1/chat/completions") # httpx.Request stores the content as a property; assignment via # ``.read `` requires content to be bytes. The simplest path is # to construct with content= via the constructor. diff --git a/tests/test_redact.py b/tests/test_redact.py index 596dbb5..da2ae4e 100644 --- a/tests/test_redact.py +++ b/tests/test_redact.py @@ -24,7 +24,6 @@ strictly safer than the pre-fix behavior, where PII was leaking. """ -import pytest from nullrun.decorators import _safe_error_str, _safe_repr, _strip_details_balanced diff --git a/tests/test_registry.py b/tests/test_registry.py index 751c00b..13a9167 100644 --- a/tests/test_registry.py +++ b/tests/test_registry.py @@ -19,9 +19,6 @@ from __future__ import annotations import threading -from typing import Any - -import pytest def test_registry_get_returns_none_initially(): @@ -31,7 +28,7 @@ def test_registry_get_returns_none_initially(): # Use a local registry instance to avoid cross-test pollution # from the global one (the global is already populated by the # test suite's runtime fixtures). - reg = get_registry() + get_registry() def test_registry_set_returns_previous_instance(): @@ -167,7 +164,6 @@ def test_module_proxy_via_install_runtime_proxy(): registry proxy. Verified by writing through the module attribute and reading from the registry directly (and vice versa).""" - import sys import types from nullrun._singleton import ( diff --git a/tests/test_runtime.py b/tests/test_runtime.py index c827c72..887a49f 100644 --- a/tests/test_runtime.py +++ b/tests/test_runtime.py @@ -474,8 +474,7 @@ def test_runtime_singleton_reset_clears_instance(self, mock_api, monkeypatch): # ─── runtime branch tests (kill/pause, mode resolution, etc.) ────────────────────────────── -from types import SimpleNamespace -from unittest.mock import MagicMock, patch +from unittest.mock import MagicMock import pytest diff --git a/tests/test_runtime_branches.py b/tests/test_runtime_branches.py index 57ab4f4..1f590bb 100644 --- a/tests/test_runtime_branches.py +++ b/tests/test_runtime_branches.py @@ -7,13 +7,11 @@ from __future__ import annotations -from types import SimpleNamespace -from unittest.mock import MagicMock, patch +from unittest.mock import MagicMock import pytest from nullrun.breaker.exceptions import ( - NullRunBlockedException, WorkflowKilledInterrupt, WorkflowPausedException, ) diff --git a/tests/test_signal_safety.py b/tests/test_signal_safety.py index bc6327b..efb5a4b 100644 --- a/tests/test_signal_safety.py +++ b/tests/test_signal_safety.py @@ -17,8 +17,6 @@ import gc import signal -import weakref -from unittest.mock import patch import pytest @@ -111,12 +109,6 @@ def test_finalize_is_registered_on_construction(self): try: # `weakref.finalize` registers a finalize on the object. # The `__call__` method exists on the finalize object. - # We can introspect by walking the weakref.finalize - # instances attached to the object. - finalize_objs = [r for r in gc.get_referrers(t) if isinstance(r, weakref.finalize)] - # The weakref is registered as a referrer of t. We can - # at minimum check that the atexit registry is not - # pinned to t. # Note: exact introspection of weakref.finalize is # implementation-dependent; we just ensure the object # is collectable when no longer referenced. @@ -132,7 +124,6 @@ def test_weakref_fires_on_gc(self): api_url="https://api.test.nullrun.io", api_key="test-key-12345678", ) - t_id = id(t) del t gc.collect() # After GC, calling any method on a new transport should @@ -212,7 +203,6 @@ def test_atexit_flush_does_not_persist_buffer(self): The DEBUG log line emitted by the finalizer is the user-visible signal that events were dropped. """ - import logging import tempfile # Use a per-test WAL path so we can verify the finalizer diff --git a/tests/test_transport.py b/tests/test_transport.py index 0cba19c..a7ec5f2 100644 --- a/tests/test_transport.py +++ b/tests/test_transport.py @@ -553,7 +553,6 @@ def test_buffer_overflow_drops_oldest(self): t.track({"event": f"e{i}"}) # Flush with CB OPEN will re-queue and enforce max_buffer_size - initial_buffer_len = len(t._buffer) t._do_flush() # After flush with CB OPEN, buffer should be capped at max_buffer_size @@ -1064,7 +1063,6 @@ def test_verify_hmac_signature_fresh_and_matching(): """Fresh timestamp + correct signature → True.""" import hashlib import hmac as _hmac - import json as _json body = '{"x":1}' ts = int(time.time()) diff --git a/tests/test_transport_branches.py b/tests/test_transport_branches.py index d8e4d47..b33807d 100644 --- a/tests/test_transport_branches.py +++ b/tests/test_transport_branches.py @@ -51,7 +51,6 @@ def test_verify_hmac_signature_fresh_and_matching(): """Fresh timestamp + correct signature → True.""" import hashlib import hmac as _hmac - import json as _json body = '{"x":1}' ts = int(time.time()) diff --git a/tests/test_unified_fingerprint.py b/tests/test_unified_fingerprint.py index 1a167ac..a5182bc 100644 --- a/tests/test_unified_fingerprint.py +++ b/tests/test_unified_fingerprint.py @@ -48,7 +48,6 @@ import respx from nullrun.instrumentation.auto import ( - NullRunSyncTransport, _fingerprint_for_llm_call, _fingerprint_is_seen, make_dedup_state, diff --git a/tests/test_v3_wire_contract.py b/tests/test_v3_wire_contract.py index 53cfa32..194f86a 100644 --- a/tests/test_v3_wire_contract.py +++ b/tests/test_v3_wire_contract.py @@ -25,7 +25,6 @@ NullRunBudgetRecheckFailedError, NullRunChainError, NullRunConsumeOverbudgetError, - NullRunError, NullRunProtocolError, NullRunRateLimitRedisError, NullRunWorkflowInactiveError, @@ -1552,8 +1551,6 @@ def test_runtime_chain_end_invalidates_cache(self, make_runtime): import respx from nullrun.context import ( - _server_minted_execution_id_var, - _server_minted_reservation_at_var, clear_server_minted_execution_id, get_server_minted_execution_id, get_server_minted_reservation_at, diff --git a/tests/test_webhook_backoff.py b/tests/test_webhook_backoff.py index 3cd8a7a..3438c9f 100644 --- a/tests/test_webhook_backoff.py +++ b/tests/test_webhook_backoff.py @@ -26,7 +26,6 @@ idle poll use real wall-clock sleeps. """ -import time from unittest.mock import MagicMock, patch import pytest diff --git a/tests/test_ws_signed_payload.py b/tests/test_ws_signed_payload.py index 4493bf4..2b5ce14 100644 --- a/tests/test_ws_signed_payload.py +++ b/tests/test_ws_signed_payload.py @@ -25,9 +25,6 @@ from __future__ import annotations -import asyncio -import hashlib -import hmac import json import time @@ -102,7 +99,6 @@ def _build_legacy_envelope(message: dict, api_key: str, secret_key: str) -> dict that on the wire so the receiver has to fall back to the legacy "verify against the full wire bytes" path. """ - timestamp = int(time.time()) # Pre-FIX-C: the server was signing the same bytes it is putting on # the wire (full envelope), so to make this envelope verify-able # under the legacy "full wire bytes" rule we have to sign the