From 1d92c02361486dcdfd60d4dee79b2cf6dc54b3c5 Mon Sep 17 00:00:00 2001 From: Eloi Date: Tue, 28 Jul 2026 16:52:09 +0200 Subject: [PATCH 1/9] Run SDK evolution update --- uv.lock | 28 ++++++++++++++-------------- 1 file changed, 14 insertions(+), 14 deletions(-) diff --git a/uv.lock b/uv.lock index bfa2e98..ad01b7c 100644 --- a/uv.lock +++ b/uv.lock @@ -12,7 +12,7 @@ required-markers = [ ] [options] -exclude-newer = "2026-07-03T20:08:59.971702Z" +exclude-newer = "2026-07-20T14:52:02.543284Z" exclude-newer-span = "P8D" [manifest] @@ -384,7 +384,7 @@ wheels = [ [[package]] name = "claude-agent-sdk" -version = "0.2.106" +version = "0.2.123" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "anyio" }, @@ -392,13 +392,13 @@ dependencies = [ { name = "sniffio" }, { name = "typing-extensions", marker = "python_full_version < '3.11'" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/69/73/f5edec88b9548c3757e429dbfc62b16e46cac1300c9463cbb83ed5281266/claude_agent_sdk-0.2.106.tar.gz", hash = "sha256:26c20cf75db1ed609aae6217cb6dd40844365c66acf3a7abf4e877671695228e", size = 255630, upload-time = "2026-06-20T21:11:46.995Z" } +sdist = { url = "https://files.pythonhosted.org/packages/be/fa/233718cb42686f5fb6ca4fd3680d5bd8b5d13421aac890865e197e49eba2/claude_agent_sdk-0.2.123.tar.gz", hash = "sha256:76a6d4f1aa90cac98829951a15fff5678e394ed0d8719cef73f0e6428733b300", size = 296996, upload-time = "2026-07-19T03:08:01.677Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/e7/ec/190cc7304f1a439a2d87a4e4ac9543eb6aed5dd97aab9ce549f83bf7f7ed/claude_agent_sdk-0.2.106-py3-none-macosx_11_0_arm64.whl", hash = "sha256:86506d6a701759a8215083283644513bce56c968b1b95f67108a3114a177eaaf", size = 64468649, upload-time = "2026-06-20T21:11:50.182Z" }, - { url = "https://files.pythonhosted.org/packages/cf/10/bf6cafd2523c3dd9c961ef496485e0540f3fab71a9b1ce125bccfdd3d448/claude_agent_sdk-0.2.106-py3-none-macosx_11_0_x86_64.whl", hash = "sha256:9894cb7bdb276b46085b0ced1e6d867454b56c86fc703ba8e6fdd2a89eec4144", size = 67808581, upload-time = "2026-06-20T21:11:53.335Z" }, - { url = "https://files.pythonhosted.org/packages/35/e0/42b6c608d3add29c2f5057fd9d31c23c18a853cd24c1a51d5962699feab0/claude_agent_sdk-0.2.106-py3-none-manylinux_2_17_aarch64.whl", hash = "sha256:d158f15ba7c4990f72b84d88362dea71d2fa315b1f29772989e22f29c130bdec", size = 73046395, upload-time = "2026-06-20T21:11:56.82Z" }, - { url = "https://files.pythonhosted.org/packages/68/09/0d0b41145e49e5b29087e541791ea74e1c0178f6afa259210ec35f27cc51/claude_agent_sdk-0.2.106-py3-none-manylinux_2_17_x86_64.whl", hash = "sha256:778eb7e6aa07363c8b6d669bcb778e889011430a245bddc624380344ac8cc00b", size = 74000856, upload-time = "2026-06-20T21:12:00.652Z" }, - { url = "https://files.pythonhosted.org/packages/61/39/c97cd496d3bb7be4e2903b0a44ee85265cd9a8d1198d2bd0ea8b355d72c8/claude_agent_sdk-0.2.106-py3-none-win_amd64.whl", hash = "sha256:a9b60131f064dd1a260c55308ed1f26a53db0a942cdf859843ff606c81006400", size = 73741662, upload-time = "2026-06-20T21:12:04.414Z" }, + { url = "https://files.pythonhosted.org/packages/77/f8/e828a9f4c172798cc1bd13d8c59c13c33f40b1fd3a097f2b951a20630caf/claude_agent_sdk-0.2.123-py3-none-macosx_11_0_arm64.whl", hash = "sha256:e333c7edd50dde407b366b1124370e18767fb366d3c5389440891e7417d3c1ca", size = 72705260, upload-time = "2026-07-19T03:08:04.998Z" }, + { url = "https://files.pythonhosted.org/packages/e2/0e/3b6d0c0003b19dd9fcb3b49bbdbea9ec32579241968e0585c69c7b24c9dd/claude_agent_sdk-0.2.123-py3-none-macosx_11_0_x86_64.whl", hash = "sha256:c146514a3d000503b505f4cd8adbee5e4e24e173f160cc05982c4ddc464edbdc", size = 77691262, upload-time = "2026-07-19T03:08:08.763Z" }, + { url = "https://files.pythonhosted.org/packages/cf/8d/0660f9e5188d7462637df06aa874ab6bb73cb90d892bdaf206cd1edf8b4b/claude_agent_sdk-0.2.123-py3-none-manylinux_2_17_aarch64.whl", hash = "sha256:2bc89df6a3ac3e5213dc2c3b6b5ba95ab80d7748da54e6e407570ec3a3ff5458", size = 82177025, upload-time = "2026-07-19T03:08:12.856Z" }, + { url = "https://files.pythonhosted.org/packages/52/38/9587f61926e0c932125aed9c719f5e4d02eda1ef00cba758c152b2725a3f/claude_agent_sdk-0.2.123-py3-none-manylinux_2_17_x86_64.whl", hash = "sha256:90e16da0e07daac71a9fbb73ae2fcaf3b481c15eb524b5d3c9ac6429685b5901", size = 83329822, upload-time = "2026-07-19T03:08:16.495Z" }, + { url = "https://files.pythonhosted.org/packages/a0/2e/04690475f20d8e1c06991574f128af8a67d8469cd15d317783f4800b1455/claude_agent_sdk-0.2.123-py3-none-win_amd64.whl", hash = "sha256:b3cf0e398bd0dc2102ec899109534d3ce02329268c5d5ae70e6e6312b734f084", size = 82872718, upload-time = "2026-07-19T03:08:20.581Z" }, ] [[package]] @@ -605,7 +605,7 @@ wheels = [ [[package]] name = "google-antigravity" -version = "0.1.4" +version = "0.1.7" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "absl-py" }, @@ -617,11 +617,11 @@ dependencies = [ { name = "websockets" }, ] wheels = [ - { url = "https://files.pythonhosted.org/packages/89/e2/1c6f9d4ffb4d97088cac763fa0dd3604504283f4b8d3ea0c07ca33d9f1ee/google_antigravity-0.1.4-py3-none-macosx_11_0_arm64.whl", hash = "sha256:cb83022853d1021abffb103dec248cdff92fe0c6d0121f0229616f5524624d01", size = 30702577, upload-time = "2026-06-18T18:15:52.589Z" }, - { url = "https://files.pythonhosted.org/packages/f7/06/6d853664783e2b17ef2e95a6202add07a331e1dd55430341db257200c251/google_antigravity-0.1.4-py3-none-manylinux_2_17_aarch64.whl", hash = "sha256:4276600a2480cb24bf74f3334f082df2059019236766ea23b5af9f9e0e2a4ae0", size = 33270837, upload-time = "2026-06-18T18:15:55.187Z" }, - { url = "https://files.pythonhosted.org/packages/5f/72/6efc89b3aa36f85f583afa9b8214f3b758ec6cd056be5798c80f58861826/google_antigravity-0.1.4-py3-none-manylinux_2_17_x86_64.whl", hash = "sha256:cbc35edfff18a6795f46c096f8daf2a87653b1739ef4c63b01a04ba795223f95", size = 36411235, upload-time = "2026-06-18T18:15:57.888Z" }, - { url = "https://files.pythonhosted.org/packages/7b/9e/175fcbc9c9f4b5139cf42feed0d58553ec8384c8d21fb5084407cd57c8ce/google_antigravity-0.1.4-py3-none-win_amd64.whl", hash = "sha256:ab2f1cca42ff6c90bd3b2a56f6df8d90958535ff8c23704bc96b939ca273922d", size = 35416341, upload-time = "2026-06-18T18:16:00.805Z" }, - { url = "https://files.pythonhosted.org/packages/c5/9a/838bb2bdb162694452287b032da9ad99661661c58e02446bd620de560621/google_antigravity-0.1.4-py3-none-win_arm64.whl", hash = "sha256:0748cec67c864769501ef5ca7886075fa8c235d750486ef3b805c903f14ce50b", size = 32000969, upload-time = "2026-06-18T18:16:03.691Z" }, + { url = "https://files.pythonhosted.org/packages/c6/aa/609ecaca0611519fe49094d7993e046f274c5d8622541d502b4a6b6bc8e5/google_antigravity-0.1.7-py3-none-macosx_11_0_arm64.whl", hash = "sha256:2707d823bba1dfc4c3384d7d78aea13eb25fdf8a28a8345944b38c16c1405c8a", size = 31644051, upload-time = "2026-07-16T21:03:58.157Z" }, + { url = "https://files.pythonhosted.org/packages/ef/d9/8936a14195c3d923c49f3ee4810ab9dd0655c9618814a7c35837734e075d/google_antigravity-0.1.7-py3-none-manylinux_2_17_aarch64.whl", hash = "sha256:c5070a0f9a6c4406ea460550ae1a543cfb7c9fc185b80921c31ea681aa645ec5", size = 37042570, upload-time = "2026-07-16T21:04:01.126Z" }, + { url = "https://files.pythonhosted.org/packages/41/55/c5e11b0f91c70540a0e247c53de117c0986b167e94e205c131dc8fadcf76/google_antigravity-0.1.7-py3-none-manylinux_2_17_x86_64.whl", hash = "sha256:0958f4481dd4700a0a32acb948f959fb1858657038f103c5a0d067727010f71e", size = 39141605, upload-time = "2026-07-16T21:04:04.392Z" }, + { url = "https://files.pythonhosted.org/packages/3f/80/1bd2f9ef2dc20b498186d6e888c09b86ce519748e85188c898d1c7bba77c/google_antigravity-0.1.7-py3-none-win_amd64.whl", hash = "sha256:c2a20d7a22b9ca086c570efce2d36b688ef15e30a371f7fbb93cd05aba49d563", size = 36434279, upload-time = "2026-07-16T21:04:07.246Z" }, + { url = "https://files.pythonhosted.org/packages/5e/3b/e2cf44fb9da8ac407199c1c351fcdc4876c0aece1b61123b2a439a36af66/google_antigravity-0.1.7-py3-none-win_arm64.whl", hash = "sha256:6acaa5ed047cb9bad6ad50992aea71616eff2a6767161b96aab5f7a1cd765ce5", size = 32907283, upload-time = "2026-07-16T21:04:10.104Z" }, ] [[package]] From 5867298be956d8bc2c718c87ab8f329cb9aa904a Mon Sep 17 00:00:00 2001 From: Eloi Date: Tue, 28 Jul 2026 16:54:04 +0200 Subject: [PATCH 2/9] Sync compatibility manifest with SDK lock --- src/agent_runtime_kit/compatibility.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/agent_runtime_kit/compatibility.py b/src/agent_runtime_kit/compatibility.py index 30422e8..2c88653 100644 --- a/src/agent_runtime_kit/compatibility.py +++ b/src/agent_runtime_kit/compatibility.py @@ -66,7 +66,7 @@ def __post_init__(self) -> None: package="claude-agent-sdk", module="claude_agent_sdk", version_specifier=">=0.2.87,<0.3", - tested_version="0.2.106", + tested_version="0.2.123", ), RuntimeCompatibility( kind=AgentRuntimeKind.CODEX_AGENT_SDK, @@ -85,7 +85,7 @@ def __post_init__(self) -> None: package="google-antigravity", module="google.antigravity", version_specifier=">=0.1.2,<0.2", - tested_version="0.1.4", + tested_version="0.1.7", ), ) From f841af1f3263782e2d8ff84351c5571a05073a02 Mon Sep 17 00:00:00 2001 From: Eloi Date: Tue, 28 Jul 2026 17:30:15 +0200 Subject: [PATCH 3/9] Upgrade vendor SDKs and fix evolution freshness --- .claude/commands/agent-runtime-kit/upgrade.md | 11 +++- .../skills/agent-runtime-kit-upgrade/SKILL.md | 11 +++- docs/sdk-evolution-agent.md | 8 ++- examples/sdk_evolution_agent/cli.py | 6 +- examples/sdk_evolution_agent/collectors.py | 15 +++++ examples/sdk_evolution_agent/snapshots.py | 2 +- pyproject.toml | 14 +++-- src/agent_runtime_kit/compatibility.py | 10 +-- tests/test_sdk_evolution_agent.py | 25 +++++++- uv.lock | 62 ++++++++++--------- 10 files changed, 113 insertions(+), 51 deletions(-) diff --git a/.claude/commands/agent-runtime-kit/upgrade.md b/.claude/commands/agent-runtime-kit/upgrade.md index d19d38a..c927bc2 100644 --- a/.claude/commands/agent-runtime-kit/upgrade.md +++ b/.claude/commands/agent-runtime-kit/upgrade.md @@ -109,6 +109,11 @@ If unspecified, inspect all packages: Run a report-only pass first. Explicitly bypass freshness cutoffs because fresh upstream SDK releases are the point of this workflow: +The runner removes environment cutoffs and passes +`--exclude-newer-package =false` for every monitored vendor package. +The project declares the same package-scoped exemptions, while every other +dependency keeps the repository's normal eight-day delay. + ```bash env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ uv run python -m examples.sdk_evolution_agent \ @@ -181,9 +186,9 @@ After implementation, run or verify: ```bash env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv lock --check -env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run ruff check . -env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run mypy -env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run pytest +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run --locked ruff check . +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run --locked mypy +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run --locked pytest ``` If a draft PR was created, watch CI until it finishes or clearly report that it diff --git a/.codex/skills/agent-runtime-kit-upgrade/SKILL.md b/.codex/skills/agent-runtime-kit-upgrade/SKILL.md index fc52cc8..14865e2 100644 --- a/.codex/skills/agent-runtime-kit-upgrade/SKILL.md +++ b/.codex/skills/agent-runtime-kit-upgrade/SKILL.md @@ -81,6 +81,11 @@ Default runtime for this Codex skill: `codex-agent-sdk`. Always run a report-only pass before implementation. Explicitly bypass uv freshness cutoffs: +The runner removes environment cutoffs and passes +`--exclude-newer-package =false` for every monitored vendor package. +The project declares the same package-scoped exemptions, while every other +dependency keeps the repository's normal eight-day delay. + ```bash env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ uv run python -m examples.sdk_evolution_agent \ @@ -152,9 +157,9 @@ Run or verify: ```bash env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv lock --check -env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run ruff check . -env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run mypy -env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run pytest +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run --locked ruff check . +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run --locked mypy +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run --locked pytest ``` If a draft PR was created, watch CI until it finishes or clearly report that it diff --git a/docs/sdk-evolution-agent.md b/docs/sdk-evolution-agent.md index 6d5f5aa..7c428b8 100644 --- a/docs/sdk-evolution-agent.md +++ b/docs/sdk-evolution-agent.md @@ -106,9 +106,11 @@ installed distributions, then compares it with upstream package metadata for: - `google-antigravity` When `--refresh-preview` is used, the targeted `uv lock --dry-run -P ...` -preview runs with freshness cutoff environment variables removed, including -`UV_EXCLUDE_NEWER`. This workflow needs fresh upstream SDK information, so local -cutoff variables must not hide candidate releases. +preview removes freshness cutoff environment variables, including +`UV_EXCLUDE_NEWER`, and passes `--exclude-newer-package =false` for +each monitored SDK/runtime package. The project declares the same package-scoped +exemptions, so approved lock updates and ordinary locked CI agree while all other +dependencies retain the repository's eight-day delay. ## Candidate API Inspection diff --git a/examples/sdk_evolution_agent/cli.py b/examples/sdk_evolution_agent/cli.py index b776339..1f69644 100644 --- a/examples/sdk_evolution_agent/cli.py +++ b/examples/sdk_evolution_agent/cli.py @@ -47,9 +47,9 @@ ) DEFAULT_VERIFICATION_COMMANDS = ( - "uv run ruff check .", - "uv run mypy", - "uv run pytest", + "uv run --locked ruff check .", + "uv run --locked mypy", + "uv run --locked pytest", "uv lock --check", ) diff --git a/examples/sdk_evolution_agent/collectors.py b/examples/sdk_evolution_agent/collectors.py index 8dc135c..7390a46 100644 --- a/examples/sdk_evolution_agent/collectors.py +++ b/examples/sdk_evolution_agent/collectors.py @@ -22,6 +22,7 @@ ) FRESHNESS_CUTOFF_ENV_VARS = ("UV_EXCLUDE_NEWER",) +SDK_EVOLUTION_EXCLUDE_NEWER_PACKAGE = "false" PACKAGE_SOURCE_HINTS: dict[str, tuple[SourceRef, ...]] = { "claude-agent-sdk": ( @@ -244,6 +245,13 @@ def build_refresh_preview_command(packages: Sequence[str]) -> tuple[str, ...]: command = ["uv", "lock", "--dry-run"] for package in packages: command.extend(("-P", package)) + for package in packages: + command.extend( + ( + "--exclude-newer-package", + f"{package}={SDK_EVOLUTION_EXCLUDE_NEWER_PACKAGE}", + ) + ) return tuple(command) @@ -281,6 +289,13 @@ def run_lock_update( command = ["uv", "lock"] for package in packages: command.extend(("-P", package)) + for package in packages: + command.extend( + ( + "--exclude-newer-package", + f"{package}={SDK_EVOLUTION_EXCLUDE_NEWER_PACKAGE}", + ) + ) result = command_runner(tuple(command), cwd=root, env=env) return CommandResult( command=result.command, diff --git a/examples/sdk_evolution_agent/snapshots.py b/examples/sdk_evolution_agent/snapshots.py index f198536..8659786 100644 --- a/examples/sdk_evolution_agent/snapshots.py +++ b/examples/sdk_evolution_agent/snapshots.py @@ -19,7 +19,7 @@ DEFAULT_MODULES = { "claude-agent-sdk": "claude_agent_sdk", "openai-codex": "openai_codex", - "openai-codex-cli-bin": "openai_codex_cli_bin", + "openai-codex-cli-bin": "codex_cli_bin", "google-antigravity": "google.antigravity", } diff --git a/pyproject.toml b/pyproject.toml index b5b4c56..9d2766a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -48,12 +48,12 @@ dependencies = ["jsonschema>=4.18,<5"] claude = ["claude-agent-sdk>=0.2.87,<0.3"] # 0.1.0b3 floor: earlier betas pin a codex binary with no manylinux wheels, # making the extra uninstallable on glibc Linux. -codex = ["openai-codex>=0.1.0b3,<0.2"] +codex = ["openai-codex>=0.1.0b3,<0.145"] antigravity = ["google-antigravity>=0.1.2,<0.2"] all = [ "claude-agent-sdk>=0.2.87,<0.3", "google-antigravity>=0.1.2,<0.2", - "openai-codex>=0.1.0b3,<0.2", + "openai-codex>=0.1.0b3,<0.145", ] [project.urls] @@ -79,9 +79,15 @@ dev = [ # freshly compromised upload cannot reach the lockfile before the ecosystem has # had time to notice. Committed here (not just a developer env var) so CI and # every contributor resolve under the same policy and `uv sync --locked` agrees -# everywhere. The SDK-evolution agent strips env-level cutoffs for its refresh -# preview; this project-level floor still applies to what it may lock. +# everywhere. Monitored vendor SDK/runtime packages are exempt so the audited +# evolution workflow and ordinary locked CI share one current dependency state. exclude-newer = "8 days" +exclude-newer-package = { + claude-agent-sdk = false, + google-antigravity = false, + openai-codex = false, + openai-codex-cli-bin = false, +} # openai-codex-cli-bin ships per-platform binary wheels; require the platforms we # develop and run CI on so resolution never locks a version missing one of them. required-environments = [ diff --git a/src/agent_runtime_kit/compatibility.py b/src/agent_runtime_kit/compatibility.py index 2c88653..a5ff706 100644 --- a/src/agent_runtime_kit/compatibility.py +++ b/src/agent_runtime_kit/compatibility.py @@ -66,17 +66,17 @@ def __post_init__(self) -> None: package="claude-agent-sdk", module="claude_agent_sdk", version_specifier=">=0.2.87,<0.3", - tested_version="0.2.123", + tested_version="0.2.128", ), RuntimeCompatibility( kind=AgentRuntimeKind.CODEX_AGENT_SDK, extra="codex", package="openai-codex", module="openai_codex", - version_specifier=">=0.1.0b3,<0.2", - tested_version="0.1.0b3", + version_specifier=">=0.1.0b3,<0.145", + tested_version="0.144.4", tested_runtime_dependencies=( - PackageVersion(package="openai-codex-cli-bin", version="0.137.0a4"), + PackageVersion(package="openai-codex-cli-bin", version="0.144.4"), ), ), RuntimeCompatibility( @@ -85,7 +85,7 @@ def __post_init__(self) -> None: package="google-antigravity", module="google.antigravity", version_specifier=">=0.1.2,<0.2", - tested_version="0.1.7", + tested_version="0.1.8", ), ) diff --git a/tests/test_sdk_evolution_agent.py b/tests/test_sdk_evolution_agent.py index 81b8817..d36bd46 100644 --- a/tests/test_sdk_evolution_agent.py +++ b/tests/test_sdk_evolution_agent.py @@ -120,6 +120,12 @@ def runner( assert seen["command"] == build_refresh_preview_command( ("claude-agent-sdk", "google-antigravity") ) + assert seen["command"][-4:] == ( + "--exclude-newer-package", + "claude-agent-sdk=false", + "--exclude-newer-package", + "google-antigravity=false", + ) assert "UV_EXCLUDE_NEWER" not in seen["env"] assert result.removed_env == ("UV_EXCLUDE_NEWER",) @@ -155,6 +161,10 @@ def runner( "claude-agent-sdk", "-P", "google-antigravity", + "--exclude-newer-package", + "claude-agent-sdk=false", + "--exclude-newer-package", + "google-antigravity=false", ) assert "UV_EXCLUDE_NEWER" not in seen["env"] assert result.removed_env == ("UV_EXCLUDE_NEWER",) @@ -646,6 +656,12 @@ def run_new(value: str, *, verbose: bool = False) -> str: assert diff.changed == ("run",) +def test_codex_cli_snapshot_uses_distribution_module_name() -> None: + snapshot = snapshot_current_api("openai-codex-cli-bin", version="0.144.4") + + assert snapshot.module == "codex_cli_bin" + + def test_parse_args_candidate_inspection_is_opt_in() -> None: # Candidate inspection pip-installs+imports upstream code, so it is off by # default and only enabled with the explicit flag. @@ -1359,7 +1375,14 @@ def runner( assert report_path.exists() assert ("git", "switch", "-c", "sdk-update-test") in commands - assert ("uv", "lock", "-P", "claude-agent-sdk") in commands + assert ( + "uv", + "lock", + "-P", + "claude-agent-sdk", + "--exclude-newer-package", + "claude-agent-sdk=false", + ) in commands assert any(command[:3] == ("git", "commit", "-m") for command in commands) assert any(command[:4] == ("gh", "pr", "create", "--draft") for command in commands) assert ("git", "commit", "-m", "Finalize SDK evolution report") in commands diff --git a/uv.lock b/uv.lock index ad01b7c..9916bfe 100644 --- a/uv.lock +++ b/uv.lock @@ -12,9 +12,15 @@ required-markers = [ ] [options] -exclude-newer = "2026-07-20T14:52:02.543284Z" +exclude-newer = "2026-07-20T15:29:02.807093Z" exclude-newer-span = "P8D" +[options.exclude-newer-package] +openai-codex-cli-bin = false +google-antigravity = false +openai-codex = false +claude-agent-sdk = false + [manifest] constraints = [{ name = "openai-codex-cli-bin", specifier = ">=0.134.0a1" }] @@ -71,8 +77,8 @@ requires-dist = [ { name = "google-antigravity", marker = "extra == 'all'", specifier = ">=0.1.2,<0.2" }, { name = "google-antigravity", marker = "extra == 'antigravity'", specifier = ">=0.1.2,<0.2" }, { name = "jsonschema", specifier = ">=4.18,<5" }, - { name = "openai-codex", marker = "extra == 'all'", specifier = ">=0.1.0b3,<0.2" }, - { name = "openai-codex", marker = "extra == 'codex'", specifier = ">=0.1.0b3,<0.2" }, + { name = "openai-codex", marker = "extra == 'all'", specifier = ">=0.1.0b3,<0.145" }, + { name = "openai-codex", marker = "extra == 'codex'", specifier = ">=0.1.0b3,<0.145" }, ] provides-extras = ["claude", "codex", "antigravity", "all"] @@ -384,7 +390,7 @@ wheels = [ [[package]] name = "claude-agent-sdk" -version = "0.2.123" +version = "0.2.128" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "anyio" }, @@ -392,13 +398,13 @@ dependencies = [ { name = "sniffio" }, { name = "typing-extensions", marker = "python_full_version < '3.11'" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/be/fa/233718cb42686f5fb6ca4fd3680d5bd8b5d13421aac890865e197e49eba2/claude_agent_sdk-0.2.123.tar.gz", hash = "sha256:76a6d4f1aa90cac98829951a15fff5678e394ed0d8719cef73f0e6428733b300", size = 296996, upload-time = "2026-07-19T03:08:01.677Z" } +sdist = { url = "https://files.pythonhosted.org/packages/a7/e8/3a9622b31f9ee22274e13a620e5e75ac38454d14538391b2fe3bc7eb76dc/claude_agent_sdk-0.2.128.tar.gz", hash = "sha256:2ac7b2b3bc56ae9037fd284c8690d3dafab9493ecd28d8974bba79a418e1b800", size = 309369, upload-time = "2026-07-25T01:48:25.884Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/77/f8/e828a9f4c172798cc1bd13d8c59c13c33f40b1fd3a097f2b951a20630caf/claude_agent_sdk-0.2.123-py3-none-macosx_11_0_arm64.whl", hash = "sha256:e333c7edd50dde407b366b1124370e18767fb366d3c5389440891e7417d3c1ca", size = 72705260, upload-time = "2026-07-19T03:08:04.998Z" }, - { url = "https://files.pythonhosted.org/packages/e2/0e/3b6d0c0003b19dd9fcb3b49bbdbea9ec32579241968e0585c69c7b24c9dd/claude_agent_sdk-0.2.123-py3-none-macosx_11_0_x86_64.whl", hash = "sha256:c146514a3d000503b505f4cd8adbee5e4e24e173f160cc05982c4ddc464edbdc", size = 77691262, upload-time = "2026-07-19T03:08:08.763Z" }, - { url = "https://files.pythonhosted.org/packages/cf/8d/0660f9e5188d7462637df06aa874ab6bb73cb90d892bdaf206cd1edf8b4b/claude_agent_sdk-0.2.123-py3-none-manylinux_2_17_aarch64.whl", hash = "sha256:2bc89df6a3ac3e5213dc2c3b6b5ba95ab80d7748da54e6e407570ec3a3ff5458", size = 82177025, upload-time = "2026-07-19T03:08:12.856Z" }, - { url = "https://files.pythonhosted.org/packages/52/38/9587f61926e0c932125aed9c719f5e4d02eda1ef00cba758c152b2725a3f/claude_agent_sdk-0.2.123-py3-none-manylinux_2_17_x86_64.whl", hash = "sha256:90e16da0e07daac71a9fbb73ae2fcaf3b481c15eb524b5d3c9ac6429685b5901", size = 83329822, upload-time = "2026-07-19T03:08:16.495Z" }, - { url = "https://files.pythonhosted.org/packages/a0/2e/04690475f20d8e1c06991574f128af8a67d8469cd15d317783f4800b1455/claude_agent_sdk-0.2.123-py3-none-win_amd64.whl", hash = "sha256:b3cf0e398bd0dc2102ec899109534d3ce02329268c5d5ae70e6e6312b734f084", size = 82872718, upload-time = "2026-07-19T03:08:20.581Z" }, + { url = "https://files.pythonhosted.org/packages/45/ff/03613a38a84285cd85f114fc26d4431f545d2b3b344eb8f69f9b0f18f3c6/claude_agent_sdk-0.2.128-py3-none-macosx_11_0_arm64.whl", hash = "sha256:2e47ee95be68cb07612fd5288a40f3307763da5a5adf66d6e03ea49dc0495c9c", size = 75183825, upload-time = "2026-07-25T01:48:30.45Z" }, + { url = "https://files.pythonhosted.org/packages/a3/df/2adbd3077f1a1cada39cd1508990d4008a8ac9ec74c801b24b1b302348b0/claude_agent_sdk-0.2.128-py3-none-macosx_11_0_x86_64.whl", hash = "sha256:55e8fa28918af5620f391692efe80527ad9f2294cec14dadeea93c5161dbefc4", size = 80305667, upload-time = "2026-07-25T01:48:35.091Z" }, + { url = "https://files.pythonhosted.org/packages/cc/56/776a41af67a53d794b420dcd2f1075b585ed3725faa9827164f4a2e945dd/claude_agent_sdk-0.2.128-py3-none-manylinux_2_17_aarch64.whl", hash = "sha256:6c10cc1c5403b2b0b3d8deb79ad5271a8e49b9db6460193599b91f49f16b4cf7", size = 84617823, upload-time = "2026-07-25T01:48:39.297Z" }, + { url = "https://files.pythonhosted.org/packages/17/1c/37044abbddf2141c4eeb6cc8e15a486491549a27283029d73db2c4f3c3b7/claude_agent_sdk-0.2.128-py3-none-manylinux_2_17_x86_64.whl", hash = "sha256:cc4a0f20337e227fe16f00edf7cbe109349707adb440fe51ca6c44f9f8d7d21a", size = 85663462, upload-time = "2026-07-25T01:48:43.982Z" }, + { url = "https://files.pythonhosted.org/packages/a3/c9/ffbd8113080f87a4cd7bbfc8b00a974ea7189e3f666b5d7a2903dec7b74f/claude_agent_sdk-0.2.128-py3-none-win_amd64.whl", hash = "sha256:37b87c8e75daa2a6dc74da6d788e0981a2e5e3a6be80cfa4c287abce2bf70e0b", size = 85566882, upload-time = "2026-07-25T01:48:48.635Z" }, ] [[package]] @@ -605,7 +611,7 @@ wheels = [ [[package]] name = "google-antigravity" -version = "0.1.7" +version = "0.1.8" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "absl-py" }, @@ -617,11 +623,11 @@ dependencies = [ { name = "websockets" }, ] wheels = [ - { url = "https://files.pythonhosted.org/packages/c6/aa/609ecaca0611519fe49094d7993e046f274c5d8622541d502b4a6b6bc8e5/google_antigravity-0.1.7-py3-none-macosx_11_0_arm64.whl", hash = "sha256:2707d823bba1dfc4c3384d7d78aea13eb25fdf8a28a8345944b38c16c1405c8a", size = 31644051, upload-time = "2026-07-16T21:03:58.157Z" }, - { url = "https://files.pythonhosted.org/packages/ef/d9/8936a14195c3d923c49f3ee4810ab9dd0655c9618814a7c35837734e075d/google_antigravity-0.1.7-py3-none-manylinux_2_17_aarch64.whl", hash = "sha256:c5070a0f9a6c4406ea460550ae1a543cfb7c9fc185b80921c31ea681aa645ec5", size = 37042570, upload-time = "2026-07-16T21:04:01.126Z" }, - { url = "https://files.pythonhosted.org/packages/41/55/c5e11b0f91c70540a0e247c53de117c0986b167e94e205c131dc8fadcf76/google_antigravity-0.1.7-py3-none-manylinux_2_17_x86_64.whl", hash = "sha256:0958f4481dd4700a0a32acb948f959fb1858657038f103c5a0d067727010f71e", size = 39141605, upload-time = "2026-07-16T21:04:04.392Z" }, - { url = "https://files.pythonhosted.org/packages/3f/80/1bd2f9ef2dc20b498186d6e888c09b86ce519748e85188c898d1c7bba77c/google_antigravity-0.1.7-py3-none-win_amd64.whl", hash = "sha256:c2a20d7a22b9ca086c570efce2d36b688ef15e30a371f7fbb93cd05aba49d563", size = 36434279, upload-time = "2026-07-16T21:04:07.246Z" }, - { url = "https://files.pythonhosted.org/packages/5e/3b/e2cf44fb9da8ac407199c1c351fcdc4876c0aece1b61123b2a439a36af66/google_antigravity-0.1.7-py3-none-win_arm64.whl", hash = "sha256:6acaa5ed047cb9bad6ad50992aea71616eff2a6767161b96aab5f7a1cd765ce5", size = 32907283, upload-time = "2026-07-16T21:04:10.104Z" }, + { url = "https://files.pythonhosted.org/packages/0b/5f/0f5d6b210faddb4e6d2b0c25072ae0b2430294f5dec0b4eabecc196ea32b/google_antigravity-0.1.8-py3-none-macosx_11_0_arm64.whl", hash = "sha256:be5a4853ae7cb7bff4fbba44073cbed96e1fdcdeebd9602b324bf3abe5614ab2", size = 31829862, upload-time = "2026-07-23T19:20:02.908Z" }, + { url = "https://files.pythonhosted.org/packages/a7/cb/2d74bfc57e9f7f579a97f8b445599639ac4a101bcf44bab2a9c58c1f634f/google_antigravity-0.1.8-py3-none-manylinux_2_17_aarch64.whl", hash = "sha256:06773b250668993cf59e3a90ba00f1d6fc52483612d9463aa3e71e4a2a30a20b", size = 37289461, upload-time = "2026-07-23T19:20:06.571Z" }, + { url = "https://files.pythonhosted.org/packages/93/54/9972ecf8e0b5e3d12067550462078be9babf8dc0f686e80ffe70daef0cad/google_antigravity-0.1.8-py3-none-manylinux_2_17_x86_64.whl", hash = "sha256:bd912cc0ac74fef8026c227500b7de06b3749a78f0148b99d9033e8c952787e0", size = 39396935, upload-time = "2026-07-23T19:20:09.493Z" }, + { url = "https://files.pythonhosted.org/packages/bb/ee/52f63973142aa00b722d7d841b60991798352564538f8533c9eb27bb9546/google_antigravity-0.1.8-py3-none-win_amd64.whl", hash = "sha256:edc1690c3116d1da31b033951d62a56a005f73d13486ced1d11564c0b98ca244", size = 36678881, upload-time = "2026-07-23T19:20:13.11Z" }, + { url = "https://files.pythonhosted.org/packages/a1/f6/503b6efb48903cfca252fb22077c15849da302e3d3f4d7a7b7cd2438e146/google_antigravity-0.1.8-py3-none-win_arm64.whl", hash = "sha256:999bdc9ed8d413a6abff96b15105e72b7c596fcd7fce41448f6613ee0947a8b4", size = 33138761, upload-time = "2026-07-23T19:20:16.811Z" }, ] [[package]] @@ -947,30 +953,30 @@ wheels = [ [[package]] name = "openai-codex" -version = "0.1.0b3" +version = "0.144.4" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "openai-codex-cli-bin" }, { name = "pydantic" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/ae/1c/1e5e8b83ea72164d32b1f4e67fc703c8b83591f498a7aaf96f39d352b453/openai_codex-0.1.0b3.tar.gz", hash = "sha256:b76b7afe97953ac65648e9b8ca116b5ff273de91086549bd7ec88037cdc16cab", size = 58995, upload-time = "2026-06-03T19:17:34.707Z" } +sdist = { url = "https://files.pythonhosted.org/packages/b9/7d/4b999b0ddd05c22d83cc2e879c95e1e92e3b7808b58ac561c4ea91fca1c4/openai_codex-0.144.4.tar.gz", hash = "sha256:91c63a7cb213441569f130e593386b34657ab9e726ae88af255f0ecb8de08ea5", size = 68324, upload-time = "2026-07-17T23:42:17.013Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/d7/ef/f77037d9ccde80a688a17a06aea5a56813ad9c365d49b3f1c7913422af8b/openai_codex-0.1.0b3-py3-none-any.whl", hash = "sha256:8d1f9d346667aeecb435c6a45d0edb3f016187276ec452cf8094d813896276c4", size = 65639, upload-time = "2026-06-03T19:17:33.208Z" }, + { url = "https://files.pythonhosted.org/packages/17/35/5d2a13e38d91278019f18757ac10426d2649b7ec6f031818c792f64559e9/openai_codex-0.144.4-py3-none-any.whl", hash = "sha256:de1513a6e94b9a8d7728a3b74298bc1469428ade10ba0ef2d5db47dd1cb606f5", size = 76244, upload-time = "2026-07-17T23:42:15.658Z" }, ] [[package]] name = "openai-codex-cli-bin" -version = "0.137.0a4" +version = "0.144.4" source = { registry = "https://pypi.org/simple" } wheels = [ - { url = "https://files.pythonhosted.org/packages/bd/60/af73ef1676cd477fa83ed4b889bf3b57c63c47dd87025b2cc4262793cff6/openai_codex_cli_bin-0.137.0a4-py3-none-macosx_10_9_x86_64.whl", hash = "sha256:b33c3917e0b58d527ee11a11a78ad390f7d8e6aa25577dd21665ab3c8bf5cf9a", size = 94300191, upload-time = "2026-06-03T18:44:36.312Z" }, - { url = "https://files.pythonhosted.org/packages/92/8f/d1a5f8c87176e00ef6a85798794f4530f5eb04e5a1a13468b5b3c3a361f9/openai_codex_cli_bin-0.137.0a4-py3-none-macosx_11_0_arm64.whl", hash = "sha256:3d0f0bc5becc88c61952fbfa9bd792ac9d74fa78b3a6bd40f545b612048b07eb", size = 83924479, upload-time = "2026-06-03T18:44:40.854Z" }, - { url = "https://files.pythonhosted.org/packages/3e/3c/fc00bcdc0c302208317d5eb1d0bfaab3024f351cd0121400f19baa6b19aa/openai_codex_cli_bin-0.137.0a4-py3-none-manylinux_2_17_aarch64.whl", hash = "sha256:2f1656339e2736868c4cce59f6d9e5c633879123687169b03b1137d42bf2c11a", size = 83363315, upload-time = "2026-06-03T18:44:44.851Z" }, - { url = "https://files.pythonhosted.org/packages/ec/09/39362e944ebeb12fcbfb86881fbb4dd6e806f77f7541c1f1f993bb9351a0/openai_codex_cli_bin-0.137.0a4-py3-none-manylinux_2_17_x86_64.whl", hash = "sha256:6454f838d44c56c1ed07a29b391fa412785e5dd2ffd06db0b62e62478c19bb64", size = 90611239, upload-time = "2026-06-03T18:44:49.338Z" }, - { url = "https://files.pythonhosted.org/packages/fa/38/87b1247fdfe95cddce7f7fe8331d6843cf037e14292c0f5004e23247133b/openai_codex_cli_bin-0.137.0a4-py3-none-musllinux_1_1_aarch64.whl", hash = "sha256:f5ae7401d00c65d56a75d9645d7bf87d809566a12d238e4b2a8b328a02f2316e", size = 83363315, upload-time = "2026-06-03T18:44:53.428Z" }, - { url = "https://files.pythonhosted.org/packages/fb/c4/3c693ad07e587f6b3a28128c417f2e831d81a40cdbd85c0e5f0f36aaff82/openai_codex_cli_bin-0.137.0a4-py3-none-musllinux_1_1_x86_64.whl", hash = "sha256:3dcec1e649448be498d6e7ec0e1f71dca83efa76063d90890dafb41e987069b7", size = 90611238, upload-time = "2026-06-03T18:44:57.612Z" }, - { url = "https://files.pythonhosted.org/packages/9e/26/81e037066b9b8d312a6f9e09015e452ce17630d5ab88e02a4c1d9503e4e8/openai_codex_cli_bin-0.137.0a4-py3-none-win_amd64.whl", hash = "sha256:9e13bf68e18e36bd3a0efd51213281c83e9f6ec22bdb7a45bd2e0211822733a9", size = 94744969, upload-time = "2026-06-03T18:45:02.23Z" }, - { url = "https://files.pythonhosted.org/packages/0d/a3/952bc2a5d62373a51fea161effe3b338b3417c2f6e65fe467ed91b205e2b/openai_codex_cli_bin-0.137.0a4-py3-none-win_arm64.whl", hash = "sha256:5ec4303ca2dcb5f838e0de3ca7f44050b6bcdd41d281a178c3a1420a985a515d", size = 86963504, upload-time = "2026-06-03T18:45:07.131Z" }, + { url = "https://files.pythonhosted.org/packages/79/30/7e457c007a32aa7333a78438a9f532504c3b77a1ea57e7808855712c2c0f/openai_codex_cli_bin-0.144.4-py3-none-macosx_10_9_x86_64.whl", hash = "sha256:4d587d152d5f0aa25f21ded4d08aac5150da48639b29fc3e57b2a8b413d06903", size = 126851183, upload-time = "2026-07-15T00:14:00.171Z" }, + { url = "https://files.pythonhosted.org/packages/65/eb/64c180514a2cc3e2500e486813f5a8d7f7e349342e9bebd78d99ddd9791a/openai_codex_cli_bin-0.144.4-py3-none-macosx_11_0_arm64.whl", hash = "sha256:05db505a9c7f020f58b70837a94e00d32a50086986c267bcc44ea97b573d4a05", size = 116474758, upload-time = "2026-07-15T00:14:08.177Z" }, + { url = "https://files.pythonhosted.org/packages/85/34/921ab692c7ed140941a91e39b84c6deafddb45072793f3a5f6dcbf86f59a/openai_codex_cli_bin-0.144.4-py3-none-manylinux_2_17_aarch64.whl", hash = "sha256:cd4bb31b8a477a3adba22139c8c493693fa5292ba21e86eef08ace3e96c41294", size = 119437596, upload-time = "2026-07-15T00:14:17.418Z" }, + { url = "https://files.pythonhosted.org/packages/25/62/39e630cf8b7b2e2444a5a6669235a6cd88004869a25bb7f3f629ed35638c/openai_codex_cli_bin-0.144.4-py3-none-manylinux_2_17_x86_64.whl", hash = "sha256:4106229c38f37245c3eea8d904426afd45b8bf19831765715647f5402f757c06", size = 128817757, upload-time = "2026-07-15T00:14:30.539Z" }, + { url = "https://files.pythonhosted.org/packages/23/24/318f91a95baaff845307e529d138d42a7202e6bd0e626556058eab3dead5/openai_codex_cli_bin-0.144.4-py3-none-musllinux_1_1_aarch64.whl", hash = "sha256:d2d3fada11731938e3d3e3660819d5324b7f92e825a0ed5aede63100b36ffc9a", size = 119437594, upload-time = "2026-07-15T00:14:39.327Z" }, + { url = "https://files.pythonhosted.org/packages/2e/a0/65e6fd3a6fba52639937801d9ce517b0c48026458723bb9f08eb07a1dd35/openai_codex_cli_bin-0.144.4-py3-none-musllinux_1_1_x86_64.whl", hash = "sha256:70fb62ed7755e332dd8b00ed48e22ee16df10f4a977335039e237cd330490df3", size = 128817755, upload-time = "2026-07-15T00:14:47.849Z" }, + { url = "https://files.pythonhosted.org/packages/e3/8a/3ab5fc97352e8f780d472c8dd6d7504d6a670cd7dede106d461ac1171999/openai_codex_cli_bin-0.144.4-py3-none-win_amd64.whl", hash = "sha256:56e142974467332f1f669b89f2636e06b3ada09413512abb2f24c33b4a4f59fc", size = 140595092, upload-time = "2026-07-15T00:14:58.484Z" }, + { url = "https://files.pythonhosted.org/packages/70/1a/3aa52ab8f89e0f596b6eea835e3f3494697da51917d3231f5f48f92e2bcb/openai_codex_cli_bin-0.144.4-py3-none-win_arm64.whl", hash = "sha256:2ad01058db7181323ae7c6217e560702e38c4f107595b99ab1d0abc9f184890a", size = 130665105, upload-time = "2026-07-15T00:15:08.861Z" }, ] [[package]] From 060faab4ca5f3c577f903fc289dcd0c00e709945 Mon Sep 17 00:00:00 2001 From: Eloi Date: Tue, 28 Jul 2026 17:32:45 +0200 Subject: [PATCH 4/9] Use portable TOML for SDK freshness exemptions --- pyproject.toml | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 9d2766a..70bb681 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -82,12 +82,6 @@ dev = [ # everywhere. Monitored vendor SDK/runtime packages are exempt so the audited # evolution workflow and ordinary locked CI share one current dependency state. exclude-newer = "8 days" -exclude-newer-package = { - claude-agent-sdk = false, - google-antigravity = false, - openai-codex = false, - openai-codex-cli-bin = false, -} # openai-codex-cli-bin ships per-platform binary wheels; require the platforms we # develop and run CI on so resolution never locks a version missing one of them. required-environments = [ @@ -98,6 +92,12 @@ required-environments = [ # marker here opts that one package into pre-release resolution. constraint-dependencies = ["openai-codex-cli-bin>=0.134.0a1"] +[tool.uv.exclude-newer-package] +claude-agent-sdk = false +google-antigravity = false +openai-codex = false +openai-codex-cli-bin = false + [tool.hatch.build.targets.wheel] packages = ["src/agent_runtime_kit"] From f030ebb12378b1a041ca0e36526a5e2efb95c652 Mon Sep 17 00:00:00 2001 From: Eloi Date: Tue, 28 Jul 2026 19:53:40 +0200 Subject: [PATCH 5/9] Plan SDK evolution demo reliability repairs (#51) ## Summary - records the reproduced false-green and missing-baseline failure modes on current `main` - defines the evidence invariants, First Principles model, Five Whys, and owning fix layers - lays out a five-PR delivery stack with exact files, tests, acceptance criteria, and report-only rehearsals - keeps the replacement stack independent of stale open PRs #19 and #40 ## Why The Agent SDK stages execute successfully, but the documented evolution workflow can label ambient import failures as locked SDK evidence, collapse incomplete probes into a passing headline, and permanently fail CLI-binary snapshots through the wrong module name. The repair needs to preserve the explicit candidate-inspection and credential-scrubbing boundary while making the supported command actionable. ## Stack This is PR 1 of 5 and contains planning documentation only. The implementation branches will be based in order on this branch: 1. truthful locked-baseline collection 2. fail-closed evidence summaries and exact transition gates 3. Codex CLI binary introspection 4. actionable runtime-specific operator commands and documentation conformance ## Validation - `git diff --cached --check` - reviewed against `origin/main` at `2e7e4c3` - checked against the current implementation, tests, documentation, and live PR topology --- ...26-07-13-sdk-evolution-demo-reliability.md | 444 ++++++++++++++++++ 1 file changed, 444 insertions(+) create mode 100644 docs/plans/2026-07-13-sdk-evolution-demo-reliability.md diff --git a/docs/plans/2026-07-13-sdk-evolution-demo-reliability.md b/docs/plans/2026-07-13-sdk-evolution-demo-reliability.md new file mode 100644 index 0000000..d0104b0 --- /dev/null +++ b/docs/plans/2026-07-13-sdk-evolution-demo-reliability.md @@ -0,0 +1,444 @@ +# SDK Evolution Demo Reliability Repair Plan + +Status: implementation-ready +Written: 2026-07-13 +Baseline: `origin/main` at `2e7e4c3` (`v0.5.0`) + +## Outcome + +Make the SDK evolution demo produce truthful, actionable evidence from the +documented local workflow, including when only the selected runtime extra is +installed. A successful decision run must compare an observed locked SDK +baseline with an observed resolver-selected candidate, report incomplete +evidence as incomplete, and preserve the existing fail-closed implementation +gates. + +This plan is intentionally narrower than the historical overhaul in draft PR +#40. It fixes the defects reproduced on current `main` without importing the +older branch's unrelated transaction, resolver, scheduling, or implementation +changes. + +## Reproduced behavior + +The following behavior was reproduced from a fresh `origin/main` worktree: + +1. The focused SDK evolution suite passes (`50 passed`). +2. The full suite passes (`418 passed, 3 skipped`), as do Ruff, mypy, and + `uv lock --check`. +3. A normal Codex-backed report runs all three structured AI stages and reuses + one Codex SDK process, but its behavior headline says `pass` while the raw + Claude and Antigravity baseline probes contain import failures. +4. Adding `--inspect-candidates` installs and probes the candidates, but the + absent locked SDKs are still probed through the ambient environment. The + report therefore compares `fail` to `pass`, emits no candidate API diff, and + is rejected by the deterministic API-diff gate. +5. Installing every vendor extra before the same command produces two valid API + diffs, passing before/after behavior probes, `safe_to_implement=true`, and a + passing reviewer decision. +6. The `openai-codex-cli-bin` distribution imports as `codex_cli_bin`; the + snapshot map currently asks for `openai_codex_cli_bin` and records a false + import error. +7. The public examples use bare `python -m`. In the fresh worktree that selected + Python 3.9 despite `.python-version` declaring 3.10, and failed before CLI + parsing because `dataclasses.field(kw_only=...)` requires Python 3.10. + +The Agent SDK execution layer is therefore healthy. The broken boundaries are +locked-baseline evidence collection, behavior aggregation, distribution +introspection, and the operator command contract. + +## First-principles model + +### Actors and boundaries + +- `uv.lock` defines the current dependency baseline. +- The active environment may contain only core plus the selected runtime extra. +- `collect_evidence` records locked, installed, latest, and resolver-selected + versions. +- API snapshots and behavior probes observe baseline and candidate packages. +- Diff and summary layers convert observations into evidence. +- Deterministic guards decide whether implementation is allowed. +- AI stages interpret the evidence but cannot override deterministic guards. +- Documentation and repo-local runbooks define the supported operator command. + +### Invariants + +1. A version comparison is valid only when both versions were actually + observed. +2. The current side of an SDK update comparison is the locked version, not an + arbitrary ambient import. +3. Missing or skipped evidence is `incomplete`; it is never equivalent to + `pass` or to a behavioral change. +4. Candidate and isolated-baseline installation remains explicit and runs only + behind `--inspect-candidates` in a credential-scrubbed environment. +5. A required failed or incomplete behavior probe blocks implementation. +6. The documented command supplies the selected runtime dependency and the + explicit candidate-inspection consent required to satisfy the gates. +7. The outer `uv run` is locked and cannot rewrite dependency state while + preparing the environment; the agent's explicit refresh preview remains the + only freshness path. +8. Vendor-specific surfaces remain explicit; this repair does not flatten + provider capabilities or change adapter contracts. + +### Failure dynamics + +The failure is environment-dependent and deterministic. It occurs when a +locked optional SDK is absent or differs from the active environment. It is +masked when all vendor extras happen to be installed, which is why unit tests +and developer environments can appear healthy. + +## Five whys and fix layer + +1. **Why is the report false-green or non-actionable?** The locked Claude and + Antigravity baselines are not observed in the normal Codex-only environment. +2. **Why are they not observed?** Isolated baseline probing requires + `installed_version` to be present and different from `locked_version`; an + absent installation takes the ambient path. +3. **Why does bad evidence reach the report?** The ambient failure is stamped + with the locked version, failed snapshots are omitted from diffs, and the + behavior summary consumes diffs rather than raw probe results. +4. **Why does this happen in the supported workflow?** The runbooks install only + enough for auth and omit `--inspect-candidates` from the decision and + implementation commands; public examples also bypass the repository's + declared Python and lockfile through bare `python -m`. +5. **Why did tests not catch it?** They cover installed-version drift and + opt-in candidate safety, but not a missing optional baseline, raw failure + aggregation, or command/document conformance. + +The proximal symptom is a contradictory report or a deterministic rejection. +The mechanism is invalid provenance followed by loss of missing/failure state. +The trigger is an absent optional SDK in a runtime-specific environment. The +root cause is the evidence contract across collection, aggregation, and +operator invocation. The fix belongs in those layers because they own what was +observed, how it is classified, and how operators request actionable evidence. + +## Stacked delivery model + +Every PR is based on the branch immediately above it. Review each PR against its +declared base so only that phase's delta is visible. + +| PR | Branch | Base | Scope | +| --- | --- | --- | --- | +| 1 | `agent/sdk-evolution-demo-reliability-plan` | `main` | This plan only | +| 2 | `agent/sdk-evolution-locked-baselines` | PR 1 branch | Observe the locked baseline correctly | +| 3 | `agent/sdk-evolution-evidence-integrity` | PR 2 branch | Make summaries and gates fail closed | +| 4 | `agent/sdk-evolution-cli-bin-snapshots` | PR 3 branch | Fix CLI binary introspection | +| 5 | `agent/sdk-evolution-operator-contract` | PR 4 branch | Make supported commands actionable | + +All PRs remain draft until their phase checks and ordinary correctness review +pass. Reviews must not run security-diff scans, per the delivery instruction. + +## PR 2: Observe locked baselines correctly + +### Production changes + +- In `collect_behavior_evidence`, treat `installed_version is None` the same as + installed-version drift when `locked_version` exists. +- With `--inspect-candidates`, probe that locked version in the existing + credential-scrubbed isolated environment and label it `current-baseline`. +- Without the opt-in, do not install anything and do not relabel an ambient + import as the locked baseline. Record only the actual ambient installed + version (or `None` when absent); PR 3 classifies that unavailable locked + comparison as incomplete. +- Apply the same missing-or-drifted decision to `_collect_snapshots`: with the + opt-in, snapshot the locked version in an isolated environment; without it, + retain only an explicitly unusable current-environment snapshot whose + version is the observed installed version, not the requested lock. +- Preserve the fast ambient path when installed and locked versions match. +- Convert isolated snapshot venv creation, package installation, execution, + timeout, and malformed-output failures into bounded structured snapshot + evidence rather than aborting the report. Preserve package, requested version, + and the scrubbed-environment boundary; do not copy unbounded subprocess output + into artifacts. Behavior subprocess error handling lands atomically with its + summary and guard in PR 3. + +### Tests + +- Missing installed SDK plus opt-in probes the locked version as + `current-baseline` before the candidate. +- Missing installed SDK without opt-in performs no isolated install and never + stamps the locked version onto the ambient result. +- Snapshot collection isolates a missing locked baseline when opted in. +- Snapshot collection performs no isolated install without the flag. +- A matching installed/locked version remains eligible for the ambient path. + Drifted or absent baselines use `current-baseline` when opted in and remain + explicitly unusable when the flag is off. +- Assert that isolated baseline and candidate subprocesses retain the scrubbed + environment contract. +- Snapshot venv creation, install, execution, timeout, and malformed-output + failures produce an explicit failed snapshot artifact and a completed report. + +### Files + +- `examples/sdk_evolution_agent/behavior.py` +- `examples/sdk_evolution_agent/snapshots.py` +- `examples/sdk_evolution_agent/cli.py` +- `tests/test_sdk_evolution_agent.py` + +### Acceptance + +- A core/Codex-only environment with `--inspect-candidates` produces valid + locked-to-candidate inputs without requiring `[all]`. +- No freshly downloaded package is installed or imported when the flag is off. +- No `current-environment` failure is presented as an observed locked baseline. + +## PR 3: Make evidence summaries and gates truthful + +### Production changes + +- Change behavior summarization to consume raw probe results as well as diffs. +- Introduce one shared resolver-transition parser returning exact + `(package, from_version, to_version)` tuples and use it for snapshot/behavior + collection plus deterministic gates. Cover prerelease and multiple-update + output so duplicated weaker parsers cannot drift. +- Define four report states: + - `pass`: required observations exist and no contract change is detected. + - `changed`: required observations exist and only non-breaking change exists. + - `incomplete`: a required observation was skipped, missing, malformed, or + failed to execute/import/install. + - `fail`: a probe completed and proved a required contract failure, or a + breaking diff exists. +- Use deterministic precedence `fail > incomplete > changed > pass`. A failure + with `details.missing` is a proven contract failure; `details.error`, skips, + and missing comparisons are incomplete evidence. +- Preserve existing summary keys and add contract-failure, probe-error, skipped, + and missing-comparison counts plus concise reasons. +- Persist the additive summary as `behavior_summary.json` and render the same + counts/reasons in the Markdown report. +- Add `behavior_summary.json` to `current_state.json` artifact references so its + path and hash participate in the durable baseline manifest. +- Keep one-sided skips out of behavior diffs; absence is not a change. +- Compute missing comparisons only for resolver-selected transitions. A package + with a passing baseline and no selected update has no required candidate and + remains eligible for `pass`. +- Convert behavior venv creation, install, probe execution, timeout, and + malformed-output failures into bounded structured probe evidence. Land this + with classification and gating so a formerly loud failure can never become an + intermediate false-green. +- Use one deterministic assessment helper for collection, reporting, and + gating; never trust a contradictory cached summary over raw results/diffs. +- Extend `with_behavior_probe_guard` so `fail` and `incomplete` both set + `safe_to_implement=false`, set `manual_design_required=true`, and preserve a + concrete reason in findings/uncertainty. +- Fail closed when the behavior summary is missing, malformed, or has an unknown + status. +- Require API-diff evidence to match the exact resolver-selected package, + locked `from_version`, and candidate `to_version`; package membership alone + is insufficient. +- Surface API snapshot import/execution errors in `report.md` so a zero diff + count cannot imply successful inspection. +- Keep `pass` and non-breaking `changed` eligible for later deterministic gates. + +### Tests + +- A baseline execution/import error with no candidate diff reports `incomplete`, + not `pass`. +- A completed baseline contract failure reports `fail`. +- Skipped baseline/candidate evidence reports `incomplete`, not `pass`. +- Candidate install/probe execution failure reports `incomplete` and blocks + implementation; a successfully executed probe with missing required contract + fields reports `fail`. +- A complete unchanged pair reports `pass`. +- A complete non-breaking changed pair reports `changed` without being treated + as breaking. +- Status precedence covers fail plus incomplete and changed plus incomplete. +- A passing baseline with no selected update has + `missing_comparison_count=0`; an expected update with a missing candidate is + incomplete. +- The Markdown report renders status, failure/skip/error counts, reasons, and + snapshot errors; `behavior_summary.json` matches it. +- `current_state.json` records the summary artifact's path and hash. +- The behavior guard blocks `fail` and `incomplete`, while allowing `pass` and + non-breaking `changed` to continue to other gates. +- Missing, non-mapping, unknown-status, and contradictory `pass` summaries with + raw fail/skip evidence all block deterministically. +- The API guard rejects a diff for the correct package with the wrong from/to + versions and accepts an exact empty transition diff. +- Shared transition parsing covers multiple packages and prerelease versions. + +### Files + +- `examples/sdk_evolution_agent/behavior.py` +- `examples/sdk_evolution_agent/collectors.py` +- `examples/sdk_evolution_agent/stages.py` +- `examples/sdk_evolution_agent/report.py` +- `examples/sdk_evolution_agent/current_state.py` +- `docs/sdk-evolution-agent.md` +- `tests/test_sdk_evolution_agent.py` + +### Acceptance + +- The headline can be derived from the raw artifact without contradiction. +- No empty diff set can hide failed or skipped required observations. +- Deterministic gates, not an AI interpretation, own the fail-closed decision. + +## PR 4: Fix Codex CLI binary introspection + +### Production changes + +- Map `openai-codex-cli-bin` to its real import module, `codex_cli_bin`. +- Keep the distribution name unchanged for package metadata and resolver work. +- Strengthen the binary behavior probe to import `codex_cli_bin`, call + `bundled_codex_path()`, and prove the bundled executable exists instead of + treating distribution metadata alone as success. +- Preserve the `binary-distribution` probe name when the isolated embedded probe + raises; it currently hardcodes `adapter-contract`, which prevents correct + before/after pairing. +- Do not introduce a special lowest-common-denominator adapter abstraction; the + snapshot remains package-specific evidence. + +### Tests + +- Current-environment snapshot imports `codex_cli_bin` for the distribution. +- The isolated snapshot script receives the same module name. +- The real installed locked package produces a snapshot without import error + when the Codex extra is available. +- The binary behavior probe passes only when the module and bundled path are + usable, and failures remain labeled `binary-distribution`. +- Existing package mappings remain unchanged. + +### Files + +- `examples/sdk_evolution_agent/snapshots.py` +- `examples/sdk_evolution_agent/behavior.py` +- `tests/test_sdk_evolution_agent.py` + +### Acceptance + +- Current reports no longer contain a synthetic CLI-bin import failure. +- A future resolver-selected CLI-bin update can produce a before/after API diff + instead of being permanently blocked by the wrong module name. + +## PR 5: Make the operator workflow actionable + +### Documentation and runbook changes + +- Update `.codex/skills/agent-runtime-kit-upgrade/SKILL.md` so Codex-backed runs + use `uv run --locked --extra codex` for auth, report, and implementation + passes. +- Update `.claude/commands/agent-runtime-kit/upgrade.md` so Claude-backed runs + use `uv run --locked --extra claude`; document the corresponding Codex and + Antigravity extras when another runtime is selected. +- Couple runtime and extra selection explicitly everywhere: + `claude-agent-sdk` uses `--extra claude`, `codex-agent-sdk` uses + `--extra codex`, and `antigravity-agent-sdk` uses `--extra antigravity`. + Remove instructions that say to replace only `--runtime`. +- Replace bare public `python -m` commands with `uv run --locked` so commands + honor the supported Python version and do not rewrite `uv.lock` while setting + up the outer environment. +- Add `--inspect-candidates` to the actionable report-only decision pass and to + the implementation pass. +- Update `docs/sdk-evolution-agent.md` and its design companion with complete + runtime-specific commands, the `incomplete` status contract, and the new + `behavior_summary.json` artifact. +- Add `behavior_summary.json` to the inspection lists in both repo-owned + runbooks. +- State explicitly that `--inspect-candidates` installs both a missing/drifted + locked baseline and resolver-selected candidates in scrubbed environments. +- Preserve the credential-free fake runtime as a pipeline-shape check, not an + upgrade decision. + +### Tests + +- Add a small documentation-conformance test that asserts each repo-owned + runbook's default real-runtime command includes its required extra, + `uv run --locked`, `--refresh-preview`, and `--inspect-candidates`. +- Assert all three runtime-to-extra mappings. +- Assert implementation examples retain the same inspection flag. +- Reject bare `python -m examples.sdk_evolution_agent` in executable operator + command blocks while allowing explanatory prose. +- Assert the docs still describe candidate inspection as explicit opt-in and + never imply that the fake reviewer proves upgrade safety. + +### Files + +- `.codex/skills/agent-runtime-kit-upgrade/SKILL.md` +- `.claude/commands/agent-runtime-kit/upgrade.md` +- `docs/sdk-evolution-agent.md` +- `docs/sdk-evolution-agent-design.md` +- `tests/test_sdk_evolution_docs.py` + +### Acceptance + +- Copying the documented Claude, Codex, or Antigravity command into a fresh + worktree supplies the selected runtime and can satisfy candidate evidence + gates. +- The command does not require installing every vendor extra. +- Runbooks and public docs agree on opt-in execution, authentication, evidence + states, and stop gates. + +## Cross-stack verification + +Run the smallest relevant test set during development, then all of these from +the top branch: + +```bash +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv lock --check +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run --locked ruff check . +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run --locked mypy +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run --locked pytest -q +``` + +Also rehearse these exact report-only paths in a fresh worktree: + +```bash +# Explicitly incomplete discovery pass: candidates exist but inspection is off. +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ + uv run --locked python -m examples.sdk_evolution_agent \ + --runtime fake \ + --refresh-preview + +# Complete deterministic evidence pass from a core-only environment. +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ + uv run --locked python -m examples.sdk_evolution_agent \ + --runtime fake \ + --refresh-preview \ + --inspect-candidates + +# Complete real-runtime pass with only the selected runtime extra. +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ + uv run --locked --extra codex python -m examples.sdk_evolution_agent \ + --runtime codex-agent-sdk \ + --refresh-preview \ + --inspect-candidates +``` + +The first command must report incomplete candidate evidence. The second must +produce valid locked/candidate comparisons. The third must complete all +structured stages, reuse the Codex process, and give the reviewer valid +evidence. A deterministic no-update regression test must separately prove that +a baseline-only package does not invent a missing candidate or downgrade. + +The live rehearsal remains report-only. It must not enable implementation, +create a branch, push, or open an autonomous upgrade PR. + +## Review protocol + +For each phase: + +1. Review only `base...head` for that PR. +2. Check phase acceptance criteria and tests. +3. Run an ordinary correctness, compatibility, and missing-test review. +4. Do not run a security diff or security scan. +5. Fix material findings on the same phase branch and repeat the targeted + review before publishing the next layer. +6. Verify live GitHub base/head topology after publication. + +## Compatibility and risk controls + +- The public `agent_runtime_kit` API is unchanged. +- Candidate inspection stays opt-in and credential-scrubbed. +- The new `incomplete` value appears only in generated example artifacts; any + downstream parser that assumed three status values must be updated in PR 3. +- No vendor dependency ranges or lockfile versions change in this repair stack. +- No scheduled automation is added. +- No direct vendor model API calls are added. +- Existing open PRs #19 and #40 are not used as stack bases because both carry + stale ancestry and broader changes. Their eventual disposition is separate + from this repair stack. + +## Completion definition + +The stack is complete when all five PRs are published with correct bases, each +phase has passed its targeted tests and non-security review, the top branch +passes the full verification matrix, and a Codex-backed report-only rehearsal +from a runtime-specific environment produces truthful actionable evidence. From 86ec7679f879dd0bbc72800c62c7313cba2693d0 Mon Sep 17 00:00:00 2001 From: Eloi Date: Tue, 28 Jul 2026 19:54:29 +0200 Subject: [PATCH 6/9] Observe locked SDK baselines accurately (#52) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Summary - treat `uv.lock` as the authoritative SDK baseline during explicit candidate inspection - isolate locked baselines when the matching SDK is absent or the ambient environment has drifted - keep non-opt-in runs honest by recording the actual ambient version instead of stamping it as locked - convert isolated API snapshot setup, install, execution, timeout, and malformed-output failures into bounded evidence ## Stack - Depends on #51 - This is layer 2 of 5 in the SDK evolution demo reliability stack ## Review boundary This layer repairs baseline provenance and snapshot collection only. Behavior-probe subprocess failures and the final `fail > incomplete > changed > pass` assessment are intentionally implemented together in the next layer so incomplete evidence cannot temporarily appear green. ## Validation - `PYTHONPATH=. .venv/bin/python -m pytest -q` — 421 passed, 12 skipped - `.venv/bin/ruff check .` - `.venv/bin/mypy` - `uv lock --check` - `git diff --check` --- examples/sdk_evolution_agent/behavior.py | 4 +- examples/sdk_evolution_agent/cli.py | 4 +- examples/sdk_evolution_agent/snapshots.py | 134 +++++--- tests/test_sdk_evolution_agent.py | 362 +++++++++++++++++++++- 4 files changed, 460 insertions(+), 44 deletions(-) diff --git a/examples/sdk_evolution_agent/behavior.py b/examples/sdk_evolution_agent/behavior.py index 96317b4..b45a264 100644 --- a/examples/sdk_evolution_agent/behavior.py +++ b/examples/sdk_evolution_agent/behavior.py @@ -41,16 +41,14 @@ def collect_behavior_evidence( continue locked_version = _string_or_none(package.get("locked_version")) installed_version = _string_or_none(package.get("installed_version")) - current_version = locked_version or installed_version if ( inspect_candidates and locked_version - and installed_version and locked_version != installed_version ): results.extend(probe_candidate_in_venv(name, locked_version, scope="current-baseline")) else: - results.extend(probe_current_package(name, version=current_version)) + results.extend(probe_current_package(name, version=installed_version)) candidate = update_versions.get(name) if candidate: if inspect_candidates: diff --git a/examples/sdk_evolution_agent/cli.py b/examples/sdk_evolution_agent/cli.py index b776339..87461df 100644 --- a/examples/sdk_evolution_agent/cli.py +++ b/examples/sdk_evolution_agent/cli.py @@ -526,10 +526,10 @@ def _collect_snapshots(evidence: dict[str, Any], *, inspect_candidates: bool = F locked = package.get("locked_version") installed = package.get("installed_version") baseline = locked or installed - if inspect_candidates and locked and installed and locked != installed: + if inspect_candidates and locked and locked != installed: snapshots.append(snapshot_candidate_in_venv(name, str(locked))) else: - snapshots.append(snapshot_current_api(name, version=baseline)) + snapshots.append(snapshot_current_api(name, version=installed)) if not inspect_candidates: continue candidate = update_versions.get(name) diff --git a/examples/sdk_evolution_agent/snapshots.py b/examples/sdk_evolution_agent/snapshots.py index f198536..31f80e8 100644 --- a/examples/sdk_evolution_agent/snapshots.py +++ b/examples/sdk_evolution_agent/snapshots.py @@ -128,51 +128,111 @@ def snapshot_candidate_in_venv( """Inspect a candidate version in an isolated temporary virtualenv.""" module_name = DEFAULT_MODULES.get(package, package.replace("-", "_")) - with tempfile.TemporaryDirectory(prefix="ark-sdk-snapshot-") as directory: - venv = Path(directory) / ".venv" - # Scrub the environment for every subprocess that touches freshly downloaded - # upstream code: give it a throwaway HOME and only PATH, so a malicious or - # buggy candidate package cannot read the caller's credentials/config. - env = isolated_env(Path(directory)) - subprocess.run( - (python, "-m", "venv", str(venv)), check=True, timeout=timeout, env=env + step = "virtual environment creation" + try: + with tempfile.TemporaryDirectory(prefix="ark-sdk-snapshot-") as directory: + venv = Path(directory) / ".venv" + # Scrub the environment for every subprocess that touches freshly downloaded + # upstream code: give it a throwaway HOME and only PATH, so a malicious or + # buggy candidate package cannot read the caller's credentials/config. + env = isolated_env(Path(directory)) + subprocess.run( + (python, "-m", "venv", str(venv)), + check=True, + timeout=timeout, + env=env, + ) + bin_dir = "Scripts" if sys.platform == "win32" else "bin" + venv_python = venv / bin_dir / "python" + step = "package installation" + subprocess.run( + (str(venv_python), "-m", "pip", "install", f"{package}=={version}"), + check=True, + text=True, + capture_output=True, + timeout=timeout, + env=env, + ) + step = "snapshot execution" + completed = subprocess.run( + ( + str(venv_python), + "-c", + _SNAPSHOT_SCRIPT, + package, + version, + module_name, + ), + check=True, + text=True, + capture_output=True, + timeout=timeout, + env=env, + ) + except subprocess.TimeoutExpired as exc: + return _failed_isolated_snapshot( + package, + version, + module_name, + f"{step} timed out after {exc.timeout}s", ) - bin_dir = "Scripts" if sys.platform == "win32" else "bin" - venv_python = venv / bin_dir / "python" - subprocess.run( - (str(venv_python), "-m", "pip", "install", f"{package}=={version}"), - check=True, - text=True, - capture_output=True, - timeout=timeout, - env=env, + except subprocess.CalledProcessError as exc: + detail = _bounded_failure_detail(exc.stderr or exc.stdout or str(exc)) + return _failed_isolated_snapshot( + package, + version, + module_name, + f"{step} failed: {detail}", ) - completed = subprocess.run( - ( - str(venv_python), - "-c", - _SNAPSHOT_SCRIPT, - package, - version, - module_name, - ), - check=True, - text=True, - capture_output=True, - timeout=timeout, - env=env, + except OSError as exc: + return _failed_isolated_snapshot( + package, + version, + module_name, + f"{step} failed: {_bounded_failure_detail(exc)}", ) - raw = json.loads(completed.stdout) + + try: + raw = json.loads(completed.stdout) + return ApiSnapshot( + package=raw["package"], + version=raw["version"], + module=raw["module"], + members=tuple(ApiMember(**item) for item in raw.get("members", ())), + import_error=raw.get("import_error"), + source="isolated-venv", + ) + except (json.JSONDecodeError, KeyError, TypeError, ValueError) as exc: + output = _bounded_failure_detail(completed.stdout) + detail = f"malformed snapshot output: {exc}" + if output: + detail += f"; stdout={output}" + return _failed_isolated_snapshot(package, version, module_name, detail) + + +def _failed_isolated_snapshot( + package: str, + version: str, + module_name: str, + error: str, +) -> ApiSnapshot: return ApiSnapshot( - package=raw["package"], - version=raw["version"], - module=raw["module"], - members=tuple(ApiMember(**item) for item in raw.get("members", ())), - import_error=raw.get("import_error"), + package=package, + version=version, + module=module_name, + import_error=_bounded_failure_detail(error, limit=560), source="isolated-venv", ) +def _bounded_failure_detail(value: object, *, limit: int = 480) -> str: + if isinstance(value, bytes): + text = value.decode(errors="replace") + else: + text = str(value) + return " ".join(text.split())[:limit] + + def _member_kind(value: Any) -> str: if inspect.isclass(value): return "class" diff --git a/tests/test_sdk_evolution_agent.py b/tests/test_sdk_evolution_agent.py index 81b8817..b3dfec2 100644 --- a/tests/test_sdk_evolution_agent.py +++ b/tests/test_sdk_evolution_agent.py @@ -2,6 +2,7 @@ import os import stat +import subprocess import sys import types from pathlib import Path @@ -53,7 +54,11 @@ SchemaValidationError, validate_mapping, ) -from examples.sdk_evolution_agent.snapshots import diff_snapshots, snapshot_current_api +from examples.sdk_evolution_agent.snapshots import ( + diff_snapshots, + snapshot_candidate_in_venv, + snapshot_current_api, +) from examples.sdk_evolution_agent.stages import ( SDK_EVOLUTION_CODEX_HOME, SDK_EVOLUTION_CODEX_MODEL, @@ -492,6 +497,156 @@ def isolated(package: str, version: str, *, scope: str = "candidate"): assert behavior["diffs"][0].severity == "changed" +def test_behavior_evidence_uses_locked_baseline_when_sdk_not_installed( + monkeypatch: pytest.MonkeyPatch, +) -> None: + calls: list[tuple[str, str, str]] = [] + + def isolated(package: str, version: str, *, scope: str = "candidate"): + calls.append((package, version, scope)) + return (_probe(package, version, scope, "pass", {"scope": scope}),) + + def current(package: str, *, version: str | None = None): + raise AssertionError(f"ambient probe must not represent {package} {version}") + + monkeypatch.setattr( + "examples.sdk_evolution_agent.behavior.probe_candidate_in_venv", + isolated, + ) + monkeypatch.setattr( + "examples.sdk_evolution_agent.behavior.probe_current_package", + current, + ) + + collect_behavior_evidence( + [ + { + "name": "claude-agent-sdk", + "locked_version": "0.2.106", + "installed_version": None, + } + ], + {"claude-agent-sdk": "0.2.110"}, + inspect_candidates=True, + ) + + assert calls == [ + ("claude-agent-sdk", "0.2.106", "current-baseline"), + ("claude-agent-sdk", "0.2.110", "candidate"), + ] + + +def test_behavior_evidence_without_opt_in_reports_actual_ambient_version( + monkeypatch: pytest.MonkeyPatch, +) -> None: + calls: list[tuple[str, str | None]] = [] + + def current(package: str, *, version: str | None = None): + calls.append((package, version)) + return (_probe(package, version, "current-environment", "fail", {}),) + + def isolated(package: str, version: str, *, scope: str = "candidate"): + raise AssertionError(f"isolated probe must not run for {package} {version} {scope}") + + monkeypatch.setattr( + "examples.sdk_evolution_agent.behavior.probe_current_package", + current, + ) + monkeypatch.setattr( + "examples.sdk_evolution_agent.behavior.probe_candidate_in_venv", + isolated, + ) + + behavior = collect_behavior_evidence( + [ + { + "name": "claude-agent-sdk", + "locked_version": "0.2.106", + "installed_version": None, + } + ], + {"claude-agent-sdk": "0.2.110"}, + ) + + assert calls == [("claude-agent-sdk", None)] + assert [result.status for result in behavior["results"]] == ["fail", "skip"] + + +def test_behavior_evidence_without_opt_in_uses_drifted_installed_version( + monkeypatch: pytest.MonkeyPatch, +) -> None: + calls: list[tuple[str, str | None]] = [] + + def current(package: str, *, version: str | None = None): + calls.append((package, version)) + return (_probe(package, version, "current-environment", "pass", {}),) + + def isolated(package: str, version: str, *, scope: str = "candidate"): + raise AssertionError(f"isolated probe must not run for {package} {version} {scope}") + + monkeypatch.setattr( + "examples.sdk_evolution_agent.behavior.probe_current_package", + current, + ) + monkeypatch.setattr( + "examples.sdk_evolution_agent.behavior.probe_candidate_in_venv", + isolated, + ) + + collect_behavior_evidence( + [ + { + "name": "claude-agent-sdk", + "locked_version": "0.2.96", + "installed_version": "0.2.106", + } + ], + {"claude-agent-sdk": "0.2.110"}, + ) + + assert calls == [("claude-agent-sdk", "0.2.106")] + + +def test_behavior_evidence_reuses_matching_ambient_baseline( + monkeypatch: pytest.MonkeyPatch, +) -> None: + calls: list[tuple[str, str, str | None]] = [] + + def current(package: str, *, version: str | None = None): + calls.append(("current", package, version)) + return (_probe(package, version, "current-environment", "pass", {}),) + + def isolated(package: str, version: str, *, scope: str = "candidate"): + calls.append((scope, package, version)) + return (_probe(package, version, scope, "pass", {}),) + + monkeypatch.setattr( + "examples.sdk_evolution_agent.behavior.probe_current_package", + current, + ) + monkeypatch.setattr( + "examples.sdk_evolution_agent.behavior.probe_candidate_in_venv", + isolated, + ) + + collect_behavior_evidence( + [ + { + "name": "claude-agent-sdk", + "locked_version": "0.2.106", + "installed_version": "0.2.106", + } + ], + {"claude-agent-sdk": "0.2.110"}, + inspect_candidates=True, + ) + + assert calls == [ + ("current", "claude-agent-sdk", "0.2.106"), + ("candidate", "claude-agent-sdk", "0.2.110"), + ] + + def test_behavior_candidate_probes_are_opt_in(monkeypatch: pytest.MonkeyPatch) -> None: # Probing a candidate pip-installs and imports freshly downloaded upstream # code; without --inspect-candidates that must never happen, and the gap @@ -646,6 +801,111 @@ def run_new(value: str, *, verbose: bool = False) -> str: assert diff.changed == ("run",) +@pytest.mark.parametrize( + ("failure_call", "detail"), + [ + (1, "venv creation failed"), + (2, "package install failed"), + (3, "snapshot execution failed"), + ], +) +def test_candidate_snapshot_records_subprocess_failures( + monkeypatch: pytest.MonkeyPatch, + failure_call: int, + detail: str, +) -> None: + calls = 0 + + def fake_run(args: Any, **kwargs: Any) -> Any: + nonlocal calls + calls += 1 + if calls == failure_call: + raise subprocess.CalledProcessError( + returncode=1, + cmd=args, + stderr=(detail + " ") * 100, + ) + return types.SimpleNamespace(stdout="{}", stderr="", returncode=0) + + monkeypatch.setattr("examples.sdk_evolution_agent.snapshots.subprocess.run", fake_run) + + snapshot = snapshot_candidate_in_venv("claude-agent-sdk", "0.2.110") + + assert snapshot.package == "claude-agent-sdk" + assert snapshot.version == "0.2.110" + assert snapshot.source == "isolated-venv" + assert snapshot.import_error is not None + assert detail in snapshot.import_error + assert len(snapshot.import_error) <= 600 + + +def test_candidate_snapshot_records_timeout(monkeypatch: pytest.MonkeyPatch) -> None: + def fake_run(args: Any, **kwargs: Any) -> Any: + raise subprocess.TimeoutExpired(cmd=args, timeout=kwargs["timeout"]) + + monkeypatch.setattr("examples.sdk_evolution_agent.snapshots.subprocess.run", fake_run) + + snapshot = snapshot_candidate_in_venv("claude-agent-sdk", "0.2.110", timeout=7) + + assert snapshot.import_error is not None + assert "timed out after 7s" in snapshot.import_error + assert snapshot.source == "isolated-venv" + + +def test_candidate_snapshot_records_malformed_output(monkeypatch: pytest.MonkeyPatch) -> None: + calls = 0 + + def fake_run(args: Any, **kwargs: Any) -> Any: + nonlocal calls + calls += 1 + return types.SimpleNamespace( + stdout="not-json" if calls == 3 else "", + stderr="", + returncode=0, + ) + + monkeypatch.setattr("examples.sdk_evolution_agent.snapshots.subprocess.run", fake_run) + + snapshot = snapshot_candidate_in_venv("claude-agent-sdk", "0.2.110") + + assert snapshot.import_error is not None + assert "malformed snapshot output" in snapshot.import_error + assert snapshot.source == "isolated-venv" + + +def test_candidate_snapshot_scrubs_subprocess_env( + monkeypatch: pytest.MonkeyPatch, +) -> None: + captured_envs: list[dict[str, str] | None] = [] + calls = 0 + + def fake_run(args: Any, **kwargs: Any) -> Any: + nonlocal calls + calls += 1 + captured_envs.append(kwargs.get("env")) + payload = ( + '{"package":"claude-agent-sdk","version":"0.2.110",' + '"module":"claude_agent_sdk","members":[],"import_error":null}' + ) + return types.SimpleNamespace( + stdout=payload if calls == 3 else "", + stderr="", + returncode=0, + ) + + monkeypatch.setenv("ARK_FAKE_SECRET", "not-for-snapshots") + monkeypatch.setattr("examples.sdk_evolution_agent.snapshots.subprocess.run", fake_run) + + snapshot = snapshot_candidate_in_venv("claude-agent-sdk", "0.2.110") + + assert snapshot.import_error is None + assert len(captured_envs) == 3 + for env in captured_envs: + assert env is not None + assert "ARK_FAKE_SECRET" not in env + assert env.get("HOME") != os.environ.get("HOME") + + def test_parse_args_candidate_inspection_is_opt_in() -> None: # Candidate inspection pip-installs+imports upstream code, so it is off by # default and only enabled with the explicit flag. @@ -803,6 +1063,56 @@ def isolated_snapshot(package: str, version: str) -> ApiSnapshot: ] +def test_collect_snapshots_uses_locked_baseline_when_sdk_not_installed( + monkeypatch: pytest.MonkeyPatch, +) -> None: + calls: list[tuple[str, str, str | None]] = [] + + def current_snapshot(package: str, *, version: str | None = None) -> ApiSnapshot: + raise AssertionError(f"ambient snapshot must not represent {package} {version}") + + def isolated_snapshot(package: str, version: str) -> ApiSnapshot: + calls.append(("isolated", package, version)) + return ApiSnapshot( + package=package, + version=version, + module=package.replace("-", "_"), + source="isolated-venv", + ) + + monkeypatch.setattr( + "examples.sdk_evolution_agent.cli.snapshot_current_api", + current_snapshot, + ) + monkeypatch.setattr( + "examples.sdk_evolution_agent.cli.snapshot_candidate_in_venv", + isolated_snapshot, + ) + + _collect_snapshots( + { + "packages": [ + { + "name": "claude-agent-sdk", + "locked_version": "0.2.106", + "installed_version": None, + "latest_version": "0.2.110", + }, + ], + "refresh_preview": { + "stdout": "", + "stderr": "Update claude-agent-sdk v0.2.106 -> v0.2.110\n", + }, + }, + inspect_candidates=True, + ) + + assert calls == [ + ("isolated", "claude-agent-sdk", "0.2.106"), + ("isolated", "claude-agent-sdk", "0.2.110"), + ] + + def test_collect_snapshots_without_opt_in_never_installs_even_when_lock_drifted( monkeypatch: pytest.MonkeyPatch, ) -> None: @@ -846,7 +1156,55 @@ def isolated_snapshot(package: str, version: str) -> ApiSnapshot: inspect_candidates=False, ) - assert calls == [("current", "claude-agent-sdk", "0.2.96")] + assert calls == [("current", "claude-agent-sdk", "0.2.106")] + + +def test_collect_snapshots_without_opt_in_reports_missing_ambient_sdk( + monkeypatch: pytest.MonkeyPatch, +) -> None: + calls: list[tuple[str, str, str | None]] = [] + + def current_snapshot(package: str, *, version: str | None = None) -> ApiSnapshot: + calls.append(("current", package, version)) + return ApiSnapshot( + package=package, + version=version, + module=package.replace("-", "_"), + import_error="not installed", + ) + + def isolated_snapshot(package: str, version: str) -> ApiSnapshot: + raise AssertionError(f"isolated snapshot must not run for {package} {version}") + + monkeypatch.setattr( + "examples.sdk_evolution_agent.cli.snapshot_current_api", + current_snapshot, + ) + monkeypatch.setattr( + "examples.sdk_evolution_agent.cli.snapshot_candidate_in_venv", + isolated_snapshot, + ) + + snapshots = _collect_snapshots( + { + "packages": [ + { + "name": "claude-agent-sdk", + "locked_version": "0.2.106", + "installed_version": None, + "latest_version": "0.2.110", + } + ], + "refresh_preview": { + "stdout": "", + "stderr": "Update claude-agent-sdk v0.2.106 -> v0.2.110\n", + }, + }, + inspect_candidates=False, + ) + + assert calls == [("current", "claude-agent-sdk", None)] + assert snapshots[0].import_error == "not installed" def test_candidate_api_diff_guard_blocks_missing_update_diff() -> None: From e94cd5f70d0737e0ccb20ec0660ab16cccabee8d Mon Sep 17 00:00:00 2001 From: Eloi Date: Tue, 28 Jul 2026 19:57:30 +0200 Subject: [PATCH 7/9] Make SDK evolution evidence fail closed (#53) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Summary - parse resolver updates once as exact package/from/to transitions and reuse them across collection and deterministic gates - derive canonical locked-baseline expectations from collected package evidence, including no-update runs - contain isolated behavior setup, install, execution, timeout, and malformed-output failures as bounded structured evidence - compute `fail > incomplete > changed > pass` from raw probes, diffs, full probe pairing, and trusted expectations - recompute the canonical assessment for reports and implementation gates instead of trusting a cached headline - persist `behavior_summary.json`, include it in the current-state manifest, and surface snapshot/behavior errors and reasons in the Markdown report - require exact API-diff transitions and fail closed on missing, malformed, contradictory, incomplete, or breaking behavior evidence ## Stack - Depends on #52 - This is layer 3 of 5 in the SDK evolution demo reliability stack ## Review notes The layer received a two-pass correctness review. Review findings covered drifted no-update baselines, contradictory cached summaries, malformed nested details, failure precedence, partial probe sets, empty self-declared expectation scope, and real breaking-payload coverage; all were repaired and re-reviewed with no remaining actionable findings. ## Validation - `PYTHONPATH=. .venv/bin/python -m pytest -q` — 449 passed, 12 skipped - focused SDK evolution suite — 90 passed - `.venv/bin/ruff check .` - `.venv/bin/mypy` - `uv lock --check` - `git diff --check` Security-diff scans were intentionally not part of this phase review. --- docs/sdk-evolution-agent.md | 27 +- examples/sdk_evolution_agent/behavior.py | 758 +++++++++++++- examples/sdk_evolution_agent/cli.py | 23 +- examples/sdk_evolution_agent/collectors.py | 44 + examples/sdk_evolution_agent/current_state.py | 1 + examples/sdk_evolution_agent/report.py | 57 +- examples/sdk_evolution_agent/stages.py | 137 ++- tests/test_sdk_evolution_agent.py | 961 +++++++++++++++++- 8 files changed, 1870 insertions(+), 138 deletions(-) diff --git a/docs/sdk-evolution-agent.md b/docs/sdk-evolution-agent.md index 6d5f5aa..03b43c1 100644 --- a/docs/sdk-evolution-agent.md +++ b/docs/sdk-evolution-agent.md @@ -82,6 +82,7 @@ Each run writes a timestamped directory under `reports/sdk-evolution/` with: - `api_diffs.json` - `behavior_probes.json` - `behavior_diffs.json` +- `behavior_summary.json` - `current_state.json` - `direction_analysis.json` - `architecture_decision.json` @@ -135,8 +136,27 @@ removed / changed diff is valid; a missing diff object is not. Behavior probes intentionally separate observed SDK surface churn from adapter contract breakage. `behavior_probes.json` records fields and parameters seen in current and candidate packages, while `behavior_diffs.json` compares the -required adapter contract. Optional field changes remain visible in the report -and API diffs, but only breaking adapter-contract diffs block implementation. +required adapter contract. `behavior_summary.json` records the deterministic +assessment, counts, and reasons used by the implementation gate. Its status is +`pass` for complete unchanged evidence, `changed` for complete non-breaking +changes, `incomplete` for probe errors, skips, malformed records, or missing +exact-version comparisons, and `fail` for a failed required contract or a +breaking diff. A package with no resolver-selected update needs only a valid +current baseline at the locked version (or the observed installed version when +no lock entry exists); it does not need a candidate probe. An ambient SDK that +has drifted away from the lock therefore produces `incomplete`, not `pass`. +Optional field changes remain visible in the report and API diffs without being +treated as a contract failure. + +Resolver transitions are parsed once as exact `(package, from, to)` triples and +shared by snapshot collection, behavior assessment, and implementation gates. +An API diff or behavior comparison for a different package or version does not +satisfy the expected transition. Snapshot import and execution errors are also +shown explicitly in `report.md` instead of being hidden behind the snapshot +count. Behavior expectations come from deterministic package and resolver +evidence; a behavior payload cannot narrow that scope itself. The report and +`behavior_summary.json` recompute their assessment from raw probes and diffs, so +a stale or contradictory cached summary is never presented as the run result. ## Implementation Gates @@ -152,7 +172,8 @@ Implementation is still blocked when: - the reviewer rejects the evidence or design, - a resolver-selected update lacks a candidate API diff, - required release-note evidence could not be collected, -- candidate behavior probes show a breaking adapter-contract difference, +- behavior evidence is failed, incomplete, malformed, internally inconsistent, + or uses the wrong package/version transition, - required structured output or permission behavior is unsupported by the selected runtime, - recursive self-adaptation is required but no safe migration plan exists. diff --git a/examples/sdk_evolution_agent/behavior.py b/examples/sdk_evolution_agent/behavior.py index b45a264..5594972 100644 --- a/examples/sdk_evolution_agent/behavior.py +++ b/examples/sdk_evolution_agent/behavior.py @@ -12,17 +12,36 @@ import textwrap from collections.abc import Mapping, Sequence from pathlib import Path -from typing import Any +from typing import Any, cast +from examples.sdk_evolution_agent.collectors import ResolverTransition, parse_refresh_transitions from examples.sdk_evolution_agent.models import BehaviorDiff, BehaviorProbeResult from examples.sdk_evolution_agent.snapshots import DEFAULT_MODULES, isolated_env +_RESULT_STATUSES = frozenset({"pass", "fail", "skip"}) +_DIFF_SEVERITIES = frozenset({"none", "changed", "breaking"}) +_RESULT_SCOPES = frozenset( + {"current-baseline", "current-environment", "candidate", "isolated-venv"} +) +_DETAIL_STRING_SEQUENCE_FIELDS = frozenset( + { + "fields", + "missing", + "required_fields", + "required_run_params", + "required_start_params", + "run_params", + "start_params", + } +) + def collect_behavior_evidence( packages: Sequence[Mapping[str, object]], update_versions: Mapping[str, str], *, inspect_candidates: bool = False, + expected_transitions: Sequence[ResolverTransition] | None = None, ) -> dict[str, Any]: """Collect current/candidate behavior probes and compare them. @@ -41,11 +60,7 @@ def collect_behavior_evidence( continue locked_version = _string_or_none(package.get("locked_version")) installed_version = _string_or_none(package.get("installed_version")) - if ( - inspect_candidates - and locked_version - and locked_version != installed_version - ): + if inspect_candidates and locked_version and locked_version != installed_version: results.extend(probe_candidate_in_venv(name, locked_version, scope="current-baseline")) else: results.extend(probe_current_package(name, version=installed_version)) @@ -56,11 +71,19 @@ def collect_behavior_evidence( else: results.append(_skipped_candidate_probe(name, candidate)) diffs = diff_behavior_results(results) - return { + transitions = tuple( + expected_transitions + if expected_transitions is not None + else _expected_transitions(packages, update_versions) + ) + expectations = build_behavior_expectations(packages, transitions) + payload: dict[str, Any] = { "results": [result for result in results], "diffs": [diff for diff in diffs], - "summary": summarize_behavior(diffs), + **expectations, } + payload["summary"] = assess_behavior_payload(payload, expectations=expectations) + return payload def probe_current_package( @@ -83,34 +106,80 @@ def probe_candidate_in_venv( ) -> tuple[BehaviorProbeResult, ...]: """Run behavior probes against a candidate package in an isolated virtualenv.""" - with tempfile.TemporaryDirectory(prefix="ark-sdk-behavior-") as directory: - venv = Path(directory) / ".venv" - # Scrub the environment for every subprocess that touches freshly - # downloaded upstream code (same scrub as the API snapshots): a - # throwaway HOME and only PATH, so a malicious or buggy candidate - # package cannot read the caller's credentials/config. - env = isolated_env(Path(directory)) - subprocess.run((python, "-m", "venv", str(venv)), check=True, timeout=timeout, env=env) - bin_dir = "Scripts" if sys.platform == "win32" else "bin" - venv_python = venv / bin_dir / "python" - subprocess.run( - (str(venv_python), "-m", "pip", "install", f"{package}=={version}"), - check=True, - text=True, - capture_output=True, - timeout=timeout, - env=env, + step = "virtual-environment-creation" + try: + with tempfile.TemporaryDirectory(prefix="ark-sdk-behavior-") as directory: + venv = Path(directory) / ".venv" + # Scrub the environment for every subprocess that touches freshly + # downloaded upstream code (same scrub as the API snapshots): a + # throwaway HOME and only PATH, so a malicious or buggy candidate + # package cannot read the caller's credentials/config. + env = isolated_env(Path(directory)) + subprocess.run( + (python, "-m", "venv", str(venv)), + check=True, + text=True, + capture_output=True, + timeout=timeout, + env=env, + ) + bin_dir = "Scripts" if sys.platform == "win32" else "bin" + venv_python = venv / bin_dir / "python" + step = "package-installation" + subprocess.run( + (str(venv_python), "-m", "pip", "install", f"{package}=={version}"), + check=True, + text=True, + capture_output=True, + timeout=timeout, + env=env, + ) + step = "probe-execution" + completed = subprocess.run( + (str(venv_python), "-c", _PROBE_SCRIPT, package, version, scope), + check=True, + text=True, + capture_output=True, + timeout=timeout, + env=env, + ) + except subprocess.TimeoutExpired as exc: + return ( + _probe_execution_failure( + package, + version, + scope, + step, + f"timed out after {exc.timeout}s", + ), ) - completed = subprocess.run( - (str(venv_python), "-c", _PROBE_SCRIPT, package, version, scope), - check=True, - text=True, - capture_output=True, - timeout=timeout, - env=env, + except subprocess.CalledProcessError as exc: + detail = _bounded_text(exc.stderr or exc.stdout or str(exc)) + return (_probe_execution_failure(package, version, scope, step, detail),) + except OSError as exc: + return (_probe_execution_failure(package, version, scope, step, exc),) + + try: + return _parse_probe_output( + completed.stdout, + package=package, + version=version, + scope=scope, + ) + except (json.JSONDecodeError, KeyError, TypeError, ValueError) as exc: + output = _bounded_text(completed.stdout) + detail = f"malformed probe output: {exc}" + if output: + detail += f"; stdout={output}" + return ( + _probe_execution_failure( + package, + version, + scope, + "probe-output-validation", + detail, + ), ) - raw = json.loads(completed.stdout) - return tuple(BehaviorProbeResult(**item) for item in raw) def diff_behavior_results(results: Sequence[BehaviorProbeResult]) -> tuple[BehaviorDiff, ...]: @@ -125,9 +194,14 @@ def diff_behavior_results(results: Sequence[BehaviorProbeResult]) -> tuple[Behav after = scopes.get("candidate") or scopes.get("isolated-venv") if before is None or after is None: continue - if (before.status == "skip") != (after.status == "skip"): - # One side was not probed (candidate installs are opt-in): absence - # of evidence is not a behavior change and must not read as one. + if ( + before.status == "skip" + or after.status == "skip" + or _is_probe_execution_error(before) + or _is_probe_execution_error(after) + ): + # Missing or failed execution is incomplete evidence, not an observed + # behavior change. The raw-result assessment preserves that state. continue if before.status == after.status and _contract_details(before) == _contract_details(after): severity = "none" @@ -156,19 +230,525 @@ def diff_behavior_results(results: Sequence[BehaviorProbeResult]) -> tuple[Behav return tuple(diffs) -def summarize_behavior(diffs: Sequence[BehaviorDiff]) -> dict[str, Any]: - """Return a compact behavior summary for reports and gates.""" +def build_behavior_expectations( + packages: Sequence[Mapping[str, object]], + transitions: Sequence[ResolverTransition], +) -> dict[str, Any]: + """Build canonical behavior expectations from deterministic package evidence.""" + + expected_packages: list[str] = [] + expected_baselines: dict[str, str | None] = {} + for package in packages: + name = str(package.get("name") or "") + if not name or name in expected_baselines: + continue + locked_version = _string_or_none(package.get("locked_version")) + installed_version = _string_or_none(package.get("installed_version")) + expected_packages.append(name) + expected_baselines[name] = ( + locked_version if locked_version is not None else installed_version + ) + return { + "expected_packages": expected_packages, + "expected_transitions": [ + { + "package": transition.package, + "from_version": transition.from_version, + "to_version": transition.to_version, + } + for transition in sorted(set(transitions)) + ], + "expected_baselines": expected_baselines, + } + + +def behavior_expectations_from_evidence(evidence: Mapping[str, object]) -> dict[str, Any]: + """Derive authoritative behavior expectations from the deterministic evidence bundle.""" + + raw_packages = evidence.get("packages") + packages: list[Mapping[str, object]] = [] + issues: list[str] = [] + if not _is_sequence_payload(raw_packages): + issues.append("deterministic evidence packages must be an array") + else: + for index, package in enumerate(cast(Sequence[object], raw_packages)): + if not isinstance(package, Mapping): + issues.append(f"deterministic evidence package {index} must be an object") + continue + name = package.get("name") + if not isinstance(name, str) or not name: + issues.append(f"deterministic evidence package {index} must have a non-empty name") + continue + packages.append(package) + expectations = build_behavior_expectations(packages, parse_refresh_transitions(evidence)) + expectations["expectation_issues"] = issues + return expectations + + +def assess_behavior_payload( + behavior: Mapping[str, object], + *, + expectations: Mapping[str, object] | None = None, +) -> dict[str, Any]: + """Recompute the canonical assessment from raw evidence and trusted expectations.""" + + context = expectations if expectations is not None else behavior + summary = summarize_behavior( + behavior.get("results"), + behavior.get("diffs"), + expected_packages=context.get("expected_packages"), + expected_transitions=context.get("expected_transitions"), + expected_baselines=context.get("expected_baselines"), + ) + if expectations is None: + return summary + + raw_expectation_issues = context.get("expectation_issues", []) + issues = ( + list(raw_expectation_issues) + if _is_sequence_payload(raw_expectation_issues) + and all(isinstance(issue, str) for issue in cast(Sequence[object], raw_expectation_issues)) + else ["deterministic behavior expectation issues are malformed"] + ) + issues.extend( + [ + f"behavior payload {key} contradicts deterministic evidence" + for key in ("expected_packages", "expected_transitions", "expected_baselines") + if behavior.get(key) != expectations.get(key) + ] + ) + if not issues: + return summary + summary["malformed_count"] = int(summary["malformed_count"]) + len(issues) + summary["reasons"] = list(dict.fromkeys([*summary["reasons"], *issues])) + if summary["status"] != "fail": + summary["status"] = "incomplete" + return summary + + +def summarize_behavior( + results: object, + diffs: object, + *, + expected_packages: object, + expected_transitions: object, + expected_baselines: object, +) -> dict[str, Any]: + """Assess raw probes and diffs without collapsing missing evidence into pass.""" + + reasons: list[str] = [] + malformed_count = 0 + + packages: list[str] = [] + if not _is_sequence_payload(expected_packages): + malformed_count += 1 + reasons.append("expected_packages must be an array") + else: + for index, package in enumerate(cast(Sequence[object], expected_packages)): + if not isinstance(package, str) or not package: + malformed_count += 1 + reasons.append(f"expected package {index} must be a non-empty string") + elif package in packages: + malformed_count += 1 + reasons.append(f"expected package {package} is duplicated") + else: + packages.append(package) + + baselines: dict[str, str | None] = {} + if not isinstance(expected_baselines, Mapping): + malformed_count += 1 + reasons.append("expected_baselines must be an object") + else: + for package, version in expected_baselines.items(): + if not isinstance(package, str) or not package: + malformed_count += 1 + reasons.append("expected baseline package names must be non-empty strings") + continue + if version is not None and (not isinstance(version, str) or not version): + malformed_count += 1 + reasons.append( + f"expected baseline for {package} must be a non-empty string or null" + ) + continue + baselines[package] = version + + package_set = set(packages) + for package in sorted(package_set - baselines.keys()): + malformed_count += 1 + reasons.append(f"expected_baselines is missing {package}") + for package in sorted(baselines.keys() - package_set): + malformed_count += 1 + reasons.append(f"expected_baselines contains unexpected package {package}") + + normalized_results: list[BehaviorProbeResult] = [] + if not _is_sequence_payload(results): + malformed_count += 1 + reasons.append("behavior results must be an array") + else: + for index, item in enumerate(cast(Sequence[object], results)): + try: + normalized_results.append(_coerce_probe_result(item)) + except (KeyError, TypeError, ValueError) as exc: + malformed_count += 1 + reasons.append(f"behavior result {index} is malformed: {_bounded_text(exc)}") + + normalized_diffs: list[BehaviorDiff] = [] + if not _is_sequence_payload(diffs): + malformed_count += 1 + reasons.append("behavior diffs must be an array") + else: + for index, item in enumerate(cast(Sequence[object], diffs)): + try: + normalized_diffs.append(_coerce_behavior_diff(item)) + except (KeyError, TypeError, ValueError) as exc: + malformed_count += 1 + reasons.append(f"behavior diff {index} is malformed: {_bounded_text(exc)}") + + expected_diffs = diff_behavior_results(normalized_results) + if sorted(normalized_diffs, key=_behavior_diff_key) != sorted( + expected_diffs, key=_behavior_diff_key + ): + malformed_count += 1 + reasons.append("behavior diffs contradict raw probe results") + + contract_failures: list[BehaviorProbeResult] = [] + probe_errors: list[BehaviorProbeResult] = [] + skipped: list[BehaviorProbeResult] = [] + for result in normalized_results: + if result.status == "skip": + skipped.append(result) + reasons.append(_probe_reason(result, "skipped")) + elif result.status == "fail" and _is_probe_execution_error(result): + probe_errors.append(result) + reasons.append(_probe_reason(result, "could not execute")) + elif result.status == "fail": + contract_failures.append(result) + reasons.append(_probe_reason(result, "failed the required contract")) + + breaking = [diff for diff in expected_diffs if diff.severity == "breaking"] + changed = [diff for diff in expected_diffs if diff.severity == "changed"] + unchanged = [diff for diff in expected_diffs if diff.severity == "none"] + for diff in breaking: + reasons.append( + f"{diff.package}:{diff.probe} {diff.from_version} -> {diff.to_version} " + f"is breaking: {_bounded_text(diff.summary)}" + ) + + transitions: list[ResolverTransition] = [] + if not _is_sequence_payload(expected_transitions): + malformed_count += 1 + reasons.append("expected_transitions must be an array") + else: + for index, item in enumerate(cast(Sequence[object], expected_transitions)): + try: + transition = _coerce_transition(item) + except (KeyError, TypeError, ValueError) as exc: + malformed_count += 1 + reasons.append(f"expected transition {index} is malformed: {_bounded_text(exc)}") + continue + transitions.append(transition) + if transition.package not in package_set: + malformed_count += 1 + reasons.append( + f"expected transition contains unexpected package {transition.package}" + ) + elif ( + transition.package in baselines + and baselines[transition.package] != transition.from_version + ): + malformed_count += 1 + reasons.append( + f"{transition.package} transition baseline {transition.from_version} " + f"contradicts expected baseline {baselines[transition.package]}" + ) + + missing_comparison_count = 0 + transition_packages = {transition.package for transition in transitions} + for package in sorted(package_set - transition_packages): + observations = [ + result + for result in normalized_results + if result.package == package + and result.scope in {"current-baseline", "current-environment"} + ] + if not observations: + missing_comparison_count += 1 + reasons.append(f"{package} has no observed current baseline") + continue + if package not in baselines: + continue + expected_version = baselines[package] + if not any( + result.package == package + and result.version == expected_version + and result.scope in {"current-baseline", "current-environment"} + for result in observations + ): + missing_comparison_count += 1 + observed_versions = sorted( + { + result.version if result.version is not None else "" + for result in observations + } + ) + expected_label = expected_version if expected_version is not None else "" + reasons.append( + f"{package} current observation does not match expected baseline " + f"{expected_label} (observed: {', '.join(observed_versions)})" + ) + + for transition in sorted(set(transitions)): + before = [ + result + for result in normalized_results + if result.package == transition.package + and result.version == transition.from_version + and result.scope in {"current-baseline", "current-environment"} + ] + after = [ + result + for result in normalized_results + if result.package == transition.package + and result.version == transition.to_version + and result.scope in {"candidate", "isolated-venv"} + ] + before_probes = {result.probe for result in before} + after_probes = {result.probe for result in after} + compared_probes = { + diff.probe + for diff in expected_diffs + if diff.package == transition.package + and diff.from_version == transition.from_version + and diff.to_version == transition.to_version + } + if not before_probes or before_probes != after_probes or compared_probes != before_probes: + missing_comparison_count += 1 + reasons.append( + f"{transition.package} lacks a paired behavior comparison for " + f"{transition.from_version} -> {transition.to_version}" + ) + + if contract_failures or breaking: + status = "fail" + elif probe_errors or skipped or missing_comparison_count or malformed_count: + status = "incomplete" + elif changed: + status = "changed" + else: + status = "pass" - breaking = [diff for diff in diffs if diff.severity == "breaking"] - changed = [diff for diff in diffs if diff.severity == "changed"] return { + "status": status, "breaking_count": len(breaking), "changed_count": len(changed), - "unchanged_count": len([diff for diff in diffs if diff.severity == "none"]), - "status": "fail" if breaking else "changed" if changed else "pass", + "unchanged_count": len(unchanged), + "contract_failure_count": len(contract_failures), + "probe_error_count": len(probe_errors), + "skipped_count": len(skipped), + "missing_comparison_count": missing_comparison_count, + "malformed_count": malformed_count, + "reasons": list(dict.fromkeys(reasons)), } +def _is_sequence_payload(value: object) -> bool: + return isinstance(value, Sequence) and not isinstance(value, str | bytes | bytearray) + + +def _coerce_probe_result(item: object) -> BehaviorProbeResult: + if isinstance(item, BehaviorProbeResult): + result = item + elif isinstance(item, Mapping): + _require_string_fields(item, "package", "scope", "probe", "status", "summary") + version = item.get("version") + if version is not None and not isinstance(version, str): + raise TypeError("version must be a string or null") + details = item.get("details", {}) + if not isinstance(details, Mapping): + raise TypeError("details must be an object") + result = BehaviorProbeResult( + package=item["package"], + version=version, + scope=item["scope"], + probe=item["probe"], + status=item["status"], + summary=item["summary"], + details=dict(details), + ) + else: + raise TypeError("result must be an object") + if not result.package or not result.scope or not result.probe: + raise ValueError("package, scope, and probe must be non-empty") + if result.version is not None and not isinstance(result.version, str): + raise TypeError("version must be a string or null") + if result.scope not in _RESULT_SCOPES: + raise ValueError(f"unknown probe scope {result.scope!r}") + if result.status not in _RESULT_STATUSES: + raise ValueError(f"unknown probe status {result.status!r}") + if not isinstance(result.summary, str): + raise TypeError("summary must be a string") + if not isinstance(result.details, Mapping): + raise TypeError("details must be an object") + details = _normalize_probe_details(result.details) + return BehaviorProbeResult( + package=result.package, + version=result.version, + scope=result.scope, + probe=result.probe, + status=result.status, + summary=result.summary, + details=details, + ) + + +def _coerce_behavior_diff(item: object) -> BehaviorDiff: + if isinstance(item, BehaviorDiff): + diff = item + elif isinstance(item, Mapping): + _require_string_fields( + item, + "package", + "probe", + "severity", + "summary", + "before_status", + "after_status", + ) + from_version = item.get("from_version") + to_version = item.get("to_version") + if from_version is not None and not isinstance(from_version, str): + raise TypeError("from_version must be a string or null") + if to_version is not None and not isinstance(to_version, str): + raise TypeError("to_version must be a string or null") + diff = BehaviorDiff( + package=item["package"], + from_version=from_version, + to_version=to_version, + probe=item["probe"], + severity=item["severity"], + summary=item["summary"], + before_status=item["before_status"], + after_status=item["after_status"], + ) + else: + raise TypeError("diff must be an object") + if not diff.package or not diff.probe: + raise ValueError("diff package and probe must be non-empty") + if diff.from_version is not None and not isinstance(diff.from_version, str): + raise TypeError("diff from_version must be a string or null") + if diff.to_version is not None and not isinstance(diff.to_version, str): + raise TypeError("diff to_version must be a string or null") + if not isinstance(diff.summary, str): + raise TypeError("diff summary must be a string") + if diff.severity not in _DIFF_SEVERITIES: + raise ValueError(f"unknown diff severity {diff.severity!r}") + if diff.before_status not in _RESULT_STATUSES or diff.after_status not in _RESULT_STATUSES: + raise ValueError("diff contains an unknown probe status") + return diff + + +def _coerce_transition(item: ResolverTransition | Mapping[str, object]) -> ResolverTransition: + if isinstance(item, ResolverTransition): + transition = item + elif isinstance(item, Mapping): + _require_string_fields(item, "package", "from_version", "to_version") + transition = ResolverTransition( + package=cast(str, item["package"]), + from_version=cast(str, item["from_version"]), + to_version=cast(str, item["to_version"]), + ) + else: + raise TypeError("transition must be an object") + if not all( + isinstance(value, str) + for value in (transition.package, transition.from_version, transition.to_version) + ): + raise TypeError("transition fields must be strings") + if not transition.package or not transition.from_version or not transition.to_version: + raise ValueError("transition fields must be non-empty") + return transition + + +def _require_string_fields(item: Mapping[str, object], *fields: str) -> None: + for field in fields: + if field not in item: + raise KeyError(field) + if not isinstance(item[field], str): + raise TypeError(f"{field} must be a string") + + +def _normalize_probe_details(details: Mapping[str, object]) -> dict[str, Any]: + normalized = dict(details) + for field in _DETAIL_STRING_SEQUENCE_FIELDS & details.keys(): + value = details[field] + if not _is_sequence_payload(value) or not all( + isinstance(item, str) for item in cast(Sequence[object], value) + ): + raise TypeError(f"details.{field} must be an array of strings") + normalized[field] = list(cast(Sequence[str], value)) + if "error" in details and not isinstance(details["error"], str): + raise TypeError("details.error must be a string") + if "failure_step" in details and not isinstance(details["failure_step"], str): + raise TypeError("details.failure_step must be a string") + return normalized + + +def _behavior_diff_key(diff: BehaviorDiff) -> tuple[str, str, str, str, str, str, str, str]: + return ( + diff.package, + diff.from_version or "", + diff.to_version or "", + diff.probe, + diff.severity, + diff.summary, + diff.before_status, + diff.after_status, + ) + + +def _expected_transitions( + packages: Sequence[Mapping[str, object]], + update_versions: Mapping[str, str], +) -> tuple[ResolverTransition, ...]: + transitions: list[ResolverTransition] = [] + for package in packages: + name = str(package.get("name") or "") + candidate = update_versions.get(name) + if not name or not candidate: + continue + baseline = _string_or_none(package.get("locked_version")) or _string_or_none( + package.get("installed_version") + ) + transitions.append( + ResolverTransition( + package=name, + from_version=baseline or "", + to_version=str(candidate), + ) + ) + return tuple(transitions) + + +def _probe_reason(result: BehaviorProbeResult, classification: str) -> str: + return ( + f"{result.package}:{result.probe} {result.scope} {classification}: " + f"{_bounded_text(result.summary, limit=240)}" + ) + + +def _is_probe_execution_error(result: BehaviorProbeResult) -> bool: + return ( + result.status == "fail" + and "error" in result.details + and not _has_contract_failure_evidence(result) + ) + + +def _has_contract_failure_evidence(result: BehaviorProbeResult) -> bool: + missing = result.details.get("missing") + return bool(missing) and _is_sequence_payload(missing) + + def _probe_package( package: str, *, @@ -346,16 +926,28 @@ def _contract_details(result: BehaviorProbeResult) -> dict[str, Any]: details = result.details if "missing" not in details: return details - contract: dict[str, Any] = {"missing": sorted(details.get("missing") or [])} + contract: dict[str, Any] = {"missing": _normalized_contract_field(details.get("missing"))} if "required_fields" in details: - contract["required_fields"] = sorted(details.get("required_fields") or []) + contract["required_fields"] = _normalized_contract_field(details.get("required_fields")) if "required_run_params" in details: - contract["required_run_params"] = sorted(details.get("required_run_params") or []) + contract["required_run_params"] = _normalized_contract_field( + details.get("required_run_params") + ) if "required_start_params" in details: - contract["required_start_params"] = sorted(details.get("required_start_params") or []) + contract["required_start_params"] = _normalized_contract_field( + details.get("required_start_params") + ) return contract +def _normalized_contract_field(value: object) -> object: + if _is_sequence_payload(value) and all( + isinstance(item, str) for item in cast(Sequence[object], value) + ): + return sorted(cast(Sequence[str], value)) + return value + + def _failed( package: str, version: str | None, @@ -363,23 +955,81 @@ def _failed( probe: str, exc: Exception, ) -> BehaviorProbeResult: + error = _bounded_text(exc) return BehaviorProbeResult( package=package, version=version, scope=scope, probe=probe, status="fail", - summary=str(exc), - details={"error": str(exc)}, + summary=error, + details={"error": error, "failure_step": "adapter-probe"}, + ) + + +def _probe_execution_failure( + package: str, + version: str, + scope: str, + step: str, + error: object, +) -> BehaviorProbeResult: + detail = _bounded_text(error) + summary = _bounded_text(f"{step} failed: {detail}", limit=560) + return BehaviorProbeResult( + package=package, + version=version, + scope=scope, + probe=_probe_identity(package), + status="fail", + summary=summary, + details={"error": summary, "failure_step": step}, ) +def _parse_probe_output( + output: str, + *, + package: str, + version: str, + scope: str, +) -> tuple[BehaviorProbeResult, ...]: + raw = json.loads(output) + if not isinstance(raw, list) or not raw: + raise ValueError("probe output must be a non-empty array") + results: list[BehaviorProbeResult] = [] + for item in raw: + result = _coerce_probe_result(item) + if result.package != package: + raise ValueError( + f"probe package {result.package!r} does not match requested package {package!r}" + ) + if result.version != version: + raise ValueError( + f"probe version {result.version!r} does not match requested version {version!r}" + ) + if result.scope != scope: + raise ValueError( + f"probe scope {result.scope!r} does not match requested scope {scope!r}" + ) + results.append(result) + return tuple(results) + + +def _probe_identity(package: str) -> str: + if package == "openai-codex-cli-bin": + return "binary-distribution" + if package in {"claude-agent-sdk", "openai-codex", "google-antigravity"}: + return "adapter-contract" + return "package-import" + + def _skipped_candidate_probe(package: str, version: str) -> BehaviorProbeResult: return BehaviorProbeResult( package=package, version=version, scope="candidate", - probe="adapter-contract", + probe=_probe_identity(package), status="skip", summary=( "Candidate behavior probe skipped: probing a candidate installs and " @@ -390,6 +1040,14 @@ def _skipped_candidate_probe(package: str, version: str) -> BehaviorProbeResult: ) +def _bounded_text(value: object, *, limit: int = 480) -> str: + if isinstance(value, bytes): + text = value.decode(errors="replace") + else: + text = str(value) + return " ".join(text.split())[:limit] + + def _string_or_none(value: object) -> str | None: if value is None: return None diff --git a/examples/sdk_evolution_agent/cli.py b/examples/sdk_evolution_agent/cli.py index 87461df..2282fcc 100644 --- a/examples/sdk_evolution_agent/cli.py +++ b/examples/sdk_evolution_agent/cli.py @@ -3,7 +3,6 @@ from __future__ import annotations import argparse -import re from datetime import datetime, timezone from pathlib import Path from typing import Any @@ -14,6 +13,8 @@ CommandRunner, PypiClient, collect_evidence, + parse_refresh_transitions, + refresh_update_versions, run_lock_update, run_verification_commands, ) @@ -183,7 +184,8 @@ async def run_agent( pypi_client=pypi_client, command_runner=command_runner, ) - update_versions = _refresh_update_versions(evidence) + transitions = parse_refresh_transitions(evidence) + update_versions = refresh_update_versions(evidence) snapshots = _collect_snapshots(evidence, inspect_candidates=options.inspect_candidates) api_diffs = [to_jsonable(diff) for diff in diff_snapshot_groups(snapshots)] release_notes = [ @@ -195,6 +197,7 @@ async def run_agent( evidence.get("packages", []), update_versions, inspect_candidates=options.inspect_candidates, + expected_transitions=transitions, ) ) direction, architecture, review = await run_analysis_pipeline( @@ -517,7 +520,7 @@ def _collect_snapshots(evidence: dict[str, Any], *, inspect_candidates: bool = F # off, every snapshot uses the already-installed version via snapshot_current_api, # which imports nothing new. snapshots = [] - update_versions = _refresh_update_versions(evidence) + update_versions = refresh_update_versions(evidence) refresh_preview_seen = evidence.get("refresh_preview") is not None for package in evidence.get("packages", []): if not isinstance(package, dict): @@ -543,14 +546,6 @@ def _collect_snapshots(evidence: dict[str, Any], *, inspect_candidates: bool = F def _refresh_update_versions(evidence: dict[str, Any]) -> dict[str, str]: - preview = evidence.get("refresh_preview") - if not isinstance(preview, dict): - return {} - text = f"{preview.get('stdout') or ''}\n{preview.get('stderr') or ''}" - return { - package: version - for package, version in re.findall( - r"Update\s+([A-Za-z0-9_.-]+)\s+v\S+\s+->\s+v(\S+)", - text, - ) - } + """Compatibility wrapper around the shared exact transition parser.""" + + return refresh_update_versions(evidence) diff --git a/examples/sdk_evolution_agent/collectors.py b/examples/sdk_evolution_agent/collectors.py index 8dc135c..4dbce5d 100644 --- a/examples/sdk_evolution_agent/collectors.py +++ b/examples/sdk_evolution_agent/collectors.py @@ -10,6 +10,7 @@ import subprocess import urllib.request from collections.abc import Callable, Mapping, Sequence +from dataclasses import dataclass from pathlib import Path from typing import Any @@ -23,6 +24,22 @@ FRESHNESS_CUTOFF_ENV_VARS = ("UV_EXCLUDE_NEWER",) +_REFRESH_TRANSITION_RE = re.compile( + r"^[ \t]*Update[ \t]+(?P[A-Za-z0-9_.-]+)[ \t]+" + r"v(?P\S+)[ \t]+->[ \t]+v(?P\S+)[ \t]*$", + re.MULTILINE, +) + + +@dataclass(frozen=True, order=True) +class ResolverTransition: + """One exact package transition selected by the uv refresh preview.""" + + package: str + from_version: str + to_version: str + + PACKAGE_SOURCE_HINTS: dict[str, tuple[SourceRef, ...]] = { "claude-agent-sdk": ( SourceRef( @@ -268,6 +285,33 @@ def run_refresh_preview( ) +def parse_refresh_transitions(evidence: Mapping[str, Any]) -> tuple[ResolverTransition, ...]: + """Parse exact resolver transitions from refresh-preview stdout and stderr.""" + + preview = evidence.get("refresh_preview") + if not isinstance(preview, Mapping): + return () + text = f"{preview.get('stdout') or ''}\n{preview.get('stderr') or ''}" + transitions = { + ResolverTransition( + package=match.group("package"), + from_version=match.group("from_version"), + to_version=match.group("to_version"), + ) + for match in _REFRESH_TRANSITION_RE.finditer(text) + } + return tuple(sorted(transitions)) + + +def refresh_update_versions(evidence: Mapping[str, Any]) -> dict[str, str]: + """Return resolver-selected target versions keyed by exact package name.""" + + return { + transition.package: transition.to_version + for transition in parse_refresh_transitions(evidence) + } + + def run_lock_update( root: Path, packages: Sequence[str], diff --git a/examples/sdk_evolution_agent/current_state.py b/examples/sdk_evolution_agent/current_state.py index 9f37bae..7c91ae1 100644 --- a/examples/sdk_evolution_agent/current_state.py +++ b/examples/sdk_evolution_agent/current_state.py @@ -48,6 +48,7 @@ def _artifact_refs(report_root: Path, *, workspace: Path) -> dict[str, dict[str, "api_diffs.json", "behavior_probes.json", "behavior_diffs.json", + "behavior_summary.json", "direction_analysis.json", "architecture_decision.json", "implementation_summary.json", diff --git a/examples/sdk_evolution_agent/report.py b/examples/sdk_evolution_agent/report.py index ec26afa..9318295 100644 --- a/examples/sdk_evolution_agent/report.py +++ b/examples/sdk_evolution_agent/report.py @@ -6,6 +6,10 @@ from pathlib import Path from typing import Any +from examples.sdk_evolution_agent.behavior import ( + assess_behavior_payload, + behavior_expectations_from_evidence, +) from examples.sdk_evolution_agent.models import RunContext, to_jsonable @@ -44,6 +48,13 @@ def write_run_report( write_json(context.report_root / "api_diffs.json", api_diffs) write_json(context.report_root / "behavior_probes.json", behavior.get("results", [])) write_json(context.report_root / "behavior_diffs.json", behavior.get("diffs", [])) + write_json( + context.report_root / "behavior_summary.json", + assess_behavior_payload( + behavior, + expectations=behavior_expectations_from_evidence(evidence), + ), + ) write_json(context.report_root / "direction_analysis.json", direction) write_json(context.report_root / "architecture_decision.json", architecture) write_json(context.report_root / "implementation_summary.json", implementation) @@ -61,6 +72,7 @@ def write_run_report( render_markdown_report( config=config, evidence=evidence, + snapshots=snapshots, api_diffs=api_diffs, release_notes=release_notes, behavior=behavior, @@ -79,6 +91,7 @@ def render_markdown_report( *, config: dict[str, Any], evidence: dict[str, Any], + snapshots: list[dict[str, Any]], api_diffs: list[dict[str, Any]], release_notes: list[dict[str, Any]], behavior: dict[str, Any], @@ -115,8 +128,31 @@ def render_markdown_report( for item in release_notes if isinstance(item, dict) and item.get("to_version") ] - behavior_summary = behavior.get("summary") if isinstance(behavior, dict) else {} - behavior_diffs = behavior.get("diffs", []) if isinstance(behavior, dict) else [] + behavior_summary = assess_behavior_payload( + behavior, + expectations=behavior_expectations_from_evidence(evidence), + ) + behavior_diffs = behavior.get("diffs", []) + behavior_reasons = behavior_summary.get("reasons", []) + snapshot_errors = [ + snapshot + for snapshot in snapshots + if isinstance(snapshot, dict) and snapshot.get("import_error") + ] + snapshot_error_lines = [ + "- {package}@{version} ({source}): {error}".format( + package=snapshot.get("package"), + version=snapshot.get("version"), + source=snapshot.get("source"), + error=_one_line(snapshot.get("import_error")), + ) + for snapshot in snapshot_errors + ] + behavior_reason_lines = [ + f"- {_one_line(reason)}" + for reason in behavior_reasons + if isinstance(reason, str) and reason + ] promotion = current_state.get("promotion", {}) if isinstance(current_state, dict) else {} return "\n".join( [ @@ -132,6 +168,13 @@ def render_markdown_report( "", *(package_lines or ["- No package evidence collected."]), "", + "## API Snapshots", + "", + f"- Status: `{'incomplete' if snapshot_errors else 'pass'}`", + f"- Snapshot count: `{len(snapshots)}`", + f"- Import or execution errors: `{len(snapshot_errors)}`", + *(snapshot_error_lines or ["- No snapshot errors recorded."]), + "", "## API Diffs", "", f"- Diff count: `{len(api_diffs)}`", @@ -145,7 +188,13 @@ def render_markdown_report( f"- Status: `{behavior_summary.get('status')}`", f"- Changed contracts: `{behavior_summary.get('changed_count')}`", f"- Breaking contracts: `{behavior_summary.get('breaking_count')}`", + f"- Contract failures: `{behavior_summary.get('contract_failure_count')}`", + f"- Probe errors: `{behavior_summary.get('probe_error_count')}`", + f"- Skipped probes: `{behavior_summary.get('skipped_count')}`", + f"- Missing comparisons: `{behavior_summary.get('missing_comparison_count')}`", + f"- Malformed evidence: `{behavior_summary.get('malformed_count')}`", f"- Diff count: `{len(behavior_diffs)}`", + *(behavior_reason_lines or ["- No behavior evidence issues recorded."]), "", "## Direction Of Travel", "", @@ -190,3 +239,7 @@ def render_markdown_report( "", ] ) + + +def _one_line(value: object, *, limit: int = 560) -> str: + return " ".join(str(value).split())[:limit] diff --git a/examples/sdk_evolution_agent/stages.py b/examples/sdk_evolution_agent/stages.py index 15c9c53..703b20e 100644 --- a/examples/sdk_evolution_agent/stages.py +++ b/examples/sdk_evolution_agent/stages.py @@ -3,7 +3,6 @@ from __future__ import annotations import json -import re from collections.abc import Mapping, Sequence from pathlib import Path from typing import Any @@ -30,6 +29,11 @@ from agent_runtime_kit.events import safe_emit, task_completed_event, task_started_event from agent_runtime_kit.registry import create_default_registry from examples.sdk_evolution_agent.auth import prepare_isolated_codex_home +from examples.sdk_evolution_agent.behavior import ( + assess_behavior_payload, + behavior_expectations_from_evidence, +) +from examples.sdk_evolution_agent.collectors import parse_refresh_transitions from examples.sdk_evolution_agent.models import ( RUNTIME_CONTRACT_SYMBOLS, ApiDiff, @@ -242,7 +246,7 @@ async def run_analysis_pipeline( architecture = with_recursive_impact(architecture, api_diffs) architecture = with_candidate_api_diff_guard(architecture, evidence, api_diffs) architecture = with_release_note_guard(architecture, release_notes) - architecture = with_behavior_probe_guard(architecture, behavior) + architecture = with_behavior_probe_guard(architecture, evidence, behavior) architecture = with_manual_design_gate(architecture) architecture = _compact_stage_output(architecture) review = await run_stage( @@ -380,14 +384,34 @@ def with_candidate_api_diff_guard( ) -> dict[str, Any]: """Block SDK update implementation when candidate API evidence is missing.""" - update_packages = _refresh_update_packages(evidence) - if not update_packages: + transitions = parse_refresh_transitions(evidence) + if not transitions: return dict(architecture) - diff_packages = { - diff.package if isinstance(diff, ApiDiff) else str(diff.get("package") or "") + observed = { + ( + diff.package, + str(diff.from_version or ""), + str(diff.to_version or ""), + ) + if isinstance(diff, ApiDiff) + else ( + str(diff.get("package") or ""), + str(diff.get("from_version") or ""), + str(diff.get("to_version") or ""), + ) for diff in api_diffs + if isinstance(diff, ApiDiff | Mapping) } - missing = tuple(sorted(package for package in update_packages if package not in diff_packages)) + missing = tuple( + transition + for transition in transitions + if ( + transition.package, + transition.from_version, + transition.to_version, + ) + not in observed + ) if not missing: return dict(architecture) @@ -402,14 +426,21 @@ def with_candidate_api_diff_guard( "SDK update candidates require candidate-version API snapshot diffs " "before implementation can be considered safe." ), - "evidence": [f"missing api_diffs for {package}" for package in missing], + "evidence": [ + f"missing api_diffs for {transition.package} " + f"{transition.from_version} -> {transition.to_version}" + for transition in missing + ], } ) result["findings"] = findings uncertainty = list(result.get("uncertainty") or []) uncertainty.append( - "Candidate API diffs were not available for update candidate(s): " - + ", ".join(missing) + "Exact candidate API transitions were not available for: " + + ", ".join( + f"{transition.package} {transition.from_version} -> {transition.to_version}" + for transition in missing + ) ) result["uncertainty"] = uncertainty plan = list(result.get("self_adaptation_plan") or []) @@ -454,37 +485,76 @@ def with_release_note_guard( def with_behavior_probe_guard( architecture: Mapping[str, Any], + evidence: Mapping[str, Any], behavior: Mapping[str, Any], ) -> dict[str, Any]: - """Block implementation when candidate behavior probes fail.""" + """Block implementation unless raw behavior evidence is complete and consistent.""" - diffs = behavior.get("diffs") - if not isinstance(diffs, list): - return dict(architecture) - breaking = [ - diff - for diff in diffs - if isinstance(diff, Mapping) and str(diff.get("severity")) == "breaking" - ] - if not breaking: + validation_issues: list[str] = [] + computed = assess_behavior_payload( + behavior, + expectations=behavior_expectations_from_evidence(evidence), + ) + supplied = behavior.get("summary") + required_summary_keys = ( + "status", + "breaking_count", + "changed_count", + "unchanged_count", + "contract_failure_count", + "probe_error_count", + "skipped_count", + "missing_comparison_count", + "malformed_count", + "reasons", + ) + count_keys = required_summary_keys[1:-1] + if not isinstance(supplied, Mapping): + validation_issues.append("behavior summary must be an object") + else: + status = supplied.get("status") + if status not in {"pass", "changed", "incomplete", "fail"}: + validation_issues.append(f"unknown behavior summary status {status!r}") + for key in count_keys: + value = supplied.get(key) + if type(value) is not int or value < 0: + validation_issues.append(f"behavior summary {key} must be a non-negative integer") + reasons = supplied.get("reasons") + if not isinstance(reasons, list) or not all(isinstance(reason, str) for reason in reasons): + validation_issues.append("behavior summary reasons must be an array of strings") + for key in required_summary_keys: + if key not in supplied: + validation_issues.append(f"behavior summary is missing {key}") + elif supplied.get(key) != computed.get(key): + validation_issues.append( + f"behavior summary {key} contradicts raw evidence: " + f"supplied={supplied.get(key)!r} computed={computed.get(key)!r}" + ) + + status = str(computed.get("status") or "incomplete") + if not validation_issues and status in {"pass", "changed"}: return dict(architecture) + result = dict(architecture) result["safe_to_implement"] = False result["manual_design_required"] = True findings = list(result.get("findings") or []) + evidence_items = validation_issues + [ + str(reason) for reason in computed.get("reasons", []) if reason + ] findings.append( { "classification": "manual-design-required", - "summary": "Candidate SDK behavior probes detected breaking adapter-contract drift.", - "evidence": [ - f"{diff.get('package')}:{diff.get('probe')} {diff.get('summary')}" - for diff in breaking - ], + "summary": ("SDK behavior evidence is incomplete or failed deterministic validation."), + "evidence": evidence_items or [f"computed behavior status is {status}"], } ) result["findings"] = findings uncertainty = list(result.get("uncertainty") or []) - uncertainty.append("Breaking behavior probes require manual adapter design review.") + uncertainty.append( + "Behavior evidence must be complete, internally consistent, and non-breaking " + "before implementation." + ) result["uncertainty"] = uncertainty return result @@ -498,16 +568,6 @@ def with_manual_design_gate(architecture: Mapping[str, Any]) -> dict[str, Any]: return result -def _refresh_update_packages(evidence: Mapping[str, Any]) -> tuple[str, ...]: - preview = evidence.get("refresh_preview") - if not isinstance(preview, Mapping): - return () - text = f"{preview.get('stdout') or ''}\n{preview.get('stderr') or ''}" - return tuple( - sorted(set(re.findall(r"Update\s+([A-Za-z0-9_.-]+)\s+v\S+\s+->\s+v\S+", text))) - ) - - def _compact_stage_output(value: Mapping[str, Any]) -> dict[str, Any]: return {key: _compact_stage_value(item) for key, item in value.items()} @@ -554,8 +614,9 @@ def _stage_system_prompt(stage: str, schema: JsonSchema) -> str: "Do not mark manual_design_required, unsafe, or review rejection solely " "because public top-level symbols were added or removed when behavior " "probes pass before and after and there is no adapter-source evidence " - "that the removed symbols are used. Breaking behavior_diffs, missing " - "candidate API diffs, unavailable required release-note evidence, " + "that the removed symbols are used. Failed, incomplete, malformed, or " + "internally inconsistent behavior evidence; missing exact candidate API " + "transitions; unavailable required release-note evidence; " "reviewer-identified unsupported vendor behavior, or recursive " "runtime-contract impact remain hard blockers. Release-note status found " "is direct release-note evidence. Status no-matching-version is source " diff --git a/tests/test_sdk_evolution_agent.py b/tests/test_sdk_evolution_agent.py index b3dfec2..6250f59 100644 --- a/tests/test_sdk_evolution_agent.py +++ b/tests/test_sdk_evolution_agent.py @@ -1,5 +1,6 @@ from __future__ import annotations +import json import os import stat import subprocess @@ -29,26 +30,39 @@ from examples.sdk_evolution_agent import auth as auth_module from examples.sdk_evolution_agent.auth import CodexAuthResult, ensure_codex_sdk_auth from examples.sdk_evolution_agent.behavior import ( + assess_behavior_payload, collect_behavior_evidence, diff_behavior_results, probe_candidate_in_venv, + summarize_behavior, ) from examples.sdk_evolution_agent.cli import RunOptions, _collect_snapshots, parse_args, run_agent from examples.sdk_evolution_agent.collectors import ( + ResolverTransition, build_refresh_preview_command, collect_evidence, cutoff_free_env, + parse_refresh_transitions, run_lock_update, run_refresh_preview, ) from examples.sdk_evolution_agent.current_state import build_current_state -from examples.sdk_evolution_agent.models import ApiSnapshot, CommandResult, RunContext, SourceRef +from examples.sdk_evolution_agent.models import ( + ApiSnapshot, + BehaviorDiff, + BehaviorProbeResult, + CommandResult, + RunContext, + SourceRef, + to_jsonable, +) from examples.sdk_evolution_agent.pr import build_draft_pr_body from examples.sdk_evolution_agent.release_notes import ( _fetch_source_text, _format_github_discussions_index, collect_release_notes, ) +from examples.sdk_evolution_agent.report import render_markdown_report, write_run_report from examples.sdk_evolution_agent.schemas import ( DIRECTION_ANALYSIS_SCHEMA, SchemaValidationError, @@ -129,6 +143,47 @@ def runner( assert result.removed_env == ("UV_EXCLUDE_NEWER",) +def test_refresh_transition_parser_reads_exact_stdout_and_stderr_updates() -> None: + transitions = parse_refresh_transitions( + { + "refresh_preview": { + "stdout": ( + "Update claude-agent-sdk v0.2.96 -> v0.2.106\n" + "Update openai-codex v0.1.0b3 -> v0.1.0rc1\n" + ), + "stderr": ( + "Update google-antigravity v0.1.2 -> v0.1.4\n" + "Update claude-agent-sdk v0.2.96 -> v0.2.106\n" + ), + } + } + ) + + assert transitions == ( + ResolverTransition("claude-agent-sdk", "0.2.96", "0.2.106"), + ResolverTransition("google-antigravity", "0.1.2", "0.1.4"), + ResolverTransition("openai-codex", "0.1.0b3", "0.1.0rc1"), + ) + + +def test_refresh_transition_parser_rejects_partial_or_decorated_lines() -> None: + transitions = parse_refresh_transitions( + { + "refresh_preview": { + "stdout": ( + "NotUpdate claude-agent-sdk v1 -> v2\n" + "Update claude-agent-sdk v1 -> v2 trailing\n" + "prefix Update claude-agent-sdk v1 -> v2\n" + "Update claude-agent-sdk-extra v1 -> v2\n" + ), + "stderr": "", + } + } + ) + + assert transitions == (ResolverTransition("claude-agent-sdk-extra", "1", "2"),) + + def test_lock_update_uses_targeted_packages_and_clean_env( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, @@ -325,9 +380,7 @@ def test_fetch_source_text_uses_plain_fetch_without_token( url="https://github.com/o/r/discussions/categories/announcements", ) - text = _fetch_source_text( - source, fetcher=lambda url: "html fallback", use_github_graphql=True - ) + text = _fetch_source_text(source, fetcher=lambda url: "html fallback", use_github_graphql=True) assert text == "html fallback" @@ -419,7 +472,7 @@ def test_behavior_diffs_track_candidate_contract_changes() -> None: ], {}, ) - assert behavior["summary"]["status"] == "pass" + assert behavior["summary"]["status"] == "incomplete" diffs = diff_behavior_results( [ @@ -495,6 +548,14 @@ def isolated(package: str, version: str, *, scope: str = "candidate"): ("claude-agent-sdk", "0.2.106", "candidate"), ] assert behavior["diffs"][0].severity == "changed" + assert behavior["expected_packages"] == ["claude-agent-sdk"] + assert behavior["expected_transitions"] == [ + { + "package": "claude-agent-sdk", + "from_version": "0.2.96", + "to_version": "0.2.106", + } + ] def test_behavior_evidence_uses_locked_baseline_when_sdk_not_installed( @@ -607,6 +668,58 @@ def isolated(package: str, version: str, *, scope: str = "candidate"): assert calls == [("claude-agent-sdk", "0.2.106")] +def test_behavior_no_update_is_incomplete_when_ambient_version_drifted( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr( + "examples.sdk_evolution_agent.behavior.probe_current_package", + lambda package, *, version=None: ( + _probe(package, version, "current-environment", "pass", {"missing": []}), + ), + ) + + behavior = collect_behavior_evidence( + [ + { + "name": "claude-agent-sdk", + "locked_version": "1.0.0", + "installed_version": "2.0.0", + } + ], + {}, + ) + + assert behavior["expected_baselines"] == {"claude-agent-sdk": "1.0.0"} + assert behavior["summary"]["status"] == "incomplete" + assert behavior["summary"]["missing_comparison_count"] == 1 + assert "observed: 2.0.0" in " ".join(behavior["summary"]["reasons"]) + + +def test_behavior_no_update_passes_when_ambient_matches_locked_baseline( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr( + "examples.sdk_evolution_agent.behavior.probe_current_package", + lambda package, *, version=None: ( + _probe(package, version, "current-environment", "pass", {"missing": []}), + ), + ) + + behavior = collect_behavior_evidence( + [ + { + "name": "claude-agent-sdk", + "locked_version": "1.0.0", + "installed_version": "1.0.0", + } + ], + {}, + ) + + assert behavior["summary"]["status"] == "pass" + assert behavior["summary"]["missing_comparison_count"] == 0 + + def test_behavior_evidence_reuses_matching_ambient_baseline( monkeypatch: pytest.MonkeyPatch, ) -> None: @@ -697,14 +810,22 @@ def test_candidate_behavior_probe_scrubs_subprocess_env( def fake_run(args: Any, **kwargs: Any) -> Any: captured_envs.append(kwargs.get("env")) - return types.SimpleNamespace(stdout="[]", returncode=0) + return types.SimpleNamespace( + stdout=( + '[{"package":"claude-agent-sdk","version":"9.9.9",' + '"scope":"candidate","probe":"adapter-contract","status":"pass",' + '"summary":"ok","details":{}}]' + ), + returncode=0, + ) monkeypatch.setenv("ARK_FAKE_SECRET", "not-for-candidates") monkeypatch.setattr("examples.sdk_evolution_agent.behavior.subprocess.run", fake_run) results = probe_candidate_in_venv("claude-agent-sdk", "9.9.9") - assert results == () + assert len(results) == 1 + assert results[0].status == "pass" # venv create, pip install, probe script: every subprocess that touches the # freshly downloaded candidate runs with the scrubbed environment. assert len(captured_envs) == 3 @@ -714,7 +835,390 @@ def fake_run(args: Any, **kwargs: Any) -> Any: assert env.get("HOME") != os.environ.get("HOME") -def test_behavior_probe_guard_blocks_breaking_candidate_diff() -> None: +@pytest.mark.parametrize( + ("package", "failure_call", "error", "failure_step", "probe"), + [ + ( + "claude-agent-sdk", + 1, + subprocess.CalledProcessError( + 1, + ("python", "-m", "venv"), + stderr="venv failed " + ("x" * 1_000), + ), + "virtual-environment-creation", + "adapter-contract", + ), + ( + "openai-codex-cli-bin", + 2, + OSError("installer unavailable " + ("x" * 1_000)), + "package-installation", + "binary-distribution", + ), + ( + "google-antigravity", + 3, + subprocess.TimeoutExpired(("python", "-c", "probe"), 7), + "probe-execution", + "adapter-contract", + ), + ], +) +def test_candidate_behavior_probe_contains_subprocess_failures( + monkeypatch: pytest.MonkeyPatch, + package: str, + failure_call: int, + error: BaseException, + failure_step: str, + probe: str, +) -> None: + calls = 0 + + def fake_run(args: Any, **kwargs: Any) -> Any: + del args, kwargs + nonlocal calls + calls += 1 + if calls == failure_call: + raise error + return types.SimpleNamespace(stdout="", stderr="", returncode=0) + + monkeypatch.setattr("examples.sdk_evolution_agent.behavior.subprocess.run", fake_run) + + results = probe_candidate_in_venv(package, "9.9.9", timeout=7) + + assert len(results) == 1 + result = results[0] + assert result.package == package + assert result.version == "9.9.9" + assert result.scope == "candidate" + assert result.probe == probe + assert result.status == "fail" + assert result.details["failure_step"] == failure_step + assert result.details["error"] == result.summary + assert len(result.summary) <= 560 + + +@pytest.mark.parametrize( + "output", + [ + "not-json", + "[]", + ( + '[{"package":"wrong-sdk","version":"9.9.9","scope":"candidate",' + '"probe":"adapter-contract","status":"pass","summary":"ok","details":{}}]' + ), + ( + '[{"package":"claude-agent-sdk","version":"9.9.9","scope":"candidate",' + '"probe":"adapter-contract","status":"pass","summary":null,"details":{}}]' + ), + ( + '[{"package":"claude-agent-sdk","version":"9.9.9","scope":"candidate",' + '"probe":"adapter-contract","status":"unknown","summary":"ok","details":{}}]' + ), + ( + '[{"package":"claude-agent-sdk","version":"9.9.9","scope":"candidate",' + '"probe":"adapter-contract","status":"fail","summary":"missing",' + '"details":{"missing":7}}]' + ), + ], +) +def test_candidate_behavior_probe_contains_malformed_output( + monkeypatch: pytest.MonkeyPatch, + output: str, +) -> None: + def fake_run(args: Any, **kwargs: Any) -> Any: + del args, kwargs + return types.SimpleNamespace(stdout=output, stderr="", returncode=0) + + monkeypatch.setattr("examples.sdk_evolution_agent.behavior.subprocess.run", fake_run) + + (result,) = probe_candidate_in_venv("claude-agent-sdk", "9.9.9") + + assert result.package == "claude-agent-sdk" + assert result.version == "9.9.9" + assert result.scope == "candidate" + assert result.probe == "adapter-contract" + assert result.status == "fail" + assert result.details["failure_step"] == "probe-output-validation" + assert "malformed probe output" in result.summary + assert len(result.summary) <= 560 + + +def test_behavior_summary_accepts_valid_no_update_and_unchanged_evidence() -> None: + baseline = _probe( + "claude-agent-sdk", + "1.0.0", + "current-environment", + "pass", + {"missing": []}, + ) + no_update = summarize_behavior( + [baseline], + [], + expected_packages=["claude-agent-sdk"], + expected_transitions=[], + expected_baselines={"claude-agent-sdk": "1.0.0"}, + ) + assert no_update["status"] == "pass" + assert no_update["missing_comparison_count"] == 0 + + candidate = _probe( + "claude-agent-sdk", + "2.0.0", + "candidate", + "pass", + {"missing": []}, + ) + diffs = diff_behavior_results([baseline, candidate]) + unchanged = summarize_behavior( + [baseline, candidate], + diffs, + expected_packages=["claude-agent-sdk"], + expected_transitions=[ResolverTransition("claude-agent-sdk", "1.0.0", "2.0.0")], + expected_baselines={"claude-agent-sdk": "1.0.0"}, + ) + assert unchanged["status"] == "pass" + assert unchanged["unchanged_count"] == 1 + + +def test_behavior_summary_reports_changed_only_for_complete_evidence() -> None: + baseline = _probe( + "claude-agent-sdk", + "1.0.0", + "current-baseline", + "pass", + {"missing": [], "required_fields": ["a"]}, + ) + candidate = _probe( + "claude-agent-sdk", + "2.0.0", + "candidate", + "pass", + {"missing": [], "required_fields": ["a", "b"]}, + ) + diffs = diff_behavior_results([baseline, candidate]) + + summary = summarize_behavior( + [baseline, candidate], + diffs, + expected_packages=["claude-agent-sdk"], + expected_transitions=[ResolverTransition("claude-agent-sdk", "1.0.0", "2.0.0")], + expected_baselines={"claude-agent-sdk": "1.0.0"}, + ) + + assert summary["status"] == "changed" + assert summary["changed_count"] == 1 + + +def test_behavior_summary_marks_errors_skips_missing_and_malformed_as_incomplete() -> None: + transition = ResolverTransition("claude-agent-sdk", "1.0.0", "2.0.0") + baseline = _probe("claude-agent-sdk", "1.0.0", "current-baseline", "pass", {"missing": []}) + error = _probe( + "claude-agent-sdk", + "2.0.0", + "candidate", + "fail", + {"error": "probe crashed", "failure_step": "probe-execution"}, + ) + skipped = _probe("claude-agent-sdk", "2.0.0", "candidate", "skip", {"reason": "opt-in"}) + wrong_version = _probe("claude-agent-sdk", "3.0.0", "candidate", "pass", {"missing": []}) + + probe_error = summarize_behavior( + [baseline, error], + [], + expected_packages=["claude-agent-sdk"], + expected_transitions=[transition], + expected_baselines={"claude-agent-sdk": "1.0.0"}, + ) + probe_skip = summarize_behavior( + [baseline, skipped], + [], + expected_packages=["claude-agent-sdk"], + expected_transitions=[transition], + expected_baselines={"claude-agent-sdk": "1.0.0"}, + ) + missing = summarize_behavior( + [baseline, wrong_version], + diff_behavior_results([baseline, wrong_version]), + expected_packages=["claude-agent-sdk"], + expected_transitions=[transition], + expected_baselines={"claude-agent-sdk": "1.0.0"}, + ) + malformed = summarize_behavior( + [{"package": "claude-agent-sdk", "summary": None}], + [], + expected_packages=["claude-agent-sdk"], + expected_transitions=[], + expected_baselines={"claude-agent-sdk": "1.0.0"}, + ) + + assert probe_error["status"] == "incomplete" + assert probe_error["probe_error_count"] == 1 + assert probe_skip["status"] == "incomplete" + assert probe_skip["skipped_count"] == 1 + assert missing["status"] == "incomplete" + assert missing["missing_comparison_count"] == 1 + assert malformed["status"] == "incomplete" + assert malformed["malformed_count"] == 1 + + +def test_behavior_summary_contains_malformed_nested_contract_details() -> None: + transition = ResolverTransition("claude-agent-sdk", "1.0.0", "2.0.0") + baseline = _probe( + "claude-agent-sdk", "1.0.0", "current-baseline", "pass", {"missing": []} + ) + malformed_candidate = BehaviorProbeResult( + package="claude-agent-sdk", + version="2.0.0", + scope="candidate", + probe="adapter-contract", + status="fail", + summary="missing fields", + details={"missing": 7}, + ) + supplied_diffs = diff_behavior_results([baseline, malformed_candidate]) + + summary = summarize_behavior( + [baseline, malformed_candidate], + supplied_diffs, + expected_packages=["claude-agent-sdk"], + expected_transitions=[transition], + expected_baselines={"claude-agent-sdk": "1.0.0"}, + ) + + assert summary["status"] == "incomplete" + assert summary["malformed_count"] == 2 + assert "details.missing" in " ".join(summary["reasons"]) + + +def test_behavior_summary_treats_missing_fields_plus_error_as_contract_failure() -> None: + transition = ResolverTransition("claude-agent-sdk", "1.0.0", "2.0.0") + baseline = _probe( + "claude-agent-sdk", "1.0.0", "current-baseline", "pass", {"missing": []} + ) + candidate = _probe( + "claude-agent-sdk", + "2.0.0", + "candidate", + "fail", + {"missing": ["required_field"], "error": "secondary diagnostic"}, + ) + diffs = diff_behavior_results([baseline, candidate]) + + summary = summarize_behavior( + [baseline, candidate], + diffs, + expected_packages=["claude-agent-sdk"], + expected_transitions=[transition], + expected_baselines={"claude-agent-sdk": "1.0.0"}, + ) + + assert diffs[0].severity == "breaking" + assert summary["status"] == "fail" + assert summary["contract_failure_count"] == 1 + assert summary["probe_error_count"] == 0 + + +def test_behavior_summary_requires_complete_probe_sets_for_transition() -> None: + transition = ResolverTransition("claude-agent-sdk", "1.0.0", "2.0.0") + results = [ + _probe( + "claude-agent-sdk", + "1.0.0", + "current-baseline", + "pass", + {"missing": []}, + ), + BehaviorProbeResult( + package="claude-agent-sdk", + version="1.0.0", + scope="current-baseline", + probe="package-import", + status="pass", + summary="imported", + details={}, + ), + _probe( + "claude-agent-sdk", + "2.0.0", + "candidate", + "pass", + {"missing": []}, + ), + ] + diffs = diff_behavior_results(results) + + summary = summarize_behavior( + results, + diffs, + expected_packages=["claude-agent-sdk"], + expected_transitions=[transition], + expected_baselines={"claude-agent-sdk": "1.0.0"}, + ) + + assert {diff.probe for diff in diffs} == {"adapter-contract"} + assert summary["status"] == "incomplete" + assert summary["missing_comparison_count"] == 1 + + +def test_behavior_summary_status_precedence_is_fail_then_incomplete_then_changed() -> None: + baseline = _probe("claude-agent-sdk", "1.0.0", "current-baseline", "pass", {"missing": []}) + contract_failure = _probe( + "claude-agent-sdk", + "2.0.0", + "candidate", + "fail", + {"missing": ["required_field"]}, + ) + extra_skip = _probe("fake-sdk", "1.0.0", "candidate", "skip", {"reason": "missing"}) + failed = summarize_behavior( + [baseline, contract_failure, extra_skip], + diff_behavior_results([baseline, contract_failure, extra_skip]), + expected_packages=["claude-agent-sdk"], + expected_transitions=[ResolverTransition("claude-agent-sdk", "1.0.0", "2.0.0")], + expected_baselines={"claude-agent-sdk": "1.0.0"}, + ) + assert failed["status"] == "fail" + assert failed["contract_failure_count"] == 1 + assert failed["skipped_count"] == 1 + + changed_candidate = _probe( + "claude-agent-sdk", + "2.0.0", + "candidate", + "pass", + {"missing": [], "required_fields": ["new"]}, + ) + changed_diffs = diff_behavior_results([baseline, changed_candidate]) + incomplete = summarize_behavior( + [baseline, changed_candidate, extra_skip], + changed_diffs, + expected_packages=["claude-agent-sdk"], + expected_transitions=[ResolverTransition("claude-agent-sdk", "1.0.0", "2.0.0")], + expected_baselines={"claude-agent-sdk": "1.0.0"}, + ) + assert incomplete["status"] == "incomplete" + assert incomplete["changed_count"] == 1 + + +def test_behavior_probe_guard_blocks_complete_breaking_candidate_payload() -> None: + baseline = _probe( + "google-antigravity", "1.0.0", "current-baseline", "pass", {"missing": []} + ) + candidate = _probe( + "google-antigravity", + "2.0.0", + "candidate", + "fail", + {"missing": ["required_field"]}, + ) + transition = ResolverTransition("google-antigravity", "1.0.0", "2.0.0") + behavior = _behavior_payload( + [baseline, candidate], + expected_packages=["google-antigravity"], + expected_transitions=[transition], + ) guarded = with_behavior_probe_guard( { "findings": [], @@ -722,20 +1226,155 @@ def test_behavior_probe_guard_blocks_breaking_candidate_diff() -> None: "manual_design_required": False, "uncertainty": [], }, + _sdk_evidence( + package="google-antigravity", + locked_version="1.0.0", + installed_version="1.0.0", + candidate_version="2.0.0", + ), + behavior, + ) + + assert guarded["safe_to_implement"] is False + assert guarded["manual_design_required"] is True + assert behavior["summary"]["status"] == "fail" + assert "breaking" in " ".join(guarded["findings"][-1]["evidence"]) + + +def test_behavior_probe_guard_allows_valid_pass_and_changed_evidence() -> None: + architecture = { + "findings": [], + "safe_to_implement": True, + "manual_design_required": False, + "uncertainty": [], + } + baseline = _probe("claude-agent-sdk", "1.0.0", "current-baseline", "pass", {"missing": []}) + pass_payload = _behavior_payload( + [baseline], + expected_packages=["claude-agent-sdk"], + ) + changed_candidate = _probe( + "claude-agent-sdk", + "2.0.0", + "candidate", + "pass", + {"missing": [], "required_fields": ["new"]}, + ) + transition = ResolverTransition("claude-agent-sdk", "1.0.0", "2.0.0") + changed_payload = _behavior_payload( + [baseline, changed_candidate], + expected_packages=["claude-agent-sdk"], + expected_transitions=[transition], + ) + + assert ( + with_behavior_probe_guard(architecture, _sdk_evidence(), pass_payload)[ + "safe_to_implement" + ] + is True + ) + assert ( + with_behavior_probe_guard( + architecture, + _sdk_evidence(candidate_version="2.0.0"), + changed_payload, + )["safe_to_implement"] + is True + ) + + +def test_behavior_probe_guard_blocks_self_declared_empty_expectations() -> None: + behavior = _behavior_payload([], expected_packages=[]) + assert behavior["summary"]["status"] == "pass" + + guarded = with_behavior_probe_guard( { - "diffs": [ - { - "package": "google-antigravity", - "probe": "adapter-contract", - "severity": "breaking", - "summary": "Candidate probe changed from pass to fail.", - } - ] + "findings": [], + "safe_to_implement": True, + "manual_design_required": False, + "uncertainty": [], }, + _sdk_evidence(), + behavior, ) assert guarded["safe_to_implement"] is False assert guarded["manual_design_required"] is True + assert "contradicts deterministic evidence" in " ".join( + guarded["findings"][-1]["evidence"] + ) + + +def test_behavior_probe_guard_blocks_failed_incomplete_and_invalid_evidence() -> None: + architecture = { + "findings": [], + "safe_to_implement": True, + "manual_design_required": False, + "uncertainty": [], + } + baseline = _probe("claude-agent-sdk", "1.0.0", "current-baseline", "pass", {"missing": []}) + transition = ResolverTransition("claude-agent-sdk", "1.0.0", "2.0.0") + failed_payload = _behavior_payload( + [ + baseline, + _probe( + "claude-agent-sdk", + "2.0.0", + "candidate", + "fail", + {"missing": ["required_field"]}, + ), + ], + expected_packages=["claude-agent-sdk"], + expected_transitions=[transition], + ) + incomplete_payload = _behavior_payload( + [ + baseline, + _probe( + "claude-agent-sdk", + "2.0.0", + "candidate", + "skip", + {"reason": "opt-in"}, + ), + ], + expected_packages=["claude-agent-sdk"], + expected_transitions=[transition], + ) + missing_summary = _behavior_payload( + [baseline], + expected_packages=["claude-agent-sdk"], + ) + missing_summary.pop("summary") + unknown_status = _behavior_payload( + [baseline], + expected_packages=["claude-agent-sdk"], + ) + unknown_status["summary"]["status"] = "mystery" + contradictory = _behavior_payload( + [baseline], + expected_packages=["claude-agent-sdk"], + ) + contradictory["results"] = failed_payload["results"] + contradictory["expected_transitions"] = failed_payload["expected_transitions"] + malformed = _behavior_payload( + [baseline], + expected_packages=["claude-agent-sdk"], + ) + malformed["results"] = [{"package": "claude-agent-sdk", "summary": None}] + + for evidence, payload in ( + (_sdk_evidence(candidate_version="2.0.0"), failed_payload), + (_sdk_evidence(candidate_version="2.0.0"), incomplete_payload), + (_sdk_evidence(), missing_summary), + (_sdk_evidence(), unknown_status), + (_sdk_evidence(), contradictory), + (_sdk_evidence(), malformed), + ): + guarded = with_behavior_probe_guard(architecture, evidence, payload) + assert guarded["safe_to_implement"] is False + assert guarded["manual_design_required"] is True def test_current_state_artifact_paths_are_repo_relative(tmp_path: Path) -> None: @@ -750,6 +1389,7 @@ def test_current_state_artifact_paths_are_repo_relative(tmp_path: Path) -> None: report_root = tmp_path / "reports" / "sdk-evolution" / "run-1" report_root.mkdir(parents=True) (report_root / "evidence.json").write_text("{}", encoding="utf-8") + (report_root / "behavior_summary.json").write_text('{"status":"pass"}', encoding="utf-8") snapshots = report_root / "api_snapshots" snapshots.mkdir() (snapshots / "01-claude-agent-sdk.json").write_text("{}", encoding="utf-8") @@ -772,11 +1412,129 @@ def test_current_state_artifact_paths_are_repo_relative(tmp_path: Path) -> None: paths = [artifact["path"] for artifact in state["artifacts"].values()] assert "reports/sdk-evolution/run-1/evidence.json" in paths + assert "reports/sdk-evolution/run-1/behavior_summary.json" in paths assert "reports/sdk-evolution/run-1/api_snapshots/01-claude-agent-sdk.json" in paths + assert state["artifacts"]["behavior_summary.json"]["sha256"] assert all(not path.startswith("/") for path in paths) assert all("/private/tmp" not in path and "/tmp/" not in path for path in paths) +def test_report_exposes_snapshot_and_behavior_evidence_failures() -> None: + behavior = _behavior_payload( + [ + BehaviorProbeResult( + package="claude-agent-sdk", + version="2.0.0", + scope="candidate", + probe="adapter-contract", + status="fail", + summary="probe crashed", + details={"error": "probe crashed", "failure_step": "probe-execution"}, + ) + ], + expected_packages=["claude-agent-sdk"], + expected_transitions=[ResolverTransition("claude-agent-sdk", "1.0.0", "2.0.0")], + ) + + report = render_markdown_report( + config={"runtime": "fake", "implementation_enabled": False, "draft_pr": False}, + evidence=_sdk_evidence(candidate_version="2.0.0"), + snapshots=[ + { + "package": "claude-agent-sdk", + "version": "2.0.0", + "source": "isolated-venv", + "import_error": "candidate import failed\nwith details", + } + ], + api_diffs=[], + release_notes=[], + behavior=behavior, + current_state={"promotion": {"status": "skipped", "promoted": False}}, + direction={}, + architecture={ + "manual_design_required": True, + "recursive_self_adaptation_impact": False, + "safe_to_implement": False, + }, + implementation={}, + review={}, + ) + + assert "## API Snapshots" in report + assert "- Status: `incomplete`" in report + assert "- Snapshot count: `1`" in report + assert "- Import or execution errors: `1`" in report + assert "candidate import failed with details" in report + assert "## Behavior Probes" in report + assert "- Probe errors: `1`" in report + assert "- Missing comparisons: `1`" in report + assert "probe crashed" in report + + +def test_report_persists_and_renders_recomputed_behavior_summary(tmp_path: Path) -> None: + baseline = _probe( + "claude-agent-sdk", "1.0.0", "current-baseline", "pass", {"missing": []} + ) + transition = ResolverTransition("claude-agent-sdk", "1.0.0", "2.0.0") + behavior = _behavior_payload( + [ + baseline, + _probe( + "claude-agent-sdk", + "2.0.0", + "candidate", + "fail", + {"missing": ["required_field"]}, + ), + ], + expected_packages=["claude-agent-sdk"], + expected_transitions=[transition], + ) + behavior["summary"] = _behavior_payload( + [baseline], expected_packages=["claude-agent-sdk"] + )["summary"] + assert behavior["summary"]["status"] == "pass" + + report_root = tmp_path / "reports" / "run-1" + context = RunContext( + run_id="run-1", + workspace=tmp_path, + report_root=report_root, + runtime="fake", + event_log_path=report_root / "events.jsonl", + implementation_enabled=False, + draft_pr=False, + ) + report_path = write_run_report( + context, + config={"runtime": "fake", "implementation_enabled": False, "draft_pr": False}, + evidence=_sdk_evidence(candidate_version="2.0.0"), + snapshots=[], + api_diffs=[], + release_notes=[], + behavior=behavior, + current_state={"promotion": {"status": "skipped", "promoted": False}}, + direction={}, + architecture={ + "manual_design_required": True, + "recursive_self_adaptation_impact": False, + "safe_to_implement": False, + }, + implementation={}, + review={}, + ) + + persisted = json.loads( + (report_root / "behavior_summary.json").read_text(encoding="utf-8") + ) + report = report_path.read_text(encoding="utf-8") + assert persisted["status"] == "fail" + assert persisted["contract_failure_count"] == 1 + assert "- Status: `fail`" in report + assert "failed the required contract" in report + + def test_snapshot_and_diff_public_api(monkeypatch: pytest.MonkeyPatch) -> None: module = types.ModuleType("fake_sdk") @@ -1227,7 +1985,52 @@ def test_candidate_api_diff_guard_blocks_missing_update_diff() -> None: assert guarded["safe_to_implement"] is False assert guarded["manual_design_required"] is True - assert "missing api_diffs for google-antigravity" in guarded["findings"][-1]["evidence"] + assert ( + guarded["findings"][-1]["evidence"][0] + == "missing api_diffs for google-antigravity 0.1.2 -> 0.1.4" + ) + + +@pytest.mark.parametrize( + "api_diff", + [ + { + "package": "google-antigravity", + "from_version": "0.1.1", + "to_version": "0.1.4", + }, + { + "package": "google-antigravity", + "from_version": "0.1.2", + "to_version": "0.1.5", + }, + { + "package": "google-antigravity-extra", + "from_version": "0.1.2", + "to_version": "0.1.4", + }, + ], +) +def test_candidate_api_diff_guard_requires_exact_transition(api_diff: dict[str, str]) -> None: + guarded = with_candidate_api_diff_guard( + { + "findings": [], + "safe_to_implement": True, + "manual_design_required": False, + "uncertainty": [], + "self_adaptation_plan": [], + }, + { + "refresh_preview": { + "stdout": "", + "stderr": "Update google-antigravity v0.1.2 -> v0.1.4\n", + } + }, + [api_diff], + ) + + assert guarded["safe_to_implement"] is False + assert guarded["manual_design_required"] is True def test_candidate_api_diff_guard_accepts_empty_update_diff() -> None: @@ -1500,6 +2303,7 @@ async def test_run_agent_report_only_generates_artifacts( assert (report_path.parent / "api_diffs.json").exists() assert (report_path.parent / "behavior_probes.json").exists() assert (report_path.parent / "behavior_diffs.json").exists() + assert (report_path.parent / "behavior_summary.json").exists() assert (report_path.parent / "current_state.json").exists() assert (report_path.parent / "direction_analysis.json").exists() assert (report_path.parent / "architecture_decision.json").exists() @@ -1540,11 +2344,21 @@ def forbidden_candidate(package: str, version: str) -> ApiSnapshot: ) monkeypatch.setattr( "examples.sdk_evolution_agent.cli.collect_behavior_evidence", - lambda packages, updates, *, inspect_candidates=False: { - "results": [], - "diffs": [], - "summary": {"status": "pass"}, - }, + lambda packages, updates, *, inspect_candidates=False, expected_transitions=None: ( + _behavior_payload( + [ + _probe( + "claude-agent-sdk", + None, + "current-environment", + "pass", + {}, + ) + ], + expected_packages=["claude-agent-sdk"], + expected_transitions=expected_transitions, + ) + ), ) # Default run (no inspect_candidates) must never pip-install/import upstream code. @@ -1667,11 +2481,28 @@ async def test_run_agent_autonomous_pr_path( ) monkeypatch.setattr( "examples.sdk_evolution_agent.cli.collect_behavior_evidence", - lambda packages, updates, *, inspect_candidates=False: { - "results": [], - "diffs": [], - "summary": {"status": "pass"}, - }, + lambda packages, updates, *, inspect_candidates=False, expected_transitions=None: ( + _behavior_payload( + [ + _probe( + "claude-agent-sdk", + "0.2.1", + "current-environment", + "pass", + {"missing": []}, + ), + _probe( + "claude-agent-sdk", + "0.3.0", + "candidate", + "pass", + {"missing": []}, + ), + ], + expected_packages=["claude-agent-sdk"], + expected_transitions=expected_transitions, + ) + ), ) commands: list[tuple[str, ...]] = [] @@ -2062,13 +2893,11 @@ async def run(self, task: AgentTask) -> AgentResult: def _probe( package: str, - version: str, + version: str | None, scope: str, status: str, details: dict[str, Any], -): - from examples.sdk_evolution_agent.models import BehaviorProbeResult - +) -> BehaviorProbeResult: return BehaviorProbeResult( package=package, version=version, @@ -2080,6 +2909,76 @@ def _probe( ) +def _behavior_payload( + results: list[BehaviorProbeResult], + *, + diffs: list[BehaviorDiff] | None = None, + expected_packages: list[str] | None = None, + expected_transitions: list[ResolverTransition] | tuple[ResolverTransition, ...] | None = None, + expected_baselines: dict[str, str | None] | None = None, +) -> dict[str, Any]: + observed_diffs = diffs if diffs is not None else list(diff_behavior_results(results)) + packages = expected_packages or [] + transitions = list(expected_transitions or []) + baselines = dict(expected_baselines or {}) + for package in packages: + transition = next( + (item for item in transitions if item.package == package), + None, + ) + if transition is not None: + baselines.setdefault(package, transition.from_version) + continue + observation = next( + ( + result + for result in results + if result.package == package + and result.scope in {"current-baseline", "current-environment"} + ), + None, + ) + baselines.setdefault(package, observation.version if observation is not None else None) + payload: dict[str, Any] = to_jsonable( + { + "results": results, + "diffs": observed_diffs, + "expected_packages": packages, + "expected_transitions": transitions, + "expected_baselines": baselines, + } + ) + payload["summary"] = assess_behavior_payload(payload) + return payload + + +def _sdk_evidence( + *, + package: str = "claude-agent-sdk", + locked_version: str | None = "1.0.0", + installed_version: str | None = "1.0.0", + candidate_version: str | None = None, +) -> dict[str, Any]: + evidence: dict[str, Any] = { + "packages": [ + { + "name": package, + "locked_version": locked_version, + "installed_version": installed_version, + } + ] + } + if candidate_version is not None: + evidence["refresh_preview"] = { + "stdout": "", + "stderr": ( + f"Update {package} v{locked_version or installed_version} " + f"-> v{candidate_version}\n" + ), + } + return evidence + + def _fake_pypi(package: str) -> dict[str, Any]: assert package == "claude-agent-sdk" return { From 34eb4a696423d8e6ac625e9d944873fa6a019546 Mon Sep 17 00:00:00 2001 From: Eloi Date: Tue, 28 Jul 2026 19:59:21 +0200 Subject: [PATCH 8/9] Inspect bundled Codex CLI correctly (#54) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Summary - map the `openai-codex-cli-bin` distribution to its real `codex_cli_bin` import module - use that module consistently for current and isolated API snapshots - require metadata, module import, `bundled_codex_path()`, and an existing regular file before the binary probe passes - preserve the `binary-distribution` probe identity for every in-process and embedded failure path - keep success evidence stable without persisting temporary executable paths - serialize binary-probe failures with bounded exception context while redacting everything from the first absolute POSIX, drive, UNC, or URL path start ## Stack - Depends on #53 - This is layer 4 of 5 in the SDK evolution demo reliability stack ## Review notes Two correctness passes covered wrong-module snapshots, metadata/import/helper/conversion/file-check failures, invalid helper returns, regular-file semantics, embedded probe labels, cross-platform path forms, diagnostic bounds, and path-bearing messages with spaces. No actionable findings remain. ## Validation - `PYTHONPATH=. .venv/bin/python -m pytest -q` — 477 passed, 8 skipped - SDK evolution test file — 114 passed - Codex CLI focused selection — 23 passed - real installed locked Codex wheel snapshot/probe — passed - `.venv/bin/ruff check .` - `.venv/bin/mypy` - `git diff --check` Security-diff scans were intentionally not part of this phase review. --- examples/sdk_evolution_agent/behavior.py | 108 +++++- examples/sdk_evolution_agent/snapshots.py | 2 +- tests/test_sdk_evolution_agent.py | 399 ++++++++++++++++++++++ 3 files changed, 499 insertions(+), 10 deletions(-) diff --git a/examples/sdk_evolution_agent/behavior.py b/examples/sdk_evolution_agent/behavior.py index 5594972..11d9d2f 100644 --- a/examples/sdk_evolution_agent/behavior.py +++ b/examples/sdk_evolution_agent/behavior.py @@ -6,6 +6,7 @@ import importlib.metadata import inspect import json +import re import subprocess import sys import tempfile @@ -34,6 +35,10 @@ "start_params", } ) +_SENSITIVE_PATH_START_RE = re.compile( + r"(?P^|[^A-Za-z0-9_.-])" + r"(?P(?:[A-Za-z][A-Za-z0-9+.-]*://|[A-Za-z]:[\\/]|\\\\|/))" +) def collect_behavior_evidence( @@ -848,8 +853,11 @@ def _probe_codex(*, version: str | None, scope: str) -> BehaviorProbeResult: def _probe_codex_cli_bin(*, version: str | None, scope: str) -> BehaviorProbeResult: package = "openai-codex-cli-bin" + module_name = DEFAULT_MODULES[package] try: installed = importlib.metadata.version(package) + module = importlib.import_module(module_name) + _require_bundled_codex_file(module) except Exception as exc: return _failed(package, version, scope, "binary-distribution", exc) return BehaviorProbeResult( @@ -858,11 +866,37 @@ def _probe_codex_cli_bin(*, version: str | None, scope: str) -> BehaviorProbeRes scope=scope, probe="binary-distribution", status="pass", - summary="Codex CLI binary distribution metadata is available.", - details={"installed_version": installed}, + summary="Codex CLI binary distribution exposes a bundled executable.", + details={ + "installed_version": installed, + "module": module_name, + "bundled_binary": "regular-file", + }, ) +def _require_bundled_codex_file(module: Any) -> None: + """Require the Codex wheel helper to resolve to a regular file without retaining its path.""" + + try: + raw_path = module.bundled_codex_path() + except Exception as exc: + detail = _safe_exception_detail(exc) + raise RuntimeError(f"codex_cli_bin.bundled_codex_path() failed ({detail})") from exc + try: + bundled_path = Path(raw_path) + is_file = bundled_path.is_file() + except (OSError, TypeError, ValueError) as exc: + detail = _safe_exception_detail(exc) + raise RuntimeError( + f"codex_cli_bin.bundled_codex_path() did not return a usable path ({detail})" + ) from exc + if not is_file: + raise RuntimeError( + "codex_cli_bin.bundled_codex_path() did not return an existing regular file" + ) + + def _probe_antigravity(*, version: str | None, scope: str) -> BehaviorProbeResult: package = "google-antigravity" try: @@ -955,7 +989,7 @@ def _failed( probe: str, exc: Exception, ) -> BehaviorProbeResult: - error = _bounded_text(exc) + error = _safe_exception_detail(exc) if package == "openai-codex-cli-bin" else _bounded_text(exc) return BehaviorProbeResult( package=package, version=version, @@ -1048,6 +1082,15 @@ def _bounded_text(value: object, *, limit: int = 480) -> str: return " ".join(text.split())[:limit] +def _safe_exception_detail(exc: Exception, *, limit: int = 240) -> str: + """Keep useful failure context while removing path-bearing message suffixes.""" + + message = " ".join(str(exc).split()) + match = _SENSITIVE_PATH_START_RE.search(message) + redacted = f"{message[: match.start('path')]}" if match is not None else message + return _bounded_text(f"{type(exc).__name__}: {redacted}", limit=limit) + + def _string_or_none(value: object) -> str | None: if value is None: return None @@ -1056,14 +1099,30 @@ def _string_or_none(value: object) -> str | None: _PROBE_SCRIPT = textwrap.dedent( - """ + r""" import importlib import importlib.metadata import inspect import json + import re import sys + from pathlib import Path package, version, scope = sys.argv[1:4] + sensitive_path_start_re = re.compile( + r"(?P^|[^A-Za-z0-9_.-])" + r"(?P(?:[A-Za-z][A-Za-z0-9+.-]*://|[A-Za-z]:[\\/]|\\\\|/))" + ) + + def safe_exception_detail(exc, limit=240): + message = " ".join(str(exc).split()) + match = sensitive_path_start_re.search(message) + redacted = ( + f"{message[:match.start('path')]}" + if match is not None + else message + ) + return f"{type(exc).__name__}: {redacted}"[:limit] def fields(cls): if hasattr(cls, "model_fields"): @@ -1076,14 +1135,19 @@ def fields(cls): return set() def failed(probe, exc): + detail = ( + safe_exception_detail(exc) + if package == "openai-codex-cli-bin" + else str(exc) + ) return { "package": package, "version": version, "scope": scope, "probe": probe, "status": "fail", - "summary": str(exc), - "details": {"error": str(exc)}, + "summary": detail, + "details": {"error": detail}, } def result(probe, status, summary, details): @@ -1142,11 +1206,36 @@ def result(probe, status, summary, details): )] elif package == "openai-codex-cli-bin": installed = importlib.metadata.version(package) + module = importlib.import_module("codex_cli_bin") + try: + raw_path = module.bundled_codex_path() + except Exception as exc: + detail = safe_exception_detail(exc) + raise RuntimeError( + f"codex_cli_bin.bundled_codex_path() failed ({detail})" + ) from exc + try: + bundled_path = Path(raw_path) + is_file = bundled_path.is_file() + except (OSError, TypeError, ValueError) as exc: + detail = safe_exception_detail(exc) + raise RuntimeError( + "codex_cli_bin.bundled_codex_path() did not return a usable path " + f"({detail})" + ) from exc + if not is_file: + raise RuntimeError( + "codex_cli_bin.bundled_codex_path() did not return an existing regular file" + ) payload = [result( "binary-distribution", "pass", - "Codex CLI binary distribution metadata is available.", - {"installed_version": installed}, + "Codex CLI binary distribution exposes a bundled executable.", + { + "installed_version": installed, + "module": "codex_cli_bin", + "bundled_binary": "regular-file", + }, )] elif package == "google-antigravity": importlib.import_module("google.antigravity") @@ -1178,7 +1267,8 @@ def result(probe, status, summary, details): else: payload = [result("package-import", "skip", "No behavior probe is defined.", {})] except Exception as exc: - payload = [failed("adapter-contract", exc)] + probe = "binary-distribution" if package == "openai-codex-cli-bin" else "adapter-contract" + payload = [failed(probe, exc)] print(json.dumps(payload, sort_keys=True)) """ diff --git a/examples/sdk_evolution_agent/snapshots.py b/examples/sdk_evolution_agent/snapshots.py index 31f80e8..dbca816 100644 --- a/examples/sdk_evolution_agent/snapshots.py +++ b/examples/sdk_evolution_agent/snapshots.py @@ -19,7 +19,7 @@ DEFAULT_MODULES = { "claude-agent-sdk": "claude_agent_sdk", "openai-codex": "openai_codex", - "openai-codex-cli-bin": "openai_codex_cli_bin", + "openai-codex-cli-bin": "codex_cli_bin", "google-antigravity": "google.antigravity", } diff --git a/tests/test_sdk_evolution_agent.py b/tests/test_sdk_evolution_agent.py index 6250f59..4fd3eb4 100644 --- a/tests/test_sdk_evolution_agent.py +++ b/tests/test_sdk_evolution_agent.py @@ -28,12 +28,14 @@ CodexAgentRuntime, ) from examples.sdk_evolution_agent import auth as auth_module +from examples.sdk_evolution_agent import behavior as behavior_module from examples.sdk_evolution_agent.auth import CodexAuthResult, ensure_codex_sdk_auth from examples.sdk_evolution_agent.behavior import ( assess_behavior_payload, collect_behavior_evidence, diff_behavior_results, probe_candidate_in_venv, + probe_current_package, summarize_behavior, ) from examples.sdk_evolution_agent.cli import RunOptions, _collect_snapshots, parse_args, run_agent @@ -69,6 +71,7 @@ validate_mapping, ) from examples.sdk_evolution_agent.snapshots import ( + DEFAULT_MODULES, diff_snapshots, snapshot_candidate_in_venv, snapshot_current_api, @@ -945,6 +948,335 @@ def fake_run(args: Any, **kwargs: Any) -> Any: assert len(result.summary) <= 560 +@pytest.mark.parametrize( + ("private_path", "secret_suffix"), + [ + ("/tmp/private folder/POSIX_SECRET_SUFFIX", "POSIX_SECRET_SUFFIX"), + (r"C:\Program Files\Private\DRIVE_SECRET_SUFFIX", "DRIVE_SECRET_SUFFIX"), + (r"\\server\Private Share\UNC_SECRET_SUFFIX", "UNC_SECRET_SUFFIX"), + ( + "https://example.invalid/private folder/URL_SECRET_SUFFIX", + "URL_SECRET_SUFFIX", + ), + ], +) +def test_codex_cli_exception_redaction_discards_entire_path_suffix( + private_path: str, + secret_suffix: str, +) -> None: + detail = behavior_module._safe_exception_detail( + ValueError( + f"REDACTION_SENTINEL at {private_path} trailing {secret_suffix} " + + ("x" * 1_000) + ) + ) + + assert detail == "ValueError: REDACTION_SENTINEL at " + assert private_path not in detail + assert secret_suffix not in detail + assert len(detail) <= 240 + + +@pytest.mark.parametrize( + ("failure", "expected_summary"), + [ + ("metadata", "RuntimeError: METADATA_SENTINEL"), + ("module", "ModuleNotFoundError: MODULE_SENTINEL"), + ("helper-missing", "bundled_codex_path() failed"), + ("helper-error", "ValueError: HELPER_SENTINEL"), + ("invalid-return", "TypeError"), + ("missing-file", "existing regular file"), + ("directory", "existing regular file"), + ], +) +def test_codex_cli_binary_probe_contains_each_failure( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + failure: str, + expected_summary: str, +) -> None: + metadata_path = tmp_path / "private metadata" / "METADATA_SECRET_SUFFIX" + drive_path = r"C:\Program Files\Private\MODULE_SECRET_SUFFIX" + unc_path = r"\\server\Private Share\HELPER_SECRET_SUFFIX" + + def metadata_version(package: str) -> str: + assert package == "openai-codex-cli-bin" + if failure == "metadata": + raise RuntimeError( + f"METADATA_SENTINEL at {metadata_path} " + ("m" * 1_000) + ) + return "1.2.3" + + def helper_error() -> Path: + raise ValueError( + f"HELPER_SENTINEL at {unc_path} " + ("h" * 1_000) + ) + + module = types.SimpleNamespace(bundled_codex_path=lambda: tmp_path / "codex") + if failure == "helper-missing": + module = types.SimpleNamespace() + elif failure == "helper-error": + module = types.SimpleNamespace(bundled_codex_path=helper_error) + elif failure == "invalid-return": + module = types.SimpleNamespace(bundled_codex_path=lambda: None) + elif failure == "directory": + module = types.SimpleNamespace(bundled_codex_path=lambda: tmp_path) + + def import_module(name: str) -> Any: + assert name == "codex_cli_bin" + if failure == "module": + raise ModuleNotFoundError( + f"MODULE_SENTINEL: No module named codex_cli_bin at {drive_path} " + + ("i" * 1_000) + ) + return module + + monkeypatch.setattr(behavior_module.importlib.metadata, "version", metadata_version) + monkeypatch.setattr(behavior_module.importlib, "import_module", import_module) + + (result,) = probe_current_package("openai-codex-cli-bin", version="1.2.3") + + assert result.probe == "binary-distribution" + assert result.status == "fail" + assert expected_summary in result.summary + assert str(tmp_path) not in result.summary + assert drive_path not in result.summary + assert unc_path not in result.summary + serialized_details = json.dumps(result.details) + assert str(tmp_path) not in serialized_details + assert drive_path not in serialized_details + assert unc_path not in serialized_details + for secret in ( + "METADATA_SECRET_SUFFIX", + "MODULE_SECRET_SUFFIX", + "HELPER_SECRET_SUFFIX", + ): + assert secret not in result.summary + assert secret not in serialized_details + assert result.details["error"] == result.summary + assert len(result.summary) <= 240 + assert len(result.details["error"]) <= 240 + if failure in {"metadata", "module", "helper-error"}: + assert "" in result.summary + + +@pytest.mark.parametrize( + ("failure", "expected_summary", "secret_suffix"), + [ + ("conversion", "ValueError: CONVERSION_SENTINEL", "CONVERSION_SECRET_SUFFIX"), + ("file-check", "OSError: FILE_CHECK_SENTINEL", "FILE_CHECK_SECRET_SUFFIX"), + ], +) +def test_codex_cli_binary_probe_redacts_path_conversion_and_file_check_errors( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + failure: str, + expected_summary: str, + secret_suffix: str, +) -> None: + conversion_url = ( + "https://example.invalid/private folder/CONVERSION_SECRET_SUFFIX" + ) + file_check_path = tmp_path / "private file check" / "FILE_CHECK_SECRET_SUFFIX" + + class InvalidPath: + def __fspath__(self) -> str: + raise ValueError( + f"CONVERSION_SENTINEL at {conversion_url} " + ("c" * 1_000) + ) + + if failure == "file-check": + + def fail_file_check(self: Path) -> bool: + del self + raise OSError( + f"FILE_CHECK_SENTINEL at {file_check_path} " + ("f" * 1_000) + ) + + monkeypatch.setattr(Path, "is_file", fail_file_check) + + raw_path: object = InvalidPath() if failure == "conversion" else tmp_path / "codex" + module = types.SimpleNamespace(bundled_codex_path=lambda: raw_path) + monkeypatch.setattr( + behavior_module.importlib.metadata, + "version", + lambda package: "1.2.3", + ) + monkeypatch.setattr( + behavior_module.importlib, + "import_module", + lambda name: module, + ) + + (result,) = probe_current_package("openai-codex-cli-bin", version="1.2.3") + + assert result.probe == "binary-distribution" + assert result.status == "fail" + assert expected_summary in result.summary + assert "" in result.summary + assert secret_suffix not in result.summary + assert secret_suffix not in json.dumps(result.details) + assert result.details["error"] == result.summary + assert len(result.summary) <= 240 + assert len(result.details["error"]) <= 240 + + +def test_codex_cli_binary_probe_requires_only_a_regular_file( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + binary = tmp_path / "codex" + binary.write_bytes(b"bundled binary") + module = types.SimpleNamespace(bundled_codex_path=lambda: binary) + monkeypatch.setattr( + behavior_module.importlib.metadata, + "version", + lambda package: "1.2.3", + ) + monkeypatch.setattr( + behavior_module.importlib, + "import_module", + lambda name: module, + ) + + (result,) = probe_current_package("openai-codex-cli-bin", version="1.2.3") + + assert result.status == "pass" + assert result.probe == "binary-distribution" + assert result.details == { + "installed_version": "1.2.3", + "module": "codex_cli_bin", + "bundled_binary": "regular-file", + } + assert str(binary) not in json.dumps(result.details) + + +@pytest.mark.parametrize( + ("failure", "expected_summary", "secret_suffix"), + [ + ( + "metadata", + "RuntimeError: EMBEDDED_METADATA_SENTINEL", + "EMBEDDED_METADATA_SECRET_SUFFIX", + ), + ( + "module", + "ModuleNotFoundError: EMBEDDED_MODULE_SENTINEL", + "EMBEDDED_MODULE_SECRET_SUFFIX", + ), + ( + "helper-error", + "ValueError: EMBEDDED_HELPER_SENTINEL", + "EMBEDDED_HELPER_SECRET_SUFFIX", + ), + ( + "conversion-error", + "ValueError: EMBEDDED_CONVERSION_SENTINEL", + "EMBEDDED_CONVERSION_SECRET_SUFFIX", + ), + ( + "file-check-error", + "OSError: EMBEDDED_FILE_CHECK_SENTINEL", + "EMBEDDED_FILE_CHECK_SECRET_SUFFIX", + ), + ("missing-file", "existing regular file", None), + ], +) +def test_embedded_codex_cli_failures_keep_binary_probe_label( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + capsys: pytest.CaptureFixture[str], + failure: str, + expected_summary: str, + secret_suffix: str | None, +) -> None: + metadata_path = ( + tmp_path / "private metadata" / "EMBEDDED_METADATA_SECRET_SUFFIX" + ) + drive_path = r"D:\Program Files\Private\EMBEDDED_MODULE_SECRET_SUFFIX" + unc_path = r"\\candidate-host\Private Share\EMBEDDED_HELPER_SECRET_SUFFIX" + conversion_url = ( + "https://example.invalid/private folder/EMBEDDED_CONVERSION_SECRET_SUFFIX" + ) + file_check_path = ( + tmp_path / "private file check" / "EMBEDDED_FILE_CHECK_SECRET_SUFFIX" + ) + + def fail_metadata(package: str) -> str: + if failure == "metadata": + raise RuntimeError( + f"EMBEDDED_METADATA_SENTINEL for {package} at {metadata_path} " + + ("m" * 1_000) + ) + return "1.2.3" + + def helper_error() -> Path: + raise ValueError( + f"EMBEDDED_HELPER_SENTINEL at {unc_path} " + ("h" * 1_000) + ) + + class InvalidPath: + def __fspath__(self) -> str: + raise ValueError( + f"EMBEDDED_CONVERSION_SENTINEL at {conversion_url} " + + ("c" * 1_000) + ) + + def bundled_path() -> object: + if failure == "helper-error": + return helper_error() + if failure == "conversion-error": + return InvalidPath() + return tmp_path / "missing" + + if failure == "file-check-error": + + def fail_file_check(self: Path) -> bool: + del self + raise OSError( + f"EMBEDDED_FILE_CHECK_SENTINEL at {file_check_path} " + + ("f" * 1_000) + ) + + monkeypatch.setattr(Path, "is_file", fail_file_check) + + def import_module(name: str) -> Any: + assert name == "codex_cli_bin" + if failure == "module": + raise ModuleNotFoundError( + f"EMBEDDED_MODULE_SENTINEL at {drive_path} " + ("i" * 1_000) + ) + return module + + helper = bundled_path + module = types.SimpleNamespace(bundled_codex_path=helper) + + monkeypatch.setattr(behavior_module.importlib.metadata, "version", fail_metadata) + monkeypatch.setattr(behavior_module.importlib, "import_module", import_module) + monkeypatch.setattr( + sys, + "argv", + ["probe", "openai-codex-cli-bin", "1.2.3", "candidate"], + ) + + exec(behavior_module._PROBE_SCRIPT, {}) + + [result] = json.loads(capsys.readouterr().out) + assert result["probe"] == "binary-distribution" + assert result["status"] == "fail" + assert expected_summary in result["summary"] + assert str(tmp_path) not in result["summary"] + assert drive_path not in result["summary"] + assert unc_path not in result["summary"] + if secret_suffix is not None: + assert secret_suffix not in result["summary"] + assert secret_suffix not in json.dumps(result["details"]) + assert "" in result["summary"] + assert result["details"]["error"] == result["summary"] + assert len(result["summary"]) <= 240 + assert len(result["details"]["error"]) <= 240 + + def test_behavior_summary_accepts_valid_no_update_and_unchanged_evidence() -> None: baseline = _probe( "claude-agent-sdk", @@ -1535,6 +1867,73 @@ def test_report_persists_and_renders_recomputed_behavior_summary(tmp_path: Path) assert "failed the required contract" in report +def test_snapshot_module_mapping_preserves_distribution_identities() -> None: + expected = { + "claude-agent-sdk": "claude_agent_sdk", + "openai-codex": "openai_codex", + "openai-codex-cli-bin": "codex_cli_bin", + "google-antigravity": "google.antigravity", + } + + assert {package: DEFAULT_MODULES[package] for package in expected} == expected + + +def test_current_codex_cli_snapshot_uses_real_import_module( + monkeypatch: pytest.MonkeyPatch, +) -> None: + module = types.ModuleType("codex_cli_bin") + module.bundled_codex_path = lambda: Path("unused") + monkeypatch.setitem(sys.modules, "codex_cli_bin", module) + + snapshot = snapshot_current_api("openai-codex-cli-bin", version="1.2.3") + + assert snapshot.package == "openai-codex-cli-bin" + assert snapshot.module == "codex_cli_bin" + assert snapshot.import_error is None + + +def test_isolated_codex_cli_snapshot_receives_real_import_module( + monkeypatch: pytest.MonkeyPatch, +) -> None: + calls: list[tuple[Any, ...]] = [] + + def fake_run(args: Any, **kwargs: Any) -> Any: + del kwargs + calls.append(tuple(args)) + payload = ( + '{"package":"openai-codex-cli-bin","version":"1.2.3",' + '"module":"codex_cli_bin","members":[],"import_error":null}' + ) + return types.SimpleNamespace( + stdout=payload if len(calls) == 3 else "", + stderr="", + returncode=0, + ) + + monkeypatch.setattr("examples.sdk_evolution_agent.snapshots.subprocess.run", fake_run) + + snapshot = snapshot_candidate_in_venv("openai-codex-cli-bin", "1.2.3") + + assert calls[1][-1] == "openai-codex-cli-bin==1.2.3" + assert calls[2][-1] == "codex_cli_bin" + assert snapshot.package == "openai-codex-cli-bin" + assert snapshot.module == "codex_cli_bin" + assert snapshot.import_error is None + + +def test_real_codex_cli_snapshot_and_probe_when_extra_is_installed() -> None: + pytest.importorskip("codex_cli_bin", reason="Codex extra is not installed") + + snapshot = snapshot_current_api("openai-codex-cli-bin") + (probe,) = probe_current_package("openai-codex-cli-bin", version=snapshot.version) + + assert snapshot.module == "codex_cli_bin" + assert snapshot.import_error is None + assert probe.probe == "binary-distribution" + assert probe.status == "pass" + assert probe.details["bundled_binary"] == "regular-file" + + def test_snapshot_and_diff_public_api(monkeypatch: pytest.MonkeyPatch) -> None: module = types.ModuleType("fake_sdk") From d649ffb57bb1588a5b452f125b47870af979d24e Mon Sep 17 00:00:00 2001 From: Eloi Date: Tue, 28 Jul 2026 20:09:12 +0200 Subject: [PATCH 9/9] Make SDK evolution operator workflow actionable (#55) ## Summary - make every supported operator command use a locked uv environment and the matching runtime extra - require explicit candidate inspection for actionable report and implementation passes - document deterministic behavior-summary handoff and stop conditions - add structural command-conformance tests that reject bare, compound, commented, echoed, or runtime/extra-mismatched examples ## Stack This is PR 5 of 5 and is based on agent/sdk-evolution-cli-bin-snapshots. ## Validation - uv lock --check - uv run --locked ruff check . - uv run --locked mypy - uv run --locked pytest -q: 491 passed, 8 skipped - focused operator-doc conformance: 14 passed - ordinary correctness review: clean after two repair passes --- .claude/commands/agent-runtime-kit/upgrade.md | 49 ++- .../skills/agent-runtime-kit-upgrade/SKILL.md | 48 ++- docs/sdk-evolution-agent-design.md | 63 +++- docs/sdk-evolution-agent.md | 55 ++- tests/test_sdk_evolution_docs.py | 353 ++++++++++++++++++ 5 files changed, 497 insertions(+), 71 deletions(-) create mode 100644 tests/test_sdk_evolution_docs.py diff --git a/.claude/commands/agent-runtime-kit/upgrade.md b/.claude/commands/agent-runtime-kit/upgrade.md index d19d38a..890c9a5 100644 --- a/.claude/commands/agent-runtime-kit/upgrade.md +++ b/.claude/commands/agent-runtime-kit/upgrade.md @@ -67,11 +67,15 @@ If unspecified, inspect all packages: ```bash gh auth status env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ - uv run python -m examples.sdk_evolution_agent --help + uv run --locked python -m examples.sdk_evolution_agent --help ``` 5. Resolve the runtime that will run the AI-backed stages. Use `claude-agent-sdk` unless the user explicitly selected another runtime. + Change the runtime and uv extra together: + - `claude-agent-sdk` -> `--extra claude` + - `codex-agent-sdk` -> `--extra codex` + - `antigravity-agent-sdk` -> `--extra antigravity` 6. Verify provider auth through supported mechanisms only: - Claude: Anthropic API key, Claude Code auth, or Claude Code provider @@ -89,7 +93,7 @@ If unspecified, inspect all packages: ```bash env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ - uv run --extra codex python -m examples.sdk_evolution_agent.auth ensure-codex + uv run --locked --extra codex python -m examples.sdk_evolution_agent.auth ensure-codex ``` The helper creates `~/.codex_agent_runtime_sdk`, removes uv freshness cutoff @@ -99,9 +103,9 @@ If unspecified, inspect all packages: and refresh the normal Codex login cache: ```bash - uv run --extra codex codex login --device-auth + uv run --locked --extra codex codex login --device-auth env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ - uv run --extra codex python -m examples.sdk_evolution_agent.auth ensure-codex + uv run --locked --extra codex python -m examples.sdk_evolution_agent.auth ensure-codex ``` ## Report-Only Evidence Pass @@ -111,17 +115,22 @@ upstream SDK releases are the point of this workflow: ```bash env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ - uv run python -m examples.sdk_evolution_agent \ + uv run --locked --extra claude python -m examples.sdk_evolution_agent \ --runtime claude-agent-sdk \ --refresh-preview \ + --inspect-candidates \ --package claude-agent-sdk \ --package openai-codex \ --package openai-codex-cli-bin \ --package google-antigravity ``` -If the user chose another runtime, replace only the `--runtime` value. Do not -add direct model calls. +`--inspect-candidates` is explicit consent to install and import a missing or +drifted locked baseline and resolver-selected candidates in credential-scrubbed +temporary environments. + +If the user chose another runtime, replace both the `--runtime` value and the +matching uv extra using the mapping above. Do not add direct model calls. Inspect the newest `reports/sdk-evolution//` directory and summarize: @@ -130,6 +139,7 @@ Inspect the newest `reports/sdk-evolution//` directory and summarize: - `api_diffs.json` - `behavior_probes.json` - `behavior_diffs.json` +- `behavior_summary.json` - `current_state.json` - `direction_analysis.json` - `architecture_decision.json` @@ -140,7 +150,8 @@ Stop before implementation if any of these are true: - required candidate API diffs are missing, - required release-note evidence could not be collected, -- `behavior_diffs.json` contains breaking adapter-contract drift, +- `behavior_summary.json` is missing, malformed, has an unknown status, or + reports `fail` / `incomplete`, - `architecture_decision.json` has `manual_design_required: true`, - the reviewer rejects the evidence or design, - recursive self-adaptation is required and the report does not include a safe @@ -148,6 +159,11 @@ Stop before implementation if any of these are true: adapters, output schemas, event sinks, permission profiles, or `AgentResult`. +`pass` means complete unchanged evidence; `changed` means complete +non-breaking evidence; `incomplete` means required observations could not be +proved; and `fail` means a required contract failed or a breaking diff was +observed. + ## Implementation Pass Only run implementation when the report-only pass supports it and the user wants @@ -157,9 +173,10 @@ an upgrade branch or PR: BRANCH="sdk-evolution-upgrade-$(date +%Y%m%d-%H%M%S)" env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ - uv run python -m examples.sdk_evolution_agent \ + uv run --locked --extra claude python -m examples.sdk_evolution_agent \ --runtime claude-agent-sdk \ --refresh-preview \ + --inspect-candidates \ --implementation-enabled \ --create-branch \ --branch-name "$BRANCH" \ @@ -173,7 +190,8 @@ env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ --package google-antigravity ``` -If the user chose another runtime, replace only the `--runtime` value. +If the user chose another runtime, replace both the `--runtime` value and the +matching uv extra using the mapping above. ## Verification @@ -181,12 +199,13 @@ After implementation, run or verify: ```bash env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv lock --check -env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run ruff check . -env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run mypy -env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run pytest +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run --locked ruff check . +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run --locked mypy +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run --locked pytest ``` If a draft PR was created, watch CI until it finishes or clearly report that it is still running. Include the PR URL, report path, changed SDK versions, -architecture decision, reviewer result, test results, uncertainty, and manual -review checklist in the final response. +`behavior_summary.json` status and reasons, architecture decision, reviewer +result, test results, uncertainty, and manual review checklist in the final +response. diff --git a/.codex/skills/agent-runtime-kit-upgrade/SKILL.md b/.codex/skills/agent-runtime-kit-upgrade/SKILL.md index fc52cc8..ad6f43b 100644 --- a/.codex/skills/agent-runtime-kit-upgrade/SKILL.md +++ b/.codex/skills/agent-runtime-kit-upgrade/SKILL.md @@ -39,11 +39,15 @@ Default runtime for this Codex skill: `codex-agent-sdk`. ```bash gh auth status env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ - uv run python -m examples.sdk_evolution_agent --help + uv run --locked python -m examples.sdk_evolution_agent --help ``` 4. Resolve the runtime that will run the AI-backed stages. Use `codex-agent-sdk` unless the user explicitly selected another runtime. + Change the runtime and uv extra together: + - `claude-agent-sdk` -> `--extra claude` + - `codex-agent-sdk` -> `--extra codex` + - `antigravity-agent-sdk` -> `--extra antigravity` 5. Use supported provider auth only: - Claude: Anthropic API key, Claude Code auth, or Claude Code provider @@ -61,7 +65,7 @@ Default runtime for this Codex skill: `codex-agent-sdk`. ```bash env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ - uv run --extra codex python -m examples.sdk_evolution_agent.auth ensure-codex + uv run --locked --extra codex python -m examples.sdk_evolution_agent.auth ensure-codex ``` The helper creates `~/.codex_agent_runtime_sdk`, removes uv freshness cutoff @@ -71,9 +75,9 @@ Default runtime for this Codex skill: `codex-agent-sdk`. and refresh the normal Codex login cache: ```bash - uv run --extra codex codex login --device-auth + uv run --locked --extra codex codex login --device-auth env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ - uv run --extra codex python -m examples.sdk_evolution_agent.auth ensure-codex + uv run --locked --extra codex python -m examples.sdk_evolution_agent.auth ensure-codex ``` ## Report-Only First @@ -83,17 +87,23 @@ freshness cutoffs: ```bash env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ - uv run python -m examples.sdk_evolution_agent \ + uv run --locked --extra codex python -m examples.sdk_evolution_agent \ --runtime codex-agent-sdk \ --refresh-preview \ + --inspect-candidates \ --package claude-agent-sdk \ --package openai-codex \ --package openai-codex-cli-bin \ --package google-antigravity ``` -If the user explicitly chooses another runtime, replace only the `--runtime` -value. Codex-backed runs should use the runner's built-in `gpt-5.5` and +`--inspect-candidates` is explicit consent to install and import a missing or +drifted locked baseline and resolver-selected candidates in credential-scrubbed +temporary environments. + +If the user explicitly chooses another runtime, replace both the `--runtime` +value and the matching uv extra using the mapping above. Codex-backed runs +should use the runner's built-in `gpt-5.5` and `reasoning_effort=xhigh` policy; do not implement model selection outside the runner. @@ -104,6 +114,7 @@ Inspect the newest `reports/sdk-evolution//` directory: - `api_diffs.json` - `behavior_probes.json` - `behavior_diffs.json` +- `behavior_summary.json` - `current_state.json` - `direction_analysis.json` - `architecture_decision.json` @@ -111,9 +122,13 @@ Inspect the newest `reports/sdk-evolution//` directory: - `report.md` Stop before implementation when candidate API diffs are missing, required -release-note evidence is missing, behavior probes show breaking adapter-contract -drift, `manual_design_required` is true, the reviewer rejects the evidence or -design, or recursive self-adaptation lacks a safe migration plan. +release-note evidence is missing, `behavior_summary.json` is missing, malformed, +has an unknown status, or reports `fail` / `incomplete`, +`manual_design_required` is true, the reviewer rejects the evidence or design, +or recursive self-adaptation lacks a safe migration plan. `pass` means complete +unchanged evidence; `changed` means complete non-breaking evidence; +`incomplete` means required observations could not be proved; and `fail` means +a required contract failed or a breaking diff was observed. Recursive self-adaptation means the upgrade affects the runner's own use of `AgentTask`, `RuntimeRegistry`, adapters, output schemas, event sinks, @@ -130,9 +145,10 @@ an upgrade branch or PR: BRANCH="sdk-evolution-upgrade-$(date +%Y%m%d-%H%M%S)" env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ - uv run python -m examples.sdk_evolution_agent \ + uv run --locked --extra codex python -m examples.sdk_evolution_agent \ --runtime codex-agent-sdk \ --refresh-preview \ + --inspect-candidates \ --implementation-enabled \ --create-branch \ --branch-name "$BRANCH" \ @@ -152,12 +168,12 @@ Run or verify: ```bash env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv lock --check -env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run ruff check . -env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run mypy -env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run pytest +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run --locked ruff check . +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run --locked mypy +env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE uv run --locked pytest ``` If a draft PR was created, watch CI until it finishes or clearly report that it is still running. Final output should include the PR URL, report path, changed -SDK versions, architecture decision, reviewer result, test results, uncertainty, -and manual review checklist. +SDK versions, `behavior_summary.json` status and reasons, architecture decision, +reviewer result, test results, uncertainty, and manual review checklist. diff --git a/docs/sdk-evolution-agent-design.md b/docs/sdk-evolution-agent-design.md index 6ad1166..b5f20d9 100644 --- a/docs/sdk-evolution-agent-design.md +++ b/docs/sdk-evolution-agent-design.md @@ -150,27 +150,34 @@ evidence, but they should not invent evidence that was not collected. The default command should be report-only: ```bash -python -m examples.sdk_evolution_agent --runtime fake --refresh-preview +uv run --locked python -m examples.sdk_evolution_agent \ + --runtime fake \ + --refresh-preview ``` This mode collects evidence, writes artifacts, runs the analysis stages through -the selected runtime, and stops before editing the workspace. The fake runtime is -allowed only as a deterministic development harness. It proves the pipeline and -schemas, not the quality of AI reasoning. +the selected runtime, and stops before editing the workspace. The fake runtime +is allowed only as a deterministic pipeline-shape harness. It proves the +pipeline and schemas, not upgrade safety or the quality of AI reasoning. When +updates exist, the default no-inspection run records incomplete candidate +evidence. A real analysis run should select one configured runtime: ```bash -python -m examples.sdk_evolution_agent --runtime claude-agent-sdk --refresh-preview -python -m examples.sdk_evolution_agent --runtime codex-agent-sdk --refresh-preview -python -m examples.sdk_evolution_agent --runtime antigravity-agent-sdk --refresh-preview +uv run --locked --extra claude python -m examples.sdk_evolution_agent \ + --runtime claude-agent-sdk --refresh-preview --inspect-candidates +uv run --locked --extra codex python -m examples.sdk_evolution_agent \ + --runtime codex-agent-sdk --refresh-preview --inspect-candidates +uv run --locked --extra antigravity python -m examples.sdk_evolution_agent \ + --runtime antigravity-agent-sdk --refresh-preview --inspect-candidates ``` Before a Codex-backed run, prepare the dedicated SDK evolution auth home: ```bash env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ - uv run --extra codex python -m examples.sdk_evolution_agent.auth ensure-codex + uv run --locked --extra codex python -m examples.sdk_evolution_agent.auth ensure-codex ``` This helper owns the operator-readiness boundary for Codex-backed runs. It @@ -188,9 +195,10 @@ Package filters narrow evidence collection for debugging, but normal evolution runs should inspect all tracked packages: ```bash -python -m examples.sdk_evolution_agent \ +uv run --locked --extra antigravity python -m examples.sdk_evolution_agent \ --runtime antigravity-agent-sdk \ --refresh-preview \ + --inspect-candidates \ --package claude-agent-sdk \ --package openai-codex \ --package openai-codex-cli-bin \ @@ -201,15 +209,18 @@ python -m examples.sdk_evolution_agent \ since update candidates without candidate API snapshots are not actionable. The shipped behavior deliberately inverts that default: candidate inspection installs and imports freshly downloaded upstream code, so it is opt-in and -runs in a credential-scrubbed environment, with explicit `skip` records when -it is off. +runs in a credential-scrubbed environment. The same explicit flag is required +to install a locked baseline that is missing or drifted in the active +environment. When inspection is off, those gaps become explicit `skip` records +rather than claimed observations. Implementation mode should remain explicitly gated: ```bash -python -m examples.sdk_evolution_agent \ +uv run --locked --extra antigravity python -m examples.sdk_evolution_agent \ --runtime antigravity-agent-sdk \ --refresh-preview \ + --inspect-candidates \ --implementation-enabled ``` @@ -347,6 +358,7 @@ Behavior probe output should be a first-class report artifact, for example: ```text behavior_probes.json behavior_diffs.json +behavior_summary.json ``` Each probe result should include: @@ -358,15 +370,21 @@ Each probe result should include: - stdout/stderr summary, - skipped reason when optional credentials are missing. -`behavior_diffs.json` compares current-environment probes against -candidate-version probes for resolver-selected updates. Breaking candidate probe -changes block implementation deterministically before any local lock update. +`behavior_diffs.json` compares observed locked-baseline probes against +candidate-version probes for resolver-selected updates. Breaking candidate +probe changes block implementation deterministically before any local lock +update. `behavior_probes.json` may include observed SDK fields or parameters that are not part of the adapter contract. `behavior_diffs.json` compares the required adapter contract, not every optional field. Public API and signature churn remains visible in `api_diffs.json` and probe details, but it should only block implementation when the required behavior contract fails or becomes ambiguous. +`behavior_summary.json` is the deterministic hand-off: `pass` means complete +unchanged evidence, `changed` means complete non-breaking evidence, +`incomplete` means required observations could not be proved, and `fail` means +a required contract failed or a breaking diff was observed. Missing, malformed, +or unknown summary states also block implementation. ### 5. Runtime-Generated Analysis @@ -518,6 +536,7 @@ api_snapshots/ api_diffs.json behavior_probes.json behavior_diffs.json +behavior_summary.json current_state.json direction_analysis.json architecture_decision.json @@ -552,6 +571,7 @@ artifact-aware. It should record: - paths or content hashes for current API snapshots, - paths or content hashes for release-note evidence, - paths or content hashes for behavior probe results, +- a path or content hash for the deterministic behavior summary, - whether the baseline was promoted, refreshed, skipped, or blocked. Promotion rules should be conservative: @@ -631,13 +651,16 @@ The example implements the deterministic evidence artifacts described above: - `behavior_probes.json` records current and candidate adapter-contract probes. - `behavior_diffs.json` records behavior differences between current and candidate probes. +- `behavior_summary.json` records the recomputed deterministic status, counts, + and reasons handed to implementation guards and operators. - `current_state.json` records the run baseline, lockfile hash, accepted package versions, artifact hashes, and promotion status. The implementation path is gated by deterministic checks before the local lockfile update runs. Missing candidate API diffs, unavailable required -release-note evidence, breaking behavior diffs, reviewer rejection, -`manual_design_required`, and unresolved recursive self-adaptation all block -implementation. When implementation is allowed, the example applies the -resolver-selected SDK lock update locally, runs verification, writes the report -artifacts, commits them, pushes the branch, and opens a draft PR when configured. +release-note evidence, behavior summaries that are failed, incomplete, missing, +malformed, or unknown, reviewer rejection, `manual_design_required`, and +unresolved recursive self-adaptation all block implementation. When +implementation is allowed, the example applies the resolver-selected SDK lock +update locally, runs verification, writes the report artifacts, commits them, +pushes the branch, and opens a draft PR when configured. diff --git a/docs/sdk-evolution-agent.md b/docs/sdk-evolution-agent.md index 03b43c1..8d2c8bf 100644 --- a/docs/sdk-evolution-agent.md +++ b/docs/sdk-evolution-agent.md @@ -11,16 +11,22 @@ changelog strategy, caveats, and alternatives, see Run it from the repository: ```bash -python -m examples.sdk_evolution_agent --runtime fake +uv run --locked python -m examples.sdk_evolution_agent --runtime fake ``` The `fake` runtime is deterministic and useful for checking the local pipeline -without credentials. For real AI reasoning, select a configured runtime: +without credentials. It proves pipeline shape only, never upgrade safety. When +updates exist, its default no-inspection run records incomplete candidate +evidence. For real AI reasoning, select a configured runtime and its matching +uv extra: ```bash -python -m examples.sdk_evolution_agent --runtime claude-agent-sdk -python -m examples.sdk_evolution_agent --runtime codex-agent-sdk -python -m examples.sdk_evolution_agent --runtime antigravity-agent-sdk +uv run --locked --extra claude python -m examples.sdk_evolution_agent \ + --runtime claude-agent-sdk --refresh-preview --inspect-candidates +uv run --locked --extra codex python -m examples.sdk_evolution_agent \ + --runtime codex-agent-sdk --refresh-preview --inspect-candidates +uv run --locked --extra antigravity python -m examples.sdk_evolution_agent \ + --runtime antigravity-agent-sdk --refresh-preview --inspect-candidates ``` Every AI-backed stage is dispatched as an `AgentTask` through a runtime resolved @@ -37,7 +43,7 @@ Run the auth preflight before real Codex-backed SDK evolution runs: ```bash env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ - uv run --extra codex python -m examples.sdk_evolution_agent.auth ensure-codex + uv run --locked --extra codex python -m examples.sdk_evolution_agent.auth ensure-codex ``` The helper checks `codex login status` against the dedicated home and can refresh @@ -46,9 +52,9 @@ the helper reports that the isolated home is not authenticated, refresh the normal Codex login cache and rerun the helper: ```bash -uv run --extra codex codex login --device-auth +uv run --locked --extra codex codex login --device-auth env -u UV_EXCLUDE_NEWER -u UV_EXCLUDE_NEWER_PACKAGE \ - uv run --extra codex python -m examples.sdk_evolution_agent.auth ensure-codex + uv run --locked --extra codex python -m examples.sdk_evolution_agent.auth ensure-codex ``` Codex-backed SDK evolution runs explicitly choose `gpt-5.5` with @@ -113,16 +119,18 @@ cutoff variables must not hide candidate releases. ## Candidate API Inspection -The command treats `uv.lock` as the current baseline. If the active `.venv` -contains a different installed version, the agent inspects the locked baseline -in a temporary isolated virtualenv instead of trusting the drifted environment. +The command treats `uv.lock` as the current baseline. If the locked SDK is +missing from the active `.venv` or the installed version differs, the agent can +inspect the locked baseline in a temporary isolated virtualenv instead of +trusting the missing or drifted environment. When a refresh preview is available, package update candidates come from the resolver's `uv lock --dry-run -P ...` output, not only from PyPI's `latest` -metadata. With `--inspect-candidates`, the agent installs each resolver update -candidate in a temporary isolated virtualenv — with a credential-scrubbed -environment (throwaway `HOME`, `PATH` only) — and writes an API snapshot plus -`api_diffs.json` entry, and runs the behavior probes against the candidate the -same way. This avoids false downgrade diffs for packages whose locked +metadata. With `--inspect-candidates`, the agent installs each missing or +drifted locked baseline and each resolver update candidate in a temporary +isolated virtualenv — with a credential-scrubbed environment (throwaway `HOME`, +`PATH` only) — and writes API snapshots plus an `api_diffs.json` entry, and runs +the behavior probes against the candidate the same way. This avoids false +downgrade diffs for packages whose locked prerelease is newer than PyPI's stable latest field. Candidate inspection is opt-in because it executes freshly downloaded upstream code; without the flag, candidates are recorded as explicit `skip` entries rather than silently @@ -163,7 +171,11 @@ a stale or contradictory cached summary is never presented as the run result. Report-only mode is the default. To allow the implementation stage, pass: ```bash -python -m examples.sdk_evolution_agent --runtime claude-agent-sdk --implementation-enabled +uv run --locked --extra claude python -m examples.sdk_evolution_agent \ + --runtime claude-agent-sdk \ + --refresh-preview \ + --inspect-candidates \ + --implementation-enabled ``` Implementation is still blocked when: @@ -172,8 +184,9 @@ Implementation is still blocked when: - the reviewer rejects the evidence or design, - a resolver-selected update lacks a candidate API diff, - required release-note evidence could not be collected, -- behavior evidence is failed, incomplete, malformed, internally inconsistent, - or uses the wrong package/version transition, +- `behavior_summary.json` is missing, malformed, has an unknown status, reports + `fail` / `incomplete`, is internally inconsistent, or uses the wrong + package/version transition, - required structured output or permission behavior is unsupported by the selected runtime, - recursive self-adaptation is required but no safe migration plan exists. @@ -190,8 +203,10 @@ design review. Draft PR creation is opt-in: ```bash -python -m examples.sdk_evolution_agent \ +uv run --locked --extra claude python -m examples.sdk_evolution_agent \ --runtime claude-agent-sdk \ + --refresh-preview \ + --inspect-candidates \ --implementation-enabled \ --create-branch \ --branch-name sdk-evolution-update \ diff --git a/tests/test_sdk_evolution_docs.py b/tests/test_sdk_evolution_docs.py new file mode 100644 index 0000000..e3c7ab8 --- /dev/null +++ b/tests/test_sdk_evolution_docs.py @@ -0,0 +1,353 @@ +from __future__ import annotations + +import shlex +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[1] +CODEX_RUNBOOK = Path(".codex/skills/agent-runtime-kit-upgrade/SKILL.md") +CLAUDE_RUNBOOK = Path(".claude/commands/agent-runtime-kit/upgrade.md") +PUBLIC_GUIDE = Path("docs/sdk-evolution-agent.md") +DESIGN_GUIDE = Path("docs/sdk-evolution-agent-design.md") +AUTHORITATIVE_DOCS = (CODEX_RUNBOOK, CLAUDE_RUNBOOK, PUBLIC_GUIDE, DESIGN_GUIDE) +RUNBOOKS = (CODEX_RUNBOOK, CLAUDE_RUNBOOK) +PUBLIC_DOCS = (PUBLIC_GUIDE, DESIGN_GUIDE) +RUNTIME_EXTRAS = { + "claude-agent-sdk": "claude", + "codex-agent-sdk": "codex", + "antigravity-agent-sdk": "antigravity", +} +AGENT_MODULE = ("python", "-m", "examples.sdk_evolution_agent") +AUTH_MODULE = ("python", "-m", "examples.sdk_evolution_agent.auth") + + +def test_sdk_evolution_bash_commands_use_locked_uv() -> None: + for path in AUTHORITATIVE_DOCS: + for command in _bash_commands(path): + _assert_locked_command(path, command) + + +def test_real_sdk_evolution_commands_match_runtime_extras() -> None: + for path in AUTHORITATIVE_DOCS: + for command in _agent_commands(path): + _assert_real_runtime_command(path, command) + + +def test_public_guides_cover_every_runtime_extra_mapping() -> None: + for path in PUBLIC_DOCS: + commands = _agent_commands(path) + observed = { + runtime + for command in commands + if (runtime := _option_value(command, "--runtime")) in RUNTIME_EXTRAS + } + assert observed == set(RUNTIME_EXTRAS), f"incomplete runtime mapping in {path}" + + for path in RUNBOOKS: + text = _read(path) + for runtime, extra in RUNTIME_EXTRAS.items(): + assert f"`{runtime}` -> `--extra {extra}`" in text + + +def test_runbooks_include_report_and_implementation_commands() -> None: + defaults = { + CODEX_RUNBOOK: "codex-agent-sdk", + CLAUDE_RUNBOOK: "claude-agent-sdk", + } + for path, runtime in defaults.items(): + commands = [ + command + for command in _agent_commands(path) + if _option_value(command, "--runtime") == runtime + ] + assert any("--implementation-enabled" not in command for command in commands) + assert any("--implementation-enabled" in command for command in commands) + + +def test_fake_commands_are_pipeline_shape_checks_only() -> None: + for path in PUBLIC_DOCS: + commands = [ + command + for command in _agent_commands(path) + if _option_value(command, "--runtime") == "fake" + ] + assert commands, f"missing fake pipeline-shape command in {path}" + for command in commands: + assert "--inspect-candidates" not in command + assert "--implementation-enabled" not in command + assert "--draft-pr" not in command + prose = _read(path).lower().replace("-", " ") + assert "pipeline shape" in prose + assert "never upgrade safety" in prose or "not upgrade safety" in prose + + +def test_candidate_inspection_remains_explicit_opt_in() -> None: + for path in AUTHORITATIVE_DOCS: + prose = " ".join(_read(path).lower().split()) + assert "--inspect-candidates" in prose + assert _has_candidate_opt_in(prose) + + +def test_behavior_summary_is_in_artifacts_and_runbook_handoffs() -> None: + for path in AUTHORITATIVE_DOCS: + text = _read(path) + assert "behavior_summary.json" in text + for status in ("pass", "changed", "incomplete", "fail"): + assert f"`{status}`" in text + + for path in RUNBOOKS: + text = _read(path) + normalized = " ".join(text.split()) + assert "`behavior_summary.json` status and reasons" in text + assert "a missing or drifted locked baseline" in normalized + assert "credential-scrubbed" in text + for blocker in ("missing", "malformed", "unknown status", "`fail`", "`incomplete`"): + assert blocker in text + + +def test_codex_auth_commands_use_the_locked_codex_extra() -> None: + for path in AUTHORITATIVE_DOCS: + for command in _bash_commands(path): + payload = _command_payload(command) + if _starts_with(payload, AUTH_MODULE) or _starts_with(payload, ("codex", "login")): + assert _starts_with( + _command_argv(command), + ("uv", "run", "--locked", "--extra", "codex"), + ) + + +@pytest.mark.parametrize( + "body", + ( + "uv run --locked python -m examples.sdk_evolution_agent " + "--runtime=codex-agent-sdk", + "COMPLIANT='uv run --locked --extra codex python -m " + "examples.sdk_evolution_agent' python -m examples.sdk_evolution_agent " + "--runtime codex-agent-sdk --refresh-preview --inspect-candidates", + "uv run --locked python -m examples.sdk_evolution_agent --help && " + "python -m examples.sdk_evolution_agent --runtime codex-agent-sdk " + "--refresh-preview --inspect-candidates", + ), +) +def test_noncompliant_shell_forms_cannot_evade_locked_command_checks( + tmp_path: Path, body: str +) -> None: + path = tmp_path / "commands.md" + path.write_text(f"```bash\n{body}\n```\n", encoding="utf-8") + commands = _bash_commands(path) + with pytest.raises(AssertionError): + for command in commands: + _assert_locked_command(path, command) + _assert_real_runtime_command(path, command) + + +def test_runtime_equals_form_cannot_evade_runtime_extra_checks(tmp_path: Path) -> None: + path = tmp_path / "commands.md" + path.write_text( + "```bash\n" + "uv run --locked python -m examples.sdk_evolution_agent " + "--runtime=codex-agent-sdk\n" + "```\n", + encoding="utf-8", + ) + command = _agent_commands(path)[0] + assert _option_value(command, "--runtime") == "codex-agent-sdk" + with pytest.raises(AssertionError, match="runtime/extra mismatch"): + _assert_real_runtime_command(path, command) + + +def test_echoes_and_comments_do_not_count_as_executable_commands(tmp_path: Path) -> None: + path = tmp_path / "commands.md" + path.write_text( + "```bash\n" + "echo uv run --locked --extra codex python -m " + "examples.sdk_evolution_agent --runtime codex-agent-sdk " + "--refresh-preview --inspect-candidates\n" + "# uv run --locked --extra codex python -m " + "examples.sdk_evolution_agent --runtime codex-agent-sdk " + "--refresh-preview --inspect-candidates\n" + "```\n", + encoding="utf-8", + ) + assert _agent_commands(path) == [] + + +def test_unrelated_opt_in_prose_does_not_satisfy_candidate_consent() -> None: + prose = "--inspect-candidates is automatic. draft pr creation is opt-in." + assert not _has_candidate_opt_in(prose) + + +def _read(path: Path) -> str: + return (ROOT / path).read_text(encoding="utf-8") + + +def _agent_commands(path: Path) -> list[list[str]]: + return [ + command + for command in _bash_commands(path) + if _starts_with(_command_payload(command), AGENT_MODULE) + ] + + +def _bash_commands(path: Path) -> list[list[str]]: + commands: list[list[str]] = [] + pending: list[str] = [] + in_bash = False + for raw_line in _read(path).splitlines(): + stripped = raw_line.strip() + if stripped == "```bash": + in_bash = True + pending = [] + continue + if in_bash and stripped == "```": + if pending: + commands.extend(_split_shell_commands(" ".join(pending))) + in_bash = False + pending = [] + continue + if not in_bash: + continue + if not stripped: + if pending: + commands.extend(_split_shell_commands(" ".join(pending))) + pending = [] + continue + continued = stripped.endswith("\\") + pending.append(stripped[:-1].rstrip() if continued else stripped) + if not continued: + commands.extend(_split_shell_commands(" ".join(pending))) + pending = [] + return commands + + +def _split_shell_commands(command: str) -> list[list[str]]: + lexer = shlex.shlex(command, posix=True, punctuation_chars=";&|") + lexer.whitespace_split = True + lexer.commenters = "#" + commands: list[list[str]] = [] + pending: list[str] = [] + for token in lexer: + if token and set(token) <= {";", "&", "|"}: + if pending: + commands.append(pending) + pending = [] + continue + pending.append(token) + if pending: + commands.append(pending) + return commands + + +def _option_value(command: list[str], option: str) -> str | None: + for index, part in enumerate(command): + if part == option: + return command[index + 1] if index + 1 < len(command) else None + prefix = f"{option}=" + if part.startswith(prefix): + return part.removeprefix(prefix) or None + return None + + +def _assert_locked_command(path: Path, command: list[str]) -> None: + argv = _command_argv(command) + if _starts_with(argv, ("uv", "run")): + assert _starts_with(argv, ("uv", "run", "--locked")), ( + f"unlocked uv command in {path}: {_display(command)}" + ) + payload = _command_payload(command) + if _starts_with(payload, AGENT_MODULE): + assert _starts_with(argv, ("uv", "run", "--locked")), ( + f"bare actionable SDK evolution command in {path}: {_display(command)}" + ) + + +def _assert_real_runtime_command(path: Path, command: list[str]) -> None: + runtime = _option_value(command, "--runtime") + if runtime is None or runtime == "fake" or "--help" in command: + return + assert runtime in RUNTIME_EXTRAS, f"unknown runtime in {path}: {_display(command)}" + extra = RUNTIME_EXTRAS[runtime] + argv = _command_argv(command) + assert _starts_with(argv, ("uv", "run", "--locked", "--extra", extra)), ( + f"runtime/extra mismatch in {path}: {_display(command)}" + ) + assert _starts_with(_command_payload(command), AGENT_MODULE), ( + f"runtime command is not the SDK evolution entrypoint in {path}: {_display(command)}" + ) + assert "--refresh-preview" in command, ( + f"missing refresh in {path}: {_display(command)}" + ) + assert "--inspect-candidates" in command, ( + f"missing candidate inspection in {path}: {_display(command)}" + ) + + +def _command_argv(command: list[str]) -> list[str]: + index = 0 + while index < len(command) and _is_assignment(command[index]): + index += 1 + if index >= len(command) or command[index] != "env": + return command[index:] + + index += 1 + while index < len(command): + token = command[index] + if _is_assignment(token): + index += 1 + elif token in {"-u", "--unset"}: + index += 2 + elif token.startswith("--unset="): + index += 1 + elif token == "--": + index += 1 + break + else: + break + return command[index:] + + +def _command_payload(command: list[str]) -> list[str]: + argv = _command_argv(command) + if not _starts_with(argv, ("uv", "run")): + return argv + + index = 2 + while index < len(argv): + token = argv[index] + if token == "--": + return argv[index + 1 :] + if token == "--extra": + index += 2 + continue + if token == "--locked" or token.startswith("--extra="): + index += 1 + continue + break + return argv[index:] + + +def _is_assignment(token: str) -> bool: + name, separator, _value = token.partition("=") + return bool(separator and name and name.replace("_", "a").isalnum() and not name[0].isdigit()) + + +def _has_candidate_opt_in(prose: str) -> bool: + normalized = prose.replace("opt in", "opt-in") + sentences = normalized.split(".") + return ( + "`--inspect-candidates` is explicit consent" in normalized + or "candidate inspection is opt-in" in normalized + or any( + "candidate inspection" in sentence and "so it is opt-in" in sentence + for sentence in sentences + ) + ) + + +def _starts_with(command: list[str], sequence: tuple[str, ...]) -> bool: + return tuple(command[: len(sequence)]) == sequence + + +def _display(command: list[str]) -> str: + return shlex.join(command)