diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 2d4c6f3497..9e4eeccb6b 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -36,6 +36,9 @@ importers: sdk/typescript: devDependencies: + '@modelcontextprotocol/conformance': + specifier: github:modelcontextprotocol/conformance#49103de6ed70804e940637bf3e9e29e4a3f54e64 + version: https://codeload.github.com/modelcontextprotocol/conformance/tar.gz/49103de6ed70804e940637bf3e9e29e4a3f54e64 '@modelcontextprotocol/sdk': specifier: 1.26.0 version: 1.26.0(zod@3.25.76) @@ -74,7 +77,7 @@ importers: version: 10.9.2(@types/node@20.19.18)(typescript@5.9.2) tsup: specifier: ^8.5.0 - version: 8.5.0(postcss@8.5.6)(typescript@5.9.2)(yaml@2.8.1) + version: 8.5.0(postcss@8.5.6)(typescript@5.9.2)(yaml@2.9.0) typescript: specifier: ^5.9.2 version: 5.9.2 @@ -572,6 +575,11 @@ packages: '@jridgewell/trace-mapping@0.3.9': resolution: {integrity: sha512-3Belt6tdc8bPgAtbcmdtNJlirVoTmEb5e2gC94PnkwEW9jI6CAHUeoG85tjWP5WquqfavoMtMwiG4P926ZKKuQ==} + '@modelcontextprotocol/conformance@https://codeload.github.com/modelcontextprotocol/conformance/tar.gz/49103de6ed70804e940637bf3e9e29e4a3f54e64': + resolution: {tarball: https://codeload.github.com/modelcontextprotocol/conformance/tar.gz/49103de6ed70804e940637bf3e9e29e4a3f54e64} + version: 0.2.0-alpha.10 + hasBin: true + '@modelcontextprotocol/sdk@1.26.0': resolution: {integrity: sha512-Y5RmPncpiDtTXDbLKswIJzTqu2hyBKxTNsgKqKclDbhIgg1wgtf1fRuvxgTnRfcnxtvvgbIEcqUOzZrJ6iSReg==} engines: {node: '>=18'} @@ -594,6 +602,58 @@ packages: resolution: {integrity: sha512-oGB+UxlgWcgQkgwo8GcEGwemoTFt3FIO9ababBmaGwXIoBKZ+GTy0pP185beGg7Llih/NSHSV2XAs1lnznocSg==} engines: {node: '>= 8'} + '@octokit/auth-token@6.0.0': + resolution: {integrity: sha512-P4YJBPdPSpWTQ1NU4XYdvHvXJJDxM6YwpS0FZHRgP7YFkdVxsWcpWGy/NVqlAA7PcPCnMacXlRm1y2PFZRWL/w==} + engines: {node: '>= 20'} + + '@octokit/core@7.0.6': + resolution: {integrity: sha512-DhGl4xMVFGVIyMwswXeyzdL4uXD5OGILGX5N8Y+f6W7LhC1Ze2poSNrkF/fedpVDHEEZ+PHFW0vL14I+mm8K3Q==} + engines: {node: '>= 20'} + + '@octokit/endpoint@11.0.3': + resolution: {integrity: sha512-FWFlNxghg4HrXkD3ifYbS/IdL/mDHjh9QcsNyhQjN8dplUoZbejsdpmuqdA76nxj2xoWPs7p8uX2SNr9rYu0Ag==} + engines: {node: '>= 20'} + + '@octokit/graphql@9.0.3': + resolution: {integrity: sha512-grAEuupr/C1rALFnXTv6ZQhFuL1D8G5y8CN04RgrO4FIPMrtm+mcZzFG7dcBm+nq+1ppNixu+Jd78aeJOYxlGA==} + engines: {node: '>= 20'} + + '@octokit/openapi-types@27.0.0': + resolution: {integrity: sha512-whrdktVs1h6gtR+09+QsNk2+FO+49j6ga1c55YZudfEG+oKJVvJLQi3zkOm5JjiUXAagWK2tI2kTGKJ2Ys7MGA==} + + '@octokit/plugin-paginate-rest@14.0.0': + resolution: {integrity: sha512-fNVRE7ufJiAA3XUrha2omTA39M6IXIc6GIZLvlbsm8QOQCYvpq/LkMNGyFlB1d8hTDzsAXa3OKtybdMAYsV/fw==} + engines: {node: '>= 20'} + peerDependencies: + '@octokit/core': '>=6' + + '@octokit/plugin-request-log@6.0.0': + resolution: {integrity: sha512-UkOzeEN3W91/eBq9sPZNQ7sUBvYCqYbrrD8gTbBuGtHEuycE4/awMXcYvx6sVYo7LypPhmQwwpUe4Yyu4QZN5Q==} + engines: {node: '>= 20'} + peerDependencies: + '@octokit/core': '>=6' + + '@octokit/plugin-rest-endpoint-methods@17.0.0': + resolution: {integrity: sha512-B5yCyIlOJFPqUUeiD0cnBJwWJO8lkJs5d8+ze9QDP6SvfiXSz1BF+91+0MeI1d2yxgOhU/O+CvtiZ9jSkHhFAw==} + engines: {node: '>= 20'} + peerDependencies: + '@octokit/core': '>=6' + + '@octokit/request-error@7.1.0': + resolution: {integrity: sha512-KMQIfq5sOPpkQYajXHwnhjCC0slzCNScLHs9JafXc4RAJI+9f+jNDlBNaIMTvazOPLgb4BnlhGJOTbnN0wIjPw==} + engines: {node: '>= 20'} + + '@octokit/request@10.0.11': + resolution: {integrity: sha512-+s7HUxjfFqOMS9VlIwDffq0MikjSAK0gSpG73W+meAvVAvX4MBrHYTK5Bj3Uot55qFT4gzUtfzE4mGWY4Br8/Q==} + engines: {node: '>= 20'} + + '@octokit/rest@22.0.1': + resolution: {integrity: sha512-Jzbhzl3CEexhnivb1iQ0KJ7s5vvjMWcmRtq5aUsKmKDrRW6z3r84ngmiFKFvpZjpiU/9/S6ITPFRpn5s/3uQJw==} + engines: {node: '>= 20'} + + '@octokit/types@16.0.0': + resolution: {integrity: sha512-sKq+9r1Mm4efXW1FCk7hFSeJo4QKreL/tTbR0rz/qx/r1Oa2VV83LTA/H/MuCOX7uCIJmQVRKBcbmWoySjAnSg==} + '@pkgjs/parseargs@0.11.0': resolution: {integrity: sha512-+1VkjdD0QBLPodGrJUeqarH8VAIvQODIbwh9XpP5Syisf7YoQgsJKPNFoqqLQlu+VQ/tVSshMR6loPMn8U+dPg==} engines: {node: '>=14'} @@ -893,6 +953,9 @@ packages: ajv@8.17.1: resolution: {integrity: sha512-B/gBuNg5SiMTrPkC+A2+cW0RszwxYmn6VYxB/inlBStS5nx6xHIt/ehKRhIMhqusl7a8LjQoZnjCs5vhwxOQ1g==} + ajv@8.20.0: + resolution: {integrity: sha512-Thbli+OlOj+iMPYFBVBfJ3OmCAnaSyNn4M1vz9T6Gka5Jt9ba/HIR56joy65tY6kx/FCF5VXNB819Y7/GUrBGA==} + ansi-escapes@4.3.2: resolution: {integrity: sha512-gKXj5ALrKWQLsYG9jlTRmR/xKluxHV+Z9QEwNIgCfM1/uwPMCuzVVnh5mwTd+OuBZcwSIMbqssNWRm1lE51QaQ==} engines: {node: '>=8'} @@ -969,6 +1032,9 @@ packages: resolution: {integrity: sha512-hY/u2lxLrbecMEWSB0IpGzGyDyeoMFQhCvZd2jGFSE5I17Fh01sYUBPCJtkWERw7zrac9+cIghxm/ytJa2X8iA==} hasBin: true + before-after-hook@4.0.0: + resolution: {integrity: sha512-q6tR3RPqIB1pMiTRMFcZwuG5T8vwp+vUvEG0vuI6B+Rikh5BfPp2fQ82c925FOs+b0lcFQ8CFrL+KbilfZFhOQ==} + body-parser@2.2.2: resolution: {integrity: sha512-oP5VkATKlNwcgvxi0vM0p/D3n2C3EReYVX+DNYs5TjZFn/oQt2j+4sVJtSMr18pdRr8wjTcBl6LoV+FUwzPmNA==} engines: {node: '>=18'} @@ -1073,6 +1139,10 @@ packages: color-name@1.1.4: resolution: {integrity: sha512-dOy+3AuW3a2wNbZHIuMZpTcgjGuLU/uBL/ubcZF9OXbDo8ff4O8yVp5Bf0efS8uEoYo5q4Fx7dY9OgQGXgAsQA==} + commander@14.0.3: + resolution: {integrity: sha512-H+y0Jo/T1RZ9qPP4Eh1pkcQcLRglraJaSLoyOtHxu6AapkjWVCy2Sit1QQ4x3Dng8qDlSsZEet7g5Pq06MvTgw==} + engines: {node: '>=20'} + commander@4.1.1: resolution: {integrity: sha512-NOKm8xhkzAjzFx8B2v5OAHT+u5pRQc2UCa2Vq9jYL/31o2wi9mxBA7LIFs3sV5VSC49z6pEhfbMULvShKj26WA==} engines: {node: '>= 6'} @@ -1095,6 +1165,10 @@ packages: resolution: {integrity: sha512-nTjqfcBFEipKdXCv4YDQWCfmcLZKm81ldF0pAopTvyrFGVbcR6P/VAAd5G7N+0tTr8QqiU0tFadD6FK4NtJwOA==} engines: {node: '>= 0.6'} + content-type@2.0.0: + resolution: {integrity: sha512-j/O/d7GcZCyNl7/hwZAb606rzqkyvaDctLmckbxLzHvFBzTJHuGEdodATcP3yIRoDrLHkIATJuvzbFlp/ki2cQ==} + engines: {node: '>=18'} + convert-source-map@2.0.0: resolution: {integrity: sha512-Kvp459HrV2FEJ1CAsi1Ku+MY3kasH19TFykTz2xWmMeq6bk2NU3XXvfJ+Q61m0xktWwt+1HSYf3JZsTms3aRJg==} @@ -1304,6 +1378,10 @@ packages: resolution: {integrity: sha512-Vo1ab+QXPzZ4tCa8SwIHJFaSzy4R6SHf7BY79rFBDf0idraZWAkYrDjDj8uWaSm3S2TK+hJ7/t1CEmZ7jXw+pg==} engines: {node: '>=18.0.0'} + eventsource-parser@3.1.0: + resolution: {integrity: sha512-kJezFj9YFAMLeORyi7aCLxLbD5/qWMQnoMVlVPyHIll7lgRJCc3JVln9Vgl9nwQi0YkMnhdGTMNn7CkRRAptMg==} + engines: {node: '>=18.0.0'} + eventsource@3.0.7: resolution: {integrity: sha512-CRT1WTyuQoD771GW56XEZFQ/ZoSfWid1alKGDYMmkt2yl8UXrVR4pspqWNEcqKvVIzg6PAltWjxcSSPrboA4iA==} engines: {node: '>=18.0.0'} @@ -1453,7 +1531,7 @@ packages: glob@7.2.3: resolution: {integrity: sha512-nFR0zLpU2YCaRxwoCJvL6UvCH2JFyFVIvwTLsIf21AuHlMskA1hhTdk+LlYJtOlYt9v6dvszD2BGRqBL+iQK9Q==} - deprecated: Glob versions prior to v9 are no longer supported + deprecated: Old versions of glob are not supported, and contain widely publicized security vulnerabilities, which have been fixed in the current version. Please update. Support for old versions may be purchased (at exorbitant rates) by contacting i@izs.me globals@14.0.0: resolution: {integrity: sha512-oahGvuMGQlPw/ivIYBjVSrWAfWLBeku5tpPE2fOPLi+WHffIWbuh2tCjhyQhTBPMf5E9jDEH4FOmTYgYwbKwtQ==} @@ -1775,6 +1853,9 @@ packages: json-stable-stringify-without-jsonify@1.0.1: resolution: {integrity: sha512-Bdboy+l7tA3OGW6FjyFHWkP5LuByj1Tk33Ljyq0axyzdk9//JSi2u3fP1QSmd1KNwq6VOKYGlAu87CisVir6Pw==} + json-with-bigint@3.5.10: + resolution: {integrity: sha512-Vcx+JVNEBts/xfcoCS69sKrOhOk/3TVlvlT+XzUOefVKnnrbYSCKpDCm10pohsJFtsJVYnwa/cXRZ4eElzaM6w==} + json5@2.2.3: resolution: {integrity: sha512-XmOWe7eyHYH14cLdVPoyg+GOH3rYX++KpzrylJwSW98t3Nk+U8XOl8FWKOgwtzdb8lXGf6zYwDUzeHMWfxasyg==} engines: {node: '>=6'} @@ -2093,10 +2174,6 @@ packages: pure-rand@6.1.0: resolution: {integrity: sha512-bVWawvoZoBYpp6yIoQtQXHZjmz35RSVHnUOTefl8Vcjr8snTPY1wnpSPMWekcFwbxI6gtmT7rSYPFvz71ldiOA==} - qs@6.14.0: - resolution: {integrity: sha512-YWWTjgABSKcvs/nWBi9PycY/JiPJqOD4JA6o9Sej2AtvSGarXxKC3OQSk4pAarbdQlKAh5D4FCQkJNkW+GAn3w==} - engines: {node: '>=0.6'} - qs@6.15.1: resolution: {integrity: sha512-6YHEFRL9mfgcAvql/XhwTvf5jKcOiiupt2FiJxHkiX1z4j7WL8J/jRHYLluORvc1XxB5rV20KoeK00gVJamspg==} engines: {node: '>=0.6'} @@ -2457,6 +2534,13 @@ packages: undici-types@6.21.0: resolution: {integrity: sha512-iwDZqg0QAGrg9Rav5H4n0M64c3mkR59cJ6wQp+7C4nI0gsmExaedaYLNO44eT4AtBBwjbTiGPMlt2Md0T9H9JQ==} + undici@7.28.0: + resolution: {integrity: sha512-cRZYrTDwWznlnRiPjggAGxZXanty6M8RV1ff8Wm4LWXBp7/IG8v5DnOm74DtUBp9OONpK75YlPnIjQqX0dBDtA==} + engines: {node: '>=20.18.1'} + + universal-user-agent@7.0.3: + resolution: {integrity: sha512-TmnEAEAsBJVZM/AADELsK76llnwcf9vMKuPz8JflO1frO8Lchitr0fNaN9d+Ap0BjKtqWqd/J17qeDnXh8CL2A==} + unpipe@1.0.0: resolution: {integrity: sha512-pjy2bYhSsufwWlKwPc+l3cN7+wuJlK6uz0YdJEOlQDbl6jo/YlPi4mb8agUkVC8BF7V8NuzeyPNqRksA3hztKQ==} engines: {node: '>= 0.8'} @@ -2524,8 +2608,8 @@ packages: yallist@3.1.1: resolution: {integrity: sha512-a4UGQaWPH59mOXUYnAG2ewncQS4i4F43Tv3JoAM+s2VDAmS9NsK8GpDMLrCHPksFT7h3K6TOoUNn2pb7RoXx4g==} - yaml@2.8.1: - resolution: {integrity: sha512-lcYcMxX2PO9XMGvAJkJ3OsNMw+/7FKes7/hgerGUYWIoWu5j/+YQqcZr5JnPZWzOsEBgMbSbiSTn/dv/69Mkpw==} + yaml@2.9.0: + resolution: {integrity: sha512-2AvhNX3mb8zd6Zy7INTtSpl1F15HW6Wnqj0srWlkKLcpYl/gMIMJiyuGq2KeI2YFxUPjdlB+3Lc10seMLtL4cA==} engines: {node: '>= 14.6'} hasBin: true @@ -2558,6 +2642,9 @@ packages: zod@3.25.76: resolution: {integrity: sha512-gzUt/qt81nXsFGKIFcC3YnfEAx5NkunCfnDlvuBSSFS02bcXu4Lmea0AFIUwbLWxWPx3d9p8S5QoaujKcNQxcQ==} + zod@4.4.3: + resolution: {integrity: sha512-ytENFjIJFl2UwYglde2jchW2Hwm4GJFLDiSXWdTrJQBIN9Fcyp7n4DhxJEiWNAJMV1/BqWfW/kkg71UDcHJyTQ==} + snapshots: '@babel/code-frame@7.27.1': @@ -3095,6 +3182,23 @@ snapshots: '@jridgewell/resolve-uri': 3.1.2 '@jridgewell/sourcemap-codec': 1.5.5 + '@modelcontextprotocol/conformance@https://codeload.github.com/modelcontextprotocol/conformance/tar.gz/49103de6ed70804e940637bf3e9e29e4a3f54e64': + dependencies: + '@modelcontextprotocol/sdk': 1.26.0(zod@4.4.3) + '@octokit/rest': 22.0.1 + ajv: 8.20.0 + ajv-formats: 3.0.1(ajv@8.20.0) + commander: 14.0.3 + eventsource-parser: 3.1.0 + express: 5.2.1 + jose: 6.1.3 + undici: 7.28.0 + yaml: 2.9.0 + zod: 4.4.3 + transitivePeerDependencies: + - '@cfworker/json-schema' + - supports-color + '@modelcontextprotocol/sdk@1.26.0(zod@3.25.76)': dependencies: '@hono/node-server': 1.19.13(hono@4.12.25) @@ -3117,6 +3221,28 @@ snapshots: transitivePeerDependencies: - supports-color + '@modelcontextprotocol/sdk@1.26.0(zod@4.4.3)': + dependencies: + '@hono/node-server': 1.19.13(hono@4.12.25) + ajv: 8.17.1 + ajv-formats: 3.0.1(ajv@8.17.1) + content-type: 1.0.5 + cors: 2.8.5 + cross-spawn: 7.0.6 + eventsource: 3.0.7 + eventsource-parser: 3.0.6 + express: 5.2.1 + express-rate-limit: 8.3.2(express@5.2.1) + hono: 4.12.25 + jose: 6.1.3 + json-schema-typed: 8.0.2 + pkce-challenge: 5.0.0 + raw-body: 3.0.1 + zod: 4.4.3 + zod-to-json-schema: 3.25.2(zod@4.4.3) + transitivePeerDependencies: + - supports-color + '@nodelib/fs.scandir@2.1.5': dependencies: '@nodelib/fs.stat': 2.0.5 @@ -3129,6 +3255,69 @@ snapshots: '@nodelib/fs.scandir': 2.1.5 fastq: 1.19.1 + '@octokit/auth-token@6.0.0': {} + + '@octokit/core@7.0.6': + dependencies: + '@octokit/auth-token': 6.0.0 + '@octokit/graphql': 9.0.3 + '@octokit/request': 10.0.11 + '@octokit/request-error': 7.1.0 + '@octokit/types': 16.0.0 + before-after-hook: 4.0.0 + universal-user-agent: 7.0.3 + + '@octokit/endpoint@11.0.3': + dependencies: + '@octokit/types': 16.0.0 + universal-user-agent: 7.0.3 + + '@octokit/graphql@9.0.3': + dependencies: + '@octokit/request': 10.0.11 + '@octokit/types': 16.0.0 + universal-user-agent: 7.0.3 + + '@octokit/openapi-types@27.0.0': {} + + '@octokit/plugin-paginate-rest@14.0.0(@octokit/core@7.0.6)': + dependencies: + '@octokit/core': 7.0.6 + '@octokit/types': 16.0.0 + + '@octokit/plugin-request-log@6.0.0(@octokit/core@7.0.6)': + dependencies: + '@octokit/core': 7.0.6 + + '@octokit/plugin-rest-endpoint-methods@17.0.0(@octokit/core@7.0.6)': + dependencies: + '@octokit/core': 7.0.6 + '@octokit/types': 16.0.0 + + '@octokit/request-error@7.1.0': + dependencies: + '@octokit/types': 16.0.0 + + '@octokit/request@10.0.11': + dependencies: + '@octokit/endpoint': 11.0.3 + '@octokit/request-error': 7.1.0 + '@octokit/types': 16.0.0 + content-type: 2.0.0 + json-with-bigint: 3.5.10 + universal-user-agent: 7.0.3 + + '@octokit/rest@22.0.1': + dependencies: + '@octokit/core': 7.0.6 + '@octokit/plugin-paginate-rest': 14.0.0(@octokit/core@7.0.6) + '@octokit/plugin-request-log': 6.0.0(@octokit/core@7.0.6) + '@octokit/plugin-rest-endpoint-methods': 17.0.0(@octokit/core@7.0.6) + + '@octokit/types@16.0.0': + dependencies: + '@octokit/openapi-types': 27.0.0 + '@pkgjs/parseargs@0.11.0': optional: true @@ -3393,6 +3582,10 @@ snapshots: optionalDependencies: ajv: 8.17.1 + ajv-formats@3.0.1(ajv@8.20.0): + optionalDependencies: + ajv: 8.20.0 + ajv@6.12.6: dependencies: fast-deep-equal: 3.1.3 @@ -3407,6 +3600,13 @@ snapshots: json-schema-traverse: 1.0.0 require-from-string: 2.0.2 + ajv@8.20.0: + dependencies: + fast-deep-equal: 3.1.3 + fast-uri: 3.1.1 + json-schema-traverse: 1.0.0 + require-from-string: 2.0.2 + ansi-escapes@4.3.2: dependencies: type-fest: 0.21.3 @@ -3499,6 +3699,8 @@ snapshots: baseline-browser-mapping@2.8.9: {} + before-after-hook@4.0.0: {} + body-parser@2.2.2: dependencies: bytes: 3.1.2 @@ -3602,6 +3804,8 @@ snapshots: color-name@1.1.4: {} + commander@14.0.3: {} + commander@4.1.1: {} concat-map@0.0.1: {} @@ -3616,6 +3820,8 @@ snapshots: content-type@1.0.5: {} + content-type@2.0.0: {} + convert-source-map@2.0.0: {} cookie-signature@1.2.2: {} @@ -3829,6 +4035,8 @@ snapshots: eventsource-parser@3.0.6: {} + eventsource-parser@3.1.0: {} + eventsource@3.0.7: dependencies: eventsource-parser: 3.0.6 @@ -3882,7 +4090,7 @@ snapshots: once: 1.4.0 parseurl: 1.3.3 proxy-addr: 2.0.7 - qs: 6.14.0 + qs: 6.15.1 range-parser: 1.2.1 router: 2.2.0 send: 1.2.0 @@ -4510,6 +4718,8 @@ snapshots: json-stable-stringify-without-jsonify@1.0.1: {} + json-with-bigint@3.5.10: {} + json5@2.2.3: {} keyv@4.5.4: @@ -4729,12 +4939,12 @@ snapshots: mlly: 1.8.0 pathe: 2.0.3 - postcss-load-config@6.0.1(postcss@8.5.6)(yaml@2.8.1): + postcss-load-config@6.0.1(postcss@8.5.6)(yaml@2.9.0): dependencies: lilconfig: 3.1.3 optionalDependencies: postcss: 8.5.6 - yaml: 2.8.1 + yaml: 2.9.0 postcss@8.5.6: dependencies: @@ -4769,10 +4979,6 @@ snapshots: pure-rand@6.1.0: {} - qs@6.14.0: - dependencies: - side-channel: 1.1.0 - qs@6.15.1: dependencies: side-channel: 1.1.0 @@ -5093,7 +5299,7 @@ snapshots: v8-compile-cache-lib: 3.0.1 yn: 3.1.1 - tsup@8.5.0(postcss@8.5.6)(typescript@5.9.2)(yaml@2.8.1): + tsup@8.5.0(postcss@8.5.6)(typescript@5.9.2)(yaml@2.9.0): dependencies: bundle-require: 5.1.0(esbuild@0.28.1) cac: 6.7.14 @@ -5104,7 +5310,7 @@ snapshots: fix-dts-default-cjs-exports: 1.0.1 joycon: 3.1.1 picocolors: 1.1.1 - postcss-load-config: 6.0.1(postcss@8.5.6)(yaml@2.8.1) + postcss-load-config: 6.0.1(postcss@8.5.6)(yaml@2.9.0) resolve-from: 5.0.0 rollup: 4.59.0 source-map: 0.8.0-beta.0 @@ -5157,6 +5363,10 @@ snapshots: undici-types@6.21.0: {} + undici@7.28.0: {} + + universal-user-agent@7.0.3: {} + unpipe@1.0.0: {} update-browserslist-db@1.1.3(browserslist@4.26.2): @@ -5222,8 +5432,7 @@ snapshots: yallist@3.1.1: {} - yaml@2.8.1: - optional: true + yaml@2.9.0: {} yargs-parser@21.1.1: {} @@ -5249,4 +5458,10 @@ snapshots: dependencies: zod: 3.25.76 + zod-to-json-schema@3.25.2(zod@4.4.3): + dependencies: + zod: 4.4.3 + zod@3.25.76: {} + + zod@4.4.3: {} diff --git a/pnpm-workspace.yaml b/pnpm-workspace.yaml index 7f5e2d915e..a7c72d1f20 100644 --- a/pnpm-workspace.yaml +++ b/pnpm-workspace.yaml @@ -14,4 +14,6 @@ strictDepBuilds: true trustPolicy: no-downgrade trustPolicyIgnoreAfter: 10080 trustPolicyExclude: [] -allowBuilds: {} +allowBuilds: + # Build only the official MCP conformance CLI pinned by the TypeScript SDK. + "@modelcontextprotocol/conformance": true diff --git a/scripts/mcp_conformance/README.md b/scripts/mcp_conformance/README.md new file mode 100644 index 0000000000..d75aa1f3b8 --- /dev/null +++ b/scripts/mcp_conformance/README.md @@ -0,0 +1,121 @@ +# MCP client conformance + +This directory tests the actual Codex executable against the official Model +Context Protocol client conformance suite. It exercises the shipping legacy, +intermediate `2025-11-25`, and modern `2026-07-28` protocols, localhost HTTP, +stdio, OAuth, and additional transport and security regression fixtures. + +The official upstream suite is pinned to +`modelcontextprotocol/conformance@49103de6ed70804e940637bf3e9e29e4a3f54e64`. +Use Node.js 22 and Python 3.10 or later. + +## Run the conformance gate + +First install the frozen workspace dependencies and build Codex: + +```bash +pnpm install --frozen-lockfile +cargo build --locked --manifest-path codex-rs/Cargo.toml -p codex-cli --bin codex +``` + +From a published Codex checkout, run: + +```bash +python3 scripts/mcp_conformance/run_codex_compliance.py \ + codex-rs/target/debug/codex \ + --conformance-cli node_modules/@modelcontextprotocol/conformance/dist/index.js \ + --baseline-report scripts/mcp_conformance/regression-baseline-v1.json \ + --report /tmp/codex-mcp-conformance.json +``` + +The positional executable can also point to an already built Codex binary. +`--conformance-cli` selects the exact, lockfile-installed upstream JavaScript +runner instead of downloading a moving version during a test. + +## What the baseline means + +`regression-baseline-v1.json` is a compact, reviewed snapshot of the upstream +revision, required protocol versions, HTTP and stdio transports, enabled modern +feature, OAuth coverage, and individual passing and failing check identities. + +The gate exits successfully only when: + +- The upstream suite and modern feature match the committed baseline. +- The shipping legacy, intermediate, and modern protocols are actually tested. +- The required HTTP, stdio, and authentication scenarios are actually run. +- Every previously passing check still passes. +- No additional check fails. + +Existing known failures remain visible in the complete JSON report. In +particular, `success` describes complete upstream conformance and +`regressionGate.success` describes the no-new-regressions merge gate; the gate +does not relabel an existing failure as a pass. + +Create a compact baseline from a reviewed complete report without contacting +the upstream suite again: + +```bash +python3 scripts/mcp_conformance/run_codex_compliance.py \ + /absolute/path/to/codex \ + --baseline-report /absolute/path/to/full-conformance-report.json \ + --extract-baseline /tmp/mcp-conformance-regression-baseline-v1.json +``` + +Alternatively, add `--write-baseline /tmp/mcp-conformance-regression-baseline-v1.json` +to a complete conformance run. Review every baseline change; do not regenerate +it to conceal a regression. + +## Run the production reviewer regression gate + +The separate reviewer gate tests the real Codex app-server across all three +shipping, legacy, and modern protocol modes. It covers stdio and localhost +HTTP, exact-integer tool and elicitation schemas, bounded multi-round requests, +malformed discovery response IDs, repeated pagination cursors, SSE framing and +keepalives, and catalog boundaries. In a published Codex checkout, run: + +```bash +python3 scripts/mcp_conformance/review_regressions.py \ + /absolute/path/to/codex \ + --mode all \ + --baseline-report scripts/mcp_conformance/review-regression-baseline-v1.json \ + --report /tmp/codex-mcp-review-regressions.json +``` + +A complete, main-derived baseline records all 186 real check identities and all +21 required cases. Existing failures remain explicitly visible in the complete +report; `regressionGate.success: true` means there are no newly failing or +missing checks. Improvements are recorded under `fixedChecks`. The gate never +classifies an existing failure as a passing check. + +Extract a compact deterministic reviewer baseline from a reviewed complete +production report without rerunning the client: + +```bash +python3 scripts/mcp_conformance/review_regressions.py \ + /absolute/path/to/codex \ + --baseline-report /absolute/path/to/full-review-regressions.json \ + --extract-baseline /tmp/review-regression-baseline-v1.json +``` + +Review every baseline update. Do not regenerate a baseline to hide a regression. + +## Run the fixture self-tests + +```bash +env PYTEST_DISABLE_PLUGIN_AUTOLOAD=1 \ + python3 -m pytest -q scripts/mcp_conformance +``` + +## Run the required SDK integration + +The existing required SDK workflow runs the complete Python fixture self-tests. +Its TypeScript job builds the actual Codex executable, sets `CODEX_EXEC_PATH`, +installs the pinned upstream conformance runner, and runs both the official +authenticated suite and the separate production reviewer regression matrix. +Neither gate can be skipped. To reproduce the focused integration locally: + +```bash +CODEX_EXEC_PATH=/absolute/path/to/codex \ + pnpm --filter @openai/codex-sdk test -- \ + --runInBand tests/mcpConformance.test.ts +``` diff --git a/scripts/mcp_conformance/codex_conformance_adapter.py b/scripts/mcp_conformance/codex_conformance_adapter.py new file mode 100644 index 0000000000..bdf2e1dfeb --- /dev/null +++ b/scripts/mcp_conformance/codex_conformance_adapter.py @@ -0,0 +1,976 @@ +#!/usr/bin/env python3 +"""Adapt one official MCP client scenario to Codex app-server requests.""" + +import ipaddress +import json +import os +import sys +import traceback +import urllib.error +import urllib.parse +from dataclasses import asdict, dataclass +from pathlib import Path +from typing import Mapping, Sequence +from urllib.request import HTTPRedirectHandler, ProxyHandler, Request, build_opener # noqa: TID251 + +_MODULE_DIR = Path(__file__).resolve().parent +if str(_MODULE_DIR) not in sys.path: + sys.path.insert(0, str(_MODULE_DIR)) + +from run_codex_compliance import ( # noqa: E402 - direct scripts must first add their sibling directory. + MODERN_VERSION, + TEST_SERVER_NAME, + AppServerClient, + AppServerError, + _call_tool, + _command_detail, + _isolated_environment, + _response_result, + _run_command, +) + + +@dataclass +class Step: + name: str + success: bool + detail: str + + +class AdapterFailure(RuntimeError): + pass + + +CIMD_CLIENT_METADATA_URL = "https://conformance-test.local/client-metadata.json" +PRE_REGISTERED_CLIENT_SECRET_ENV_VAR = "MCP_CONFORMANCE_CLIENT_SECRET" +AUTH_COMPLETION_METHOD = "mcpServer/oauthLogin/completed" +EXPECTED_AUTH_REJECTION_SCENARIOS = frozenset( + { + "auth/resource-mismatch", + "auth/iss-supported-missing", + "auth/iss-wrong-issuer", + "auth/iss-unexpected", + "auth/iss-normalized", + "auth/metadata-issuer-mismatch", + } +) + + +def _required_path(name: str) -> Path: + value = os.environ.get(name) + if not value: + raise AdapterFailure(f"{name} is required") + return Path(value).expanduser().resolve() + + +def _required_env(name: str) -> str: + value = os.environ.get(name) + if not value: + raise AdapterFailure(f"{name} is required") + return value + + +def _conformance_context() -> dict[str, object]: + raw = os.environ.get("MCP_CONFORMANCE_CONTEXT") + if not raw: + return {} + try: + decoded = json.loads(raw) + except json.JSONDecodeError as exc: + raise AdapterFailure(f"invalid MCP_CONFORMANCE_CONTEXT: {exc}") from exc + if not isinstance(decoded, dict): + raise AdapterFailure("MCP_CONFORMANCE_CONTEXT was not an object") + return decoded + + +def _result_or_raise( + response: Mapping[str, object], + operation: str, +) -> dict[str, object]: + result, detail = _response_result(response) + if result is None: + raise AdapterFailure(f"{operation}: {detail}") + return result + + +def _server_entry(inventory: Mapping[str, object]) -> dict[str, object]: + entries = inventory.get("data") + if not isinstance(entries, list): + raise AdapterFailure("mcpServerStatus/list returned no data array") + for entry in entries: + if isinstance(entry, dict) and entry.get("name") == TEST_SERVER_NAME: + return entry + raise AdapterFailure(f"{TEST_SERVER_NAME!r} was absent from MCP status") + + +def _thread_id(client: AppServerClient, workspace: Path) -> str: + result = _result_or_raise( + client.request( + "thread/start", + {"cwd": str(workspace), "ephemeral": True}, + ), + "thread/start", + ) + thread = result.get("thread") + if not isinstance(thread, dict) or not isinstance(thread.get("id"), str): + raise AdapterFailure("thread/start did not return a thread id") + return str(thread["id"]) + + +def _call( + client: AppServerClient, + *, + thread_id: str, + tool: str, + arguments: Mapping[str, object], +) -> None: + result, detail = _call_tool( + client, + thread_id=thread_id, + tool=tool, + arguments=arguments, + ) + if result is None: + raise AdapterFailure(f"tool {tool}: {detail}") + + +def _elicitation_content( + scenario: str, + _params: Mapping[str, object], +) -> Mapping[str, object]: + if scenario == "elicitation-sep1034-client-defaults": + # The official scenario deliberately supplies no values. Codex, as the + # MCP client under test, must materialize the JSON Schema defaults. + return {} + if scenario == "sep-2322-client-request-state": + return {"confirmed": True} + return {"confirmation": "confirmed"} + + +def _context_tool_calls() -> list[tuple[str, dict[str, object]]]: + context = _conformance_context() + if not context: + raise AdapterFailure( + "official scenario did not provide MCP_CONFORMANCE_CONTEXT" + ) + calls = context.get("toolCalls") + if not isinstance(calls, list): + raise AdapterFailure("official scenario context did not contain toolCalls") + result: list[tuple[str, dict[str, object]]] = [] + for call in calls: + if ( + not isinstance(call, dict) + or not isinstance(call.get("name"), str) + or not isinstance(call.get("arguments"), dict) + ): + raise AdapterFailure("official scenario contained an invalid tool call") + result.append((str(call["name"]), dict(call["arguments"]))) + return result + + +def _is_loopback_hostname(hostname: str | None) -> bool: + if hostname is None: + return False + if hostname.lower() == "localhost": + return True + try: + return ipaddress.ip_address(hostname).is_loopback + except ValueError: + return False + + +def _validated_callback_url(authorization_url: str, location: str) -> str: + authorization = urllib.parse.urlsplit(authorization_url) + query = urllib.parse.parse_qs(authorization.query, keep_blank_values=True) + redirect_values = query.get("redirect_uri") + state_values = query.get("state") + if not redirect_values or len(redirect_values) != 1: + raise AdapterFailure( + "authorization request did not contain exactly one redirect_uri" + ) + if not state_values or len(state_values) != 1: + raise AdapterFailure("authorization request did not contain exactly one state") + + redirect = urllib.parse.urlsplit(redirect_values[0]) + if ( + redirect.scheme != "http" + or not _is_loopback_hostname(redirect.hostname) + or redirect.username is not None + or redirect.password is not None + or redirect.port is None + or redirect.query + or redirect.fragment + ): + raise AdapterFailure("OAuth redirect_uri was not a safe loopback HTTP endpoint") + + callback_url = urllib.parse.urljoin(authorization_url, location) + callback = urllib.parse.urlsplit(callback_url) + expected_endpoint = ( + redirect.scheme, + redirect.hostname, + redirect.port, + redirect.path, + ) + actual_endpoint = ( + callback.scheme, + callback.hostname, + callback.port, + callback.path, + ) + if ( + actual_endpoint != expected_endpoint + or callback.username is not None + or callback.password is not None + or callback.fragment + ): + raise AdapterFailure( + "authorization server redirect did not target Codex's exact loopback callback" + ) + + callback_query = urllib.parse.parse_qs(callback.query, keep_blank_values=True) + callback_states = callback_query.get("state") + if callback_states != state_values: + raise AdapterFailure( + "authorization server redirect did not preserve OAuth state" + ) + for parameter in ("code", "error", "iss"): + values = callback_query.get(parameter) + if values is not None and (len(values) != 1 or not values[0]): + raise AdapterFailure( + f"authorization server redirect must contain exactly one nonempty {parameter}" + ) + if ("code" in callback_query) == ("error" in callback_query): + raise AdapterFailure( + "authorization server redirect must contain exactly one of code or error" + ) + return callback_url + + +class _NoRedirect(HTTPRedirectHandler): + def redirect_request( + self, + req: Request, + fp: object, + code: int, + msg: str, + headers: Mapping[str, str], + newurl: str, + ) -> Request | None: + del req, fp, code, msg, headers, newurl + return None + + +def _open_without_redirects(url: str, timeout_seconds: float) -> tuple[int, str | None]: + opener = build_opener( + ProxyHandler({}), + _NoRedirect(), + ) + request = Request( + url, + method="GET", + headers={"Accept": "text/html,application/xhtml+xml"}, + ) + try: + with opener.open(request, timeout=timeout_seconds) as response: + return response.status, response.headers.get("Location") + except urllib.error.HTTPError as exc: + # With redirects disabled urllib represents 3xx as HTTPError. Closing + # the body promptly prevents an authorization page from being retained. + try: + return exc.code, exc.headers.get("Location") + finally: + exc.close() + + +def _drive_headless_authorization( + authorization_url: str, + *, + timeout_seconds: float, +) -> None: + status, location = _open_without_redirects(authorization_url, timeout_seconds) + if status not in {301, 302, 303, 307, 308} or not location: + raise AdapterFailure( + "authorization endpoint did not issue the expected callback redirect" + ) + callback_url = _validated_callback_url(authorization_url, location) + callback_status, _ = _open_without_redirects(callback_url, timeout_seconds) + # Error callbacks may intentionally return a 4xx after notifying the OAuth + # waiter. The completion notification is the authoritative outcome. + if not 200 <= callback_status < 500: + raise AdapterFailure("Codex OAuth callback endpoint returned an invalid status") + + +def _oauth_client_id( + scenario: str, + context: Mapping[str, object], + *, + require_production_client_identity: bool = False, +) -> str | None: + if scenario == "auth/basic-cimd": + return None if require_production_client_identity else CIMD_CLIENT_METADATA_URL + if scenario == "auth/pre-registration": + client_id = context.get("client_id") + if not isinstance(client_id, str) or not client_id: + raise AdapterFailure("pre-registration context did not contain client_id") + return client_id + return None + + +def _write_auth_registration( + config_path: Path, + *, + server_url: str, + oauth_client_id: str | None, + oauth_client_secret_env_var: str | None = None, +) -> None: + existing = config_path.read_text(encoding="utf-8") if config_path.exists() else "" + block = f"\n[mcp_servers.{TEST_SERVER_NAME}]\nurl = {json.dumps(server_url, ensure_ascii=False)}\n" + if oauth_client_id is not None: + block += ( + f"\n[mcp_servers.{TEST_SERVER_NAME}.oauth]\n" + f"client_id = {json.dumps(oauth_client_id, ensure_ascii=False)}\n" + ) + if oauth_client_secret_env_var is not None: + block += ( + "client_secret_env_var = " + f"{json.dumps(oauth_client_secret_env_var, ensure_ascii=False)}\n" + ) + config_path.write_text(existing.rstrip() + "\n" + block, encoding="utf-8") + + +def _validate_oauth_secret_not_persisted(codex_home: Path, client_secret: str) -> None: + if not client_secret: + return + + candidates = [codex_home / "config.toml", *codex_home.rglob(".credentials.json")] + secret_bytes = client_secret.encode("utf-8") + for candidate in dict.fromkeys(candidates): + if not candidate.is_file(): + continue + if secret_bytes in candidate.read_bytes(): + raise AdapterFailure( + "environment-provided OAuth client secret was persisted in " + f"{candidate.relative_to(codex_home)}" + ) + + +def _oauth_login( + client: AppServerClient, + *, + scopes: Sequence[str] | None, + timeout_seconds: float, +) -> tuple[bool, str | None]: + event_index = len(client.events) + params: dict[str, object] = { + "name": TEST_SERVER_NAME, + "timeoutSecs": max(1, round(timeout_seconds)), + } + if scopes is not None: + params["scopes"] = list(scopes) + response = client.request("mcpServer/oauth/login", params) + result, detail = _response_result(response) + if result is None: + return False, detail + authorization_url = result.get("authorizationUrl") + if not isinstance(authorization_url, str): + raise AdapterFailure("mcpServer/oauth/login returned no authorization URL") + + _drive_headless_authorization( + authorization_url, + timeout_seconds=timeout_seconds, + ) + event = client.wait_for_notification( + AUTH_COMPLETION_METHOD, + predicate=lambda params: params.get("name") == TEST_SERVER_NAME, + after_event_index=event_index, + ) + params_value = event.get("params") + if not isinstance(params_value, dict): + raise AdapterFailure("OAuth completion notification did not contain params") + success = params_value.get("success") is True + error = params_value.get("error") + return success, str(error) if error is not None else None + + +def _reload_mcp(client: AppServerClient) -> None: + _result_or_raise( + client.request("config/mcpServer/reload", None), + "config/mcpServer/reload", + ) + + +def _auth_inventory(client: AppServerClient) -> dict[str, object]: + inventory = _result_or_raise( + client.request("mcpServerStatus/list", {"detail": "full"}), + "mcpServerStatus/list", + ) + _server_entry(inventory) + return inventory + + +def _auth_tool_call(client: AppServerClient, workspace: Path) -> None: + thread_id = _thread_id(client, workspace) + _call( + client, + thread_id=thread_id, + tool="test-tool", + arguments={}, + ) + + +def _login_reload_and_call( + client: AppServerClient, + *, + workspace: Path, + timeout_seconds: float, + scopes: Sequence[str] | None = None, +) -> None: + success, error = _oauth_login( + client, + scopes=scopes, + timeout_seconds=timeout_seconds, + ) + if not success: + raise AdapterFailure(f"OAuth login failed: {error or 'unknown error'}") + _reload_mcp(client) + _auth_inventory(client) + _auth_tool_call(client, workspace) + + +def _exercise_auth_scenario( + client: AppServerClient, + *, + scenario: str, + workspace: Path, + timeout_seconds: float, + require_automatic_auth: bool = False, +) -> str: + if scenario in EXPECTED_AUTH_REJECTION_SCENARIOS: + success, error = _oauth_login( + client, + scopes=None, + timeout_seconds=timeout_seconds, + ) + if success: + raise AdapterFailure( + "OAuth flow unexpectedly accepted authorization metadata that must be rejected" + ) + return f"rejected invalid authorization flow: {error or 'request rejected'}" + + if scenario == "auth/scope-step-up": + success, error = _oauth_login( + client, + scopes=None, + timeout_seconds=timeout_seconds, + ) + if not success: + raise AdapterFailure( + f"initial OAuth login failed: {error or 'unknown error'}" + ) + _reload_mcp(client) + _auth_inventory(client) + if require_automatic_auth: + _auth_tool_call(client, workspace) + return "Codex automatically recovered from the challenged OAuth scope" + try: + _auth_tool_call(client, workspace) + except AdapterFailure: + pass + else: + raise AdapterFailure( + "scope-step-up tool call did not request additional scope" + ) + + # The resource server challenges with only the missing scope. The Rust + # client must union it with the previously granted scope. + _login_reload_and_call( + client, + workspace=workspace, + timeout_seconds=timeout_seconds, + scopes=("mcp:write",), + ) + return "completed initial and scope-upgrade authorization flows" + + if scenario == "auth/scope-retry-limit": + if require_automatic_auth: + success, error = _oauth_login( + client, + scopes=None, + timeout_seconds=timeout_seconds, + ) + if not success: + raise AdapterFailure( + f"initial OAuth login failed: {error or 'unknown error'}" + ) + _reload_mcp(client) + _auth_inventory(client) + try: + _auth_tool_call(client, workspace) + except AdapterFailure: + return "observed Codex's production OAuth retry-limit behavior" + raise AdapterFailure( + "retry-limit scenario unexpectedly completed the tool call" + ) + for attempt in range(3): + success, error = _oauth_login( + client, + scopes=None if attempt == 0 else ("mcp:write",), + timeout_seconds=timeout_seconds, + ) + if not success: + raise AdapterFailure(f"OAuth retry failed: {error or 'unknown error'}") + _reload_mcp(client) + _auth_inventory(client) + try: + _auth_tool_call(client, workspace) + except AdapterFailure: + continue + raise AdapterFailure( + "retry-limit scenario unexpectedly completed the tool call" + ) + return "stopped after three unsuccessful authorization attempts" + + if scenario == "auth/authorization-server-migration": + success, error = _oauth_login( + client, + scopes=None, + timeout_seconds=timeout_seconds, + ) + if not success: + raise AdapterFailure( + f"initial OAuth login failed: {error or 'unknown error'}" + ) + _reload_mcp(client) + _auth_inventory(client) + if require_automatic_auth: + _auth_tool_call(client, workspace) + return ( + "Codex automatically registered with the migrated authorization server" + ) + try: + _auth_tool_call(client, workspace) + except AdapterFailure: + pass + else: + raise AdapterFailure("migration scenario did not require re-authorization") + _login_reload_and_call( + client, + workspace=workspace, + timeout_seconds=timeout_seconds, + ) + return ( + "re-authorized after the protected resource changed authorization servers" + ) + + _login_reload_and_call( + client, + workspace=workspace, + timeout_seconds=timeout_seconds, + ) + return "completed OAuth login, authenticated discovery, and tool call" + + +def _exercise_scenario( + client: AppServerClient, + *, + scenario: str, + workspace: Path, + inventory: Mapping[str, object], +) -> str: + thread_id = _thread_id(client, workspace) + + if scenario == "tools_call": + _call( + client, + thread_id=thread_id, + tool="add_numbers", + arguments={"a": 2, "b": 3}, + ) + return "called add_numbers" + + if scenario == "elicitation-sep1034-client-defaults": + _call( + client, + thread_id=thread_id, + tool="test_client_elicitation_defaults", + arguments={}, + ) + return "completed legacy elicitation with omitted optional fields" + + if scenario == "sse-retry": + _call( + client, + thread_id=thread_id, + tool="test_reconnection", + arguments={}, + ) + return "completed the SSE reconnection tool call" + + if scenario == "sep-2322-client-request-state": + for tool in ( + "test_mrtr_unrelated", + "test_mrtr_no_result_type", + "test_mrtr_echo_state", + "test_mrtr_no_state", + ): + _call(client, thread_id=thread_id, tool=tool, arguments={}) + return "completed all four MRTR flows" + + if scenario == "http-standard-headers": + _call( + client, + thread_id=thread_id, + tool="test_headers", + arguments={}, + ) + entry = _server_entry(inventory) + resources = entry.get("resources") + if not isinstance(resources, list) or not resources: + raise AdapterFailure("standard-header scenario exposed no resources") + first = resources[0] + if not isinstance(first, dict) or not isinstance(first.get("uri"), str): + raise AdapterFailure("standard-header scenario resource had no URI") + _result_or_raise( + client.request( + "mcpServer/resource/read", + { + "threadId": thread_id, + "server": TEST_SERVER_NAME, + "uri": first["uri"], + }, + ), + "mcpServer/resource/read", + ) + return "called a tool and read a resource" + + if scenario == "http-custom-headers": + for tool, arguments in _context_tool_calls(): + _call( + client, + thread_id=thread_id, + tool=tool, + arguments=arguments, + ) + return "called both custom-header tools with official values" + + if scenario == "http-invalid-tool-headers": + _call( + client, + thread_id=thread_id, + tool="valid_tool", + arguments={"region": "us-west1"}, + ) + return "called the valid tool after filtering malformed definitions" + + if scenario in { + "initialize", + "request-metadata", + "json-schema-ref-no-deref", + }: + # Discovery performed by mcpServerStatus/list is the behavior these + # scenarios observe. Starting a thread also exercises the initialized + # server through the same public app-server interface as the other + # scenarios. + return "completed MCP discovery" + + raise AdapterFailure(f"unsupported official scenario: {scenario}") + + +def run_adapter(server_url: str) -> dict[str, object]: + codex_binary = _required_path("CODEX_CONFORMANCE_BINARY") + codex_home = _required_path("CODEX_CONFORMANCE_HOME") + scenario = _required_env("MCP_CONFORMANCE_SCENARIO") + protocol_version = _required_env("MCP_CONFORMANCE_PROTOCOL_VERSION") + timeout_seconds = float(os.environ.get("CODEX_CONFORMANCE_TIMEOUT", "30")) + enable_modern_feature = ( + os.environ.get("CODEX_CONFORMANCE_ENABLE_MODERN_FEATURE", "1") != "0" + ) + require_automatic_auth = ( + os.environ.get("CODEX_CONFORMANCE_REQUIRE_AUTOMATIC_AUTH", "0") == "1" + ) + context = _conformance_context() + + codex_home.mkdir(parents=True, exist_ok=True) + workspace = codex_home / "workspace" + workspace.mkdir() + env = _isolated_environment(codex_home) + # Scenario context can contain ephemeral OAuth client secrets. The adapter + # consumes it directly and does not expose the full blob to Codex. + env.pop("MCP_CONFORMANCE_CONTEXT", None) + steps: list[Step] = [] + registered = False + error: str | None = None + + try: + if scenario.startswith("auth/"): + config_path = codex_home / "config.toml" + config_path.write_text( + 'mcp_oauth_credentials_store = "file"\n', + encoding="utf-8", + ) + steps.append( + Step( + "oauth_store_configuration", + True, + "configured the isolated file OAuth credential store", + ) + ) + + if protocol_version == MODERN_VERSION and enable_modern_feature: + feature = _run_command( + [ + str(codex_binary), + "features", + "enable", + "mcp_2026_07_28", + ], + env=env, + cwd=workspace, + timeout_seconds=timeout_seconds, + ) + feature_configured = feature.returncode == 0 + steps.append( + Step( + "modern_feature_configuration", + feature_configured, + "configured mcp_2026_07_28 before MCP startup" + if feature_configured + else _command_detail(feature), + ) + ) + if not feature_configured: + raise AdapterFailure("could not configure the modern MCP feature") + + oauth_client_id = _oauth_client_id( + scenario, + context, + require_production_client_identity=require_automatic_auth, + ) + oauth_client_secret = context.get("client_secret") + oauth_client_secret_env_var = None + if ( + scenario == "auth/pre-registration" + and isinstance(oauth_client_secret, str) + and oauth_client_secret + ): + env[PRE_REGISTERED_CLIENT_SECRET_ENV_VAR] = oauth_client_secret + oauth_client_secret_env_var = PRE_REGISTERED_CLIENT_SECRET_ENV_VAR + if scenario.startswith("auth/"): + _write_auth_registration( + codex_home / "config.toml", + server_url=server_url, + oauth_client_id=oauth_client_id, + oauth_client_secret_env_var=oauth_client_secret_env_var, + ) + registered = True + steps.append( + Step( + "mcp_registration", + True, + "wrote isolated registration without triggering CLI auto-login", + ) + ) + else: + add = _run_command( + [ + str(codex_binary), + "mcp", + "add", + TEST_SERVER_NAME, + "--url", + server_url, + ], + env=env, + cwd=workspace, + timeout_seconds=timeout_seconds, + ) + registered = add.returncode == 0 + steps.append(Step("mcp_add", registered, _command_detail(add))) + if not registered: + raise AdapterFailure("codex mcp add failed") + + get = _run_command( + [ + str(codex_binary), + "mcp", + "get", + TEST_SERVER_NAME, + "--json", + ], + env=env, + cwd=workspace, + timeout_seconds=timeout_seconds, + ) + registration_ok = False + if get.returncode == 0: + try: + decoded = json.loads(get.stdout) + transport = ( + decoded.get("transport") if isinstance(decoded, dict) else None + ) + registration_ok = ( + isinstance(transport, dict) + and transport.get("type") == "streamable_http" + and transport.get("url") == server_url + ) + except json.JSONDecodeError: + pass + steps.append( + Step( + "mcp_get", + registration_ok, + "registered official scenario URL" + if registration_ok + else _command_detail(get), + ) + ) + if not registration_ok: + raise AdapterFailure("Codex registration did not preserve the scenario URL") + + with AppServerClient( + codex_binary, + env=env, + cwd=workspace, + timeout_seconds=timeout_seconds, + elicitation_content=lambda params: _elicitation_content(scenario, params), + ) as client: + initialize = _result_or_raise( + client.request( + "initialize", + { + "clientInfo": { + "name": "official-mcp-conformance-adapter", + "title": "Official MCP conformance adapter", + "version": "1.0.0", + }, + "capabilities": { + "experimentalApi": True, + "requestAttestation": False, + "mcpServerOpenaiFormElicitation": True, + }, + }, + ), + "app-server initialize", + ) + steps.append(Step("app_server_initialize", bool(initialize), "initialized")) + client.notify("initialized") + + if protocol_version == MODERN_VERSION and enable_modern_feature: + feature = _result_or_raise( + client.request( + "experimentalFeature/enablement/set", + {"enablement": {"mcp_2026_07_28": True}}, + ), + "experimentalFeature/enablement/set", + ) + enabled = ( + isinstance(feature.get("enablement"), dict) + and feature["enablement"].get("mcp_2026_07_28") is True + ) + steps.append( + Step( + "modern_feature_enablement", + enabled, + "enabled mcp_2026_07_28" + if enabled + else f"unexpected response: {feature!r}", + ) + ) + if not enabled: + raise AdapterFailure("could not enable the modern MCP feature") + + if scenario.startswith("auth/"): + detail = _exercise_auth_scenario( + client, + scenario=scenario, + workspace=workspace, + timeout_seconds=timeout_seconds, + require_automatic_auth=require_automatic_auth, + ) + if scenario == "auth/pre-registration" and isinstance( + oauth_client_secret, + str, + ): + _validate_oauth_secret_not_persisted( + codex_home, oauth_client_secret + ) + steps.append( + Step( + "oauth_client_secret_not_persisted", + True, + "environment-provided confidential-client secret was not " + "written to configuration or file-backed credentials", + ) + ) + steps.append(Step("authentication", True, detail)) + else: + inventory = _result_or_raise( + client.request("mcpServerStatus/list", {"detail": "full"}), + "mcpServerStatus/list", + ) + _server_entry(inventory) + steps.append(Step("inventory", True, "official server discovered")) + detail = _exercise_scenario( + client, + scenario=scenario, + workspace=workspace, + inventory=inventory, + ) + steps.append(Step("scenario", True, detail)) + except (AdapterFailure, AppServerError, OSError, ValueError) as exc: + error = str(exc) + if not steps or steps[-1].success: + steps.append(Step("scenario", False, error)) + finally: + if registered: + remove = _run_command( + [str(codex_binary), "mcp", "remove", TEST_SERVER_NAME], + env=env, + cwd=workspace, + timeout_seconds=timeout_seconds, + ) + steps.append( + Step("mcp_remove", remove.returncode == 0, _command_detail(remove)) + ) + + success = bool(steps) and all(step.success for step in steps) + return { + "success": success, + "scenario": scenario, + "protocolVersion": protocol_version, + "automaticAuthRequired": require_automatic_auth, + "serverUrl": server_url, + "steps": [asdict(step) for step in steps], + "error": error, + } + + +def main(argv: Sequence[str] | None = None) -> int: + values = list(sys.argv[1:] if argv is None else argv) + report_path_value = os.environ.get("CODEX_CONFORMANCE_ADAPTER_REPORT") + report: dict[str, object] + try: + if len(values) != 1: + raise AdapterFailure("expected exactly one official scenario server URL") + report = run_adapter(values[0]) + except Exception as exc: # Preserve a diagnostic artifact for the parent. + report = { + "success": False, + "error": str(exc), + "traceback": traceback.format_exc(limit=20), + "steps": [], + } + + if report_path_value: + report_path = Path(report_path_value) + report_path.parent.mkdir(parents=True, exist_ok=True) + report_path.write_text( + json.dumps(report, ensure_ascii=False, indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + print(json.dumps(report, ensure_ascii=False, sort_keys=True)) + return 0 if report.get("success") is True else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/mcp_conformance/official_conformance.py b/scripts/mcp_conformance/official_conformance.py new file mode 100644 index 0000000000..a6cee00a17 --- /dev/null +++ b/scripts/mcp_conformance/official_conformance.py @@ -0,0 +1,518 @@ +"""Driver for the upstream MCP client conformance suite. + +The upstream suite owns the localhost HTTP server and the wire-level checks. +This module deliberately invokes one scenario at a time because the upstream +parallel suite path does not currently propagate the client adapter result. +""" + +import json +import os +import re +import shlex +import signal +import subprocess +import sys +import tempfile +import time +from dataclasses import dataclass, field +from pathlib import Path +from typing import Mapping, Sequence + +SHIPPING_LEGACY_VERSION = "2025-06-18" +LEGACY_VERSION = "2025-11-25" +MODERN_VERSION = "2026-07-28" + +# Reviewed on 2026-07-29. Pinning a commit instead of an npm prerelease keeps +# the scenario definitions and result semantics reproducible. +OFFICIAL_CONFORMANCE_GIT_REF = "49103de6ed70804e940637bf3e9e29e4a3f54e64" +OFFICIAL_CONFORMANCE_REPOSITORY = "modelcontextprotocol/conformance" + +OFFICIAL_NON_AUTH_SCENARIOS: Mapping[str, tuple[str, ...]] = { + SHIPPING_LEGACY_VERSION: ( + "initialize", + "tools_call", + ), + LEGACY_VERSION: ( + "initialize", + "tools_call", + "elicitation-sep1034-client-defaults", + "sse-retry", + ), + MODERN_VERSION: ( + "tools_call", + "request-metadata", + "sep-2322-client-request-state", + "http-standard-headers", + "http-custom-headers", + "http-invalid-tool-headers", + "json-schema-ref-no-deref", + ), +} + +OFFICIAL_AUTH_SCENARIOS: Mapping[str, tuple[str, ...]] = { + SHIPPING_LEGACY_VERSION: ( + "auth/token-endpoint-auth-basic", + "auth/token-endpoint-auth-post", + "auth/token-endpoint-auth-none", + ), + LEGACY_VERSION: ( + "auth/metadata-default", + "auth/metadata-var1", + "auth/metadata-var2", + "auth/metadata-var3", + "auth/basic-cimd", + "auth/scope-from-www-authenticate", + "auth/scope-from-scopes-supported", + "auth/scope-omitted-when-undefined", + "auth/scope-step-up", + "auth/scope-retry-limit", + "auth/token-endpoint-auth-basic", + "auth/token-endpoint-auth-post", + "auth/token-endpoint-auth-none", + "auth/pre-registration", + ), + MODERN_VERSION: ( + "auth/metadata-default", + "auth/metadata-var1", + "auth/metadata-var2", + "auth/metadata-var3", + "auth/basic-cimd", + "auth/scope-from-www-authenticate", + "auth/scope-from-scopes-supported", + "auth/scope-omitted-when-undefined", + "auth/scope-step-up", + "auth/scope-retry-limit", + "auth/token-endpoint-auth-basic", + "auth/token-endpoint-auth-post", + "auth/token-endpoint-auth-none", + "auth/pre-registration", + "auth/resource-mismatch", + "auth/offline-access-scope", + "auth/offline-access-not-supported", + "auth/authorization-server-migration", + "auth/iss-supported", + "auth/iss-not-advertised", + "auth/iss-supported-missing", + "auth/iss-wrong-issuer", + "auth/iss-unexpected", + "auth/iss-normalized", + "auth/metadata-issuer-mismatch", + ), +} + +# These scenarios cover separately negotiated protocol extensions rather than +# either dated release. Keep them visible for future adapters, but do not mix +# them into versioned release conformance percentages. +OFFICIAL_AUTH_EXTENSION_SCENARIOS: tuple[str, ...] = ( + "auth/client-credentials-jwt", + "auth/client-credentials-basic", + "auth/enterprise-managed-authorization", + "auth/dpop", + "auth/dpop-nonce", + "auth/wif-jwt-bearer", +) + +_SENSITIVE_JSON_FIELD = re.compile( + r'("(?:client_secret|private_key_pem|valid_jwt|wrong_audience_jwt|' + r'expired_jwt|idp_id_token)"\s*:\s*)"(?:\\.|[^"\\])*"' +) +_SENSITIVE_ESCAPED_JSON_FIELD = re.compile( + r'(\\"(?:client_secret|private_key_pem|valid_jwt|wrong_audience_jwt|' + r'expired_jwt|idp_id_token)\\"\s*:\s*)\\"(?:\\\\.|[^"\\])*\\"' +) +_SENSITIVE_URL_PARAMETER = re.compile( + r"([?&](?:code|state|code_challenge|code_verifier|access_token|" + r"""refresh_token|client_secret)=)[^&\s"'\\<>]+""", + re.IGNORECASE, +) +_BEARER_TOKEN = re.compile(r"(\bBearer\s+)[A-Za-z0-9._~+/=-]+", re.IGNORECASE) + + +def redact_sensitive_text(value: str) -> str: + value = _SENSITIVE_JSON_FIELD.sub(r'\1"[REDACTED]"', value) + value = _SENSITIVE_ESCAPED_JSON_FIELD.sub(r'\1\\"[REDACTED]\\"', value) + value = _SENSITIVE_URL_PARAMETER.sub(r"\1[REDACTED]", value) + return _BEARER_TOKEN.sub(r"\1[REDACTED]", value) + + +def _scrub_retained_artifacts(scenario_dir: Path) -> None: + for name in ("stdout.txt", "stderr.txt"): + for path in scenario_dir.rglob(name): + try: + original = path.read_text(encoding="utf-8") + redacted = redact_sensitive_text(original) + if redacted != original: + path.write_text(redacted, encoding="utf-8") + except OSError: + continue + for path in scenario_dir.rglob(".credentials.json"): + try: + path.unlink(missing_ok=True) + except OSError: + continue + + +@dataclass(frozen=True) +class OfficialCheck: + scenario: str + check_id: str + name: str + status: str + description: str + error_message: str | None = None + + +@dataclass +class OfficialScenarioResult: + scenario: str + success: bool + checks: list[OfficialCheck] = field(default_factory=list) + adapter_success: bool = False + adapter_detail: str = "" + runner_detail: str = "" + + +def default_conformance_command() -> list[str]: + return [ + "npx", + "--yes", + (f"github:{OFFICIAL_CONFORMANCE_REPOSITORY}#{OFFICIAL_CONFORMANCE_GIT_REF}"), + ] + + +def scenarios_for_mode( + mode: str, + requested: Sequence[str] | None = None, + *, + include_auth: bool = True, +) -> tuple[str, ...]: + non_auth = OFFICIAL_NON_AUTH_SCENARIOS.get(mode) + auth = OFFICIAL_AUTH_SCENARIOS.get(mode) + if non_auth is None or auth is None: + raise ValueError(f"unsupported MCP protocol version: {mode}") + available = non_auth + (auth if include_auth else ()) + if not requested: + return available + all_scenarios = { + scenario + for scenario_map in ( + OFFICIAL_NON_AUTH_SCENARIOS, + OFFICIAL_AUTH_SCENARIOS, + ) + for mode_scenarios in scenario_map.values() + for scenario in mode_scenarios + } + unknown = sorted(set(requested) - all_scenarios) + if unknown: + raise ValueError( + "scenarios are not part of the pinned versioned client suite: " + + ", ".join(unknown) + ) + disabled = sorted( + set(requested) + & { + scenario + for mode_scenarios in OFFICIAL_AUTH_SCENARIOS.values() + for scenario in mode_scenarios + } + if not include_auth + else () + ) + if disabled: + raise ValueError( + "authentication scenarios require auth coverage to be enabled: " + + ", ".join(disabled) + ) + requested_set = set(requested) + selected = tuple(scenario for scenario in available if scenario in requested_set) + if not selected: + raise ValueError( + f"requested scenarios are unavailable for MCP protocol version {mode}: " + + ", ".join(sorted(requested_set)) + ) + return selected + + +def _safe_scenario_name(scenario: str) -> str: + return re.sub(r"[^A-Za-z0-9_.-]+", "-", scenario) + + +def _make_adapter_launcher( + adapter_script: Path, + *, + require_automatic_auth: bool = False, +) -> Path: + # The official runner currently splits --command on spaces before spawning + # it. Put a one-token launcher in the system temp directory so arbitrary + # checkout paths (including paths with spaces) remain supported. + launcher_dir = Path(tempfile.mkdtemp(prefix="mcp-conformance-adapter-")) + automatic_auth = "1" if require_automatic_auth else "0" + if sys.platform == "win32": + launcher = launcher_dir / "client.cmd" + adapter_command = subprocess.list2cmdline([sys.executable, str(adapter_script)]) + launcher.write_text( + "@echo off\n" + f'set "CODEX_CONFORMANCE_REQUIRE_AUTOMATIC_AUTH={automatic_auth}"\n' + f"{adapter_command} %*\n", + encoding="utf-8", + ) + else: + launcher = launcher_dir / "client" + launcher.write_text( + "#!/bin/sh\n" + f"export CODEX_CONFORMANCE_REQUIRE_AUTOMATIC_AUTH={automatic_auth}\n" + f'exec {shlex.quote(sys.executable)} {shlex.quote(str(adapter_script))} "$@"\n', + encoding="utf-8", + ) + launcher.chmod(0o755) + return launcher + + +def _terminate_process_group( + process: subprocess.Popen[str], + *, + force: bool = False, +) -> None: + if sys.platform == "win32": + command = ["taskkill", "/T", "/PID", str(process.pid)] + if force: + command.insert(1, "/F") + try: + result = subprocess.run( + command, + check=False, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + ) + if result.returncode == 0: + return + except OSError: + pass + try: + if force: + process.kill() + else: + process.terminate() + except (OSError, ProcessLookupError): + pass + return + + try: + os.killpg(process.pid, signal.SIGKILL if force else signal.SIGTERM) + except (OSError, ProcessLookupError): + pass + + +def _load_checks(result_dir: Path, scenario: str) -> list[OfficialCheck]: + check_files = sorted(result_dir.rglob("checks.json")) + if len(check_files) != 1: + return [] + try: + decoded = json.loads(check_files[0].read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return [] + if not isinstance(decoded, list): + return [] + + checks: list[OfficialCheck] = [] + for index, raw in enumerate(decoded, start=1): + if not isinstance(raw, dict): + continue + check_id = str(raw.get("id") or f"unnamed-{index}") + name = str(raw.get("name") or check_id) + status = str(raw.get("status") or "FAILURE").upper() + # INFO entries are the suite's HTTP trace, not conformance assertions. + # They remain available in the retained upstream checks.json artifact. + if status == "INFO": + continue + checks.append( + OfficialCheck( + scenario=scenario, + check_id=check_id, + name=name, + status=status, + description=str(raw.get("description") or name), + error_message=( + redact_sensitive_text(str(raw["errorMessage"])) + if raw.get("errorMessage") is not None + else None + ), + ) + ) + return checks + + +def _load_adapter_report(report_path: Path) -> tuple[bool, str]: + try: + decoded = json.loads(report_path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as exc: + return False, f"Codex adapter did not write a valid report: {exc}" + if not isinstance(decoded, dict): + return False, "Codex adapter report was not an object" + success = decoded.get("success") is True + steps = decoded.get("steps") + failed_steps: list[str] = [] + if isinstance(steps, list): + for step in steps: + if isinstance(step, dict) and step.get("success") is not True: + failed_steps.append( + f"{step.get('name')}: {step.get('detail', 'failed')}" + ) + if success: + return True, "Codex adapter completed the scenario" + if failed_steps: + return False, redact_sensitive_text("; ".join(failed_steps))[-4_000:] + return False, redact_sensitive_text( + str(decoded.get("error") or "Codex adapter failed") + )[-4_000:] + + +def _run_scenario( + *, + conformance_command: Sequence[str], + adapter_launcher: Path, + codex_binary: Path, + mode: str, + scenario: str, + output_dir: Path, + timeout_seconds: float, + base_env: Mapping[str, str], + process_grace_seconds: float = 15, +) -> OfficialScenarioResult: + scenario_dir = output_dir / _safe_scenario_name(scenario) + scenario_dir.mkdir(parents=True, exist_ok=True) + adapter_report = scenario_dir / "codex-adapter.json" + adapter_home = scenario_dir / "codex-home" + adapter_home.mkdir() + + env = dict(base_env) + env.update( + { + "CODEX_CONFORMANCE_BINARY": str(codex_binary), + "CODEX_CONFORMANCE_HOME": str(adapter_home), + "CODEX_CONFORMANCE_ADAPTER_REPORT": str(adapter_report), + } + ) + command = [ + *conformance_command, + "client", + "--command", + str(adapter_launcher), + "--scenario", + scenario, + "--spec-version", + mode, + "--timeout", + str(round(timeout_seconds * 1_000)), + "--output-dir", + str(scenario_dir), + ] + process: subprocess.Popen[str] | None = None + try: + process = subprocess.Popen( + command, + env=env, + text=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + start_new_session=sys.platform != "win32", + creationflags=( + getattr(subprocess, "CREATE_NEW_PROCESS_GROUP", 0) + if sys.platform == "win32" + else 0 + ), + ) + try: + process_timeout = timeout_seconds + process_grace_seconds + stdout, stderr = process.communicate(timeout=process_timeout) + runner_detail = redact_sensitive_text((stderr or stdout).strip())[-8_000:] + returncode = process.returncode + except subprocess.TimeoutExpired: + _terminate_process_group(process) + try: + stdout, stderr = process.communicate(timeout=3) + except subprocess.TimeoutExpired: + _terminate_process_group(process, force=True) + stdout, stderr = process.communicate() + runner_detail = redact_sensitive_text( + (stderr or stdout) + + f"\nofficial scenario timed out after {process_timeout:g}s" + ).strip()[-8_000:] + returncode = 124 + except OSError as exc: + runner_detail = redact_sensitive_text(str(exc)) + returncode = 127 + + # A crashing upstream runner can exit before its still-running adapter has + # atomically written the diagnostic report. Give that report a short grace + # period so an upstream fixture crash is not mislabeled as an adapter + # failure, then terminate any orphaned descendants in the runner's process + # group. + if returncode != 0 and not adapter_report.is_file(): + report_deadline = time.monotonic() + min(process_grace_seconds, 10) + while time.monotonic() < report_deadline and not adapter_report.is_file(): + time.sleep(0.05) + if returncode != 0 and process is not None: + _terminate_process_group(process) + + _scrub_retained_artifacts(scenario_dir) + checks = _load_checks(scenario_dir, scenario) + adapter_success, adapter_detail = _load_adapter_report(adapter_report) + official_failure = any(check.status in {"FAILURE", "WARNING"} for check in checks) + success = ( + returncode == 0 and bool(checks) and adapter_success and not official_failure + ) + if not checks: + runner_detail = ( + "official runner did not produce exactly one checks.json; " + runner_detail + ).strip() + elif returncode != 0 and not official_failure and adapter_success: + runner_detail = ( + f"official runner exited with code {returncode}; " + runner_detail + ).strip() + return OfficialScenarioResult( + scenario=scenario, + success=success, + checks=checks, + adapter_success=adapter_success, + adapter_detail=adapter_detail, + runner_detail=runner_detail, + ) + + +def run_official_mode( + *, + conformance_command: Sequence[str], + adapter_script: Path, + codex_binary: Path, + mode: str, + scenarios: Sequence[str], + output_dir: Path, + timeout_seconds: float, + base_env: Mapping[str, str], +) -> list[OfficialScenarioResult]: + launcher = _make_adapter_launcher( + adapter_script, + require_automatic_auth=base_env.get("CODEX_CONFORMANCE_REQUIRE_AUTOMATIC_AUTH") + == "1", + ) + try: + return [ + _run_scenario( + conformance_command=conformance_command, + adapter_launcher=launcher, + codex_binary=codex_binary, + mode=mode, + scenario=scenario, + output_dir=output_dir, + timeout_seconds=timeout_seconds, + base_env=base_env, + ) + for scenario in scenarios + ] + finally: + try: + launcher.unlink() + launcher.parent.rmdir() + except OSError: + pass diff --git a/scripts/mcp_conformance/regression-baseline-v1.json b/scripts/mcp_conformance/regression-baseline-v1.json new file mode 100644 index 0000000000..08f01ed1fd --- /dev/null +++ b/scripts/mcp_conformance/regression-baseline-v1.json @@ -0,0 +1,4208 @@ +{ + "automaticAuthRequired": false, + "baselineKind": "mcp-conformance-regression-baseline-v1", + "checks": { + "failing": [ + { + "check_id": "harness/auth/metadata-var3/codex-adapter", + "mode": "2025-11-25", + "scenario": "auth/metadata-var3", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/pre-registration/codex-adapter", + "mode": "2025-11-25", + "scenario": "auth/pre-registration", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/scope-from-www-authenticate/codex-adapter", + "mode": "2025-11-25", + "scenario": "auth/scope-from-www-authenticate", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/scope-step-up/codex-adapter", + "mode": "2025-11-25", + "scenario": "auth/scope-step-up", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-11-25", + "scenario": "auth/metadata-var3", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-11-25", + "scenario": "auth/metadata-var3", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2025-11-25", + "scenario": "auth/metadata-var3", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-11-25", + "scenario": "auth/metadata-var3", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pre-registration-auth", + "mode": "2025-11-25", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "scope-from-www-authenticate", + "mode": "2025-11-25", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "scope-step-up-initial", + "mode": "2025-11-25", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-elicitation-sep1034-boolean-default", + "mode": "2025-11-25", + "scenario": "elicitation-sep1034-client-defaults", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-elicitation-sep1034-enum-default", + "mode": "2025-11-25", + "scenario": "elicitation-sep1034-client-defaults", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-elicitation-sep1034-integer-default", + "mode": "2025-11-25", + "scenario": "elicitation-sep1034-client-defaults", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-elicitation-sep1034-number-default", + "mode": "2025-11-25", + "scenario": "elicitation-sep1034-client-defaults", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-elicitation-sep1034-string-default", + "mode": "2025-11-25", + "scenario": "elicitation-sep1034-client-defaults", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "harness/auth/metadata-var3/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/metadata-var3", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/pre-registration/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/pre-registration", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/scope-from-www-authenticate/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/scope-from-www-authenticate", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/scope-step-up/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/scope-step-up", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/metadata-var3", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/metadata-var3", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/metadata-var3", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/metadata-var3", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pre-registration-auth", + "mode": "2026-07-28", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "scope-from-www-authenticate", + "mode": "2026-07-28", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "scope-step-up-initial", + "mode": "2026-07-28", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2350-scope-union-on-reauth", + "mode": "2026-07-28", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + } + ], + "passing": [ + { + "check_id": "harness/auth/token-endpoint-auth-basic/codex-adapter", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-basic", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/token-endpoint-auth-none/codex-adapter", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-none", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/token-endpoint-auth-post/codex-adapter", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-post", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/initialize/codex-adapter", + "mode": "2025-06-18", + "scenario": "initialize", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/tools_call/codex-adapter", + "mode": "2025-06-18", + "scenario": "tools_call", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-consistency", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-authorization", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-token", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-valid-uri", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-endpoint-auth-method", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-consistency", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-authorization", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-token", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-valid-uri", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-endpoint-auth-method", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-consistency", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-authorization", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-token", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-valid-uri", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-endpoint-auth-method", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2025-06-18", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "mcp-client-initialization", + "mode": "2025-06-18", + "scenario": "initialize", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "tool-add-numbers", + "mode": "2025-06-18", + "scenario": "tools_call", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "wire-schema-valid", + "mode": "2025-06-18", + "scenario": "tools_call", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "app_server_initialize", + "mode": "2025-06-18", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "echo_tool", + "mode": "2025-06-18", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "ephemeral_thread", + "mode": "2025-06-18", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "inventory", + "mode": "2025-06-18", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "isolated_config_cleanup", + "mode": "2025-06-18", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "mcp_add", + "mode": "2025-06-18", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "mcp_get", + "mode": "2025-06-18", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "mcp_remove", + "mode": "2025-06-18", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "unicode_resource_read", + "mode": "2025-06-18", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "harness/auth/basic-cimd/codex-adapter", + "mode": "2025-11-25", + "scenario": "auth/basic-cimd", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/metadata-default/codex-adapter", + "mode": "2025-11-25", + "scenario": "auth/metadata-default", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/metadata-var1/codex-adapter", + "mode": "2025-11-25", + "scenario": "auth/metadata-var1", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/metadata-var2/codex-adapter", + "mode": "2025-11-25", + "scenario": "auth/metadata-var2", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/scope-from-scopes-supported/codex-adapter", + "mode": "2025-11-25", + "scenario": "auth/scope-from-scopes-supported", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/scope-omitted-when-undefined/codex-adapter", + "mode": "2025-11-25", + "scenario": "auth/scope-omitted-when-undefined", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/scope-retry-limit/codex-adapter", + "mode": "2025-11-25", + "scenario": "auth/scope-retry-limit", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/token-endpoint-auth-basic/codex-adapter", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-basic", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/token-endpoint-auth-none/codex-adapter", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-none", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/token-endpoint-auth-post/codex-adapter", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-post", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/elicitation-sep1034-client-defaults/codex-adapter", + "mode": "2025-11-25", + "scenario": "elicitation-sep1034-client-defaults", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/initialize/codex-adapter", + "mode": "2025-11-25", + "scenario": "initialize", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/sse-retry/codex-adapter", + "mode": "2025-11-25", + "scenario": "sse-retry", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/tools_call/codex-adapter", + "mode": "2025-11-25", + "scenario": "tools_call", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-11-25", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-11-25", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "cimd-client-id-used", + "mode": "2025-11-25", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2025-11-25", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2025-11-25", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2025-11-25", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2025-11-25", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2025-11-25", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-11-25", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2025-11-25", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-11-25", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-11-25", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2025-11-25", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2025-11-25", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2025-11-25", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2025-11-25", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2025-11-25", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2025-11-25", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-11-25", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2025-11-25", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-11-25", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-11-25", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2025-11-25", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2025-11-25", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2025-11-25", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2025-11-25", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2025-11-25", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2025-11-25", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-11-25", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2025-11-25", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-11-25", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-11-25", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2025-11-25", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2025-11-25", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2025-11-25", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2025-11-25", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2025-11-25", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2025-11-25", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-11-25", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2025-11-25", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-11-25", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-11-25", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2025-11-25", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2025-11-25", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2025-11-25", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2025-11-25", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2025-11-25", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-11-25", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-11-25", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-11-25", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2025-11-25", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2025-11-25", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2025-11-25", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2025-11-25", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2025-11-25", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2025-11-25", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "scope-from-scopes-supported", + "mode": "2025-11-25", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-11-25", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2025-11-25", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-11-25", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-11-25", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2025-11-25", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2025-11-25", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2025-11-25", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2025-11-25", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2025-11-25", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2025-11-25", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-11-25", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2025-11-25", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-11-25", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-11-25", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2025-11-25", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2025-11-25", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2025-11-25", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2025-11-25", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2025-11-25", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2025-11-25", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "scope-omitted-when-undefined", + "mode": "2025-11-25", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-11-25", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2025-11-25", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-11-25", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-11-25", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2025-11-25", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2025-11-25", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2025-11-25", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2025-11-25", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2025-11-25", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2025-11-25", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "scope-retry-limit", + "mode": "2025-11-25", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-11-25", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-11-25", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-11-25", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2025-11-25", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2025-11-25", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2025-11-25", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2025-11-25", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2025-11-25", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2025-11-25", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "scope-step-up-escalation", + "mode": "2025-11-25", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-11-25", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2025-11-25", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-consistency", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-authorization", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-token", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-valid-uri", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-endpoint-auth-method", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-consistency", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-authorization", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-token", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-valid-uri", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-endpoint-auth-method", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-consistency", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-authorization", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-token", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-valid-uri", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-endpoint-auth-method", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2025-11-25", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "mcp-client-initialization", + "mode": "2025-11-25", + "scenario": "initialize", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-sse-graceful-reconnect", + "mode": "2025-11-25", + "scenario": "sse-retry", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-sse-last-event-id", + "mode": "2025-11-25", + "scenario": "sse-retry", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-sse-retry-timing", + "mode": "2025-11-25", + "scenario": "sse-retry", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "tool-add-numbers", + "mode": "2025-11-25", + "scenario": "tools_call", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "wire-schema-valid", + "mode": "2025-11-25", + "scenario": "tools_call", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "app_server_initialize", + "mode": "2025-11-25", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "echo_tool", + "mode": "2025-11-25", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "ephemeral_thread", + "mode": "2025-11-25", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "inventory", + "mode": "2025-11-25", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "isolated_config_cleanup", + "mode": "2025-11-25", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "mcp_add", + "mode": "2025-11-25", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "mcp_get", + "mode": "2025-11-25", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "mcp_remove", + "mode": "2025-11-25", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "unicode_resource_read", + "mode": "2025-11-25", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "harness/auth/authorization-server-migration/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/authorization-server-migration", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/basic-cimd/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/basic-cimd", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/iss-normalized/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/iss-normalized", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/iss-not-advertised/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/iss-not-advertised", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/iss-supported/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/iss-supported", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/iss-supported-missing/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/iss-supported-missing", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/iss-unexpected/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/iss-unexpected", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/iss-wrong-issuer/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/iss-wrong-issuer", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/metadata-default/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/metadata-default", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/metadata-issuer-mismatch/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/metadata-issuer-mismatch", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/metadata-var1/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/metadata-var1", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/metadata-var2/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/metadata-var2", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/offline-access-not-supported/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/offline-access-not-supported", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/offline-access-scope/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/offline-access-scope", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/resource-mismatch/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/resource-mismatch", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/scope-from-scopes-supported/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/scope-from-scopes-supported", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/scope-omitted-when-undefined/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/scope-omitted-when-undefined", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/scope-retry-limit/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/scope-retry-limit", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/token-endpoint-auth-basic/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/token-endpoint-auth-none/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/auth/token-endpoint-auth-post/codex-adapter", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/http-custom-headers/codex-adapter", + "mode": "2026-07-28", + "scenario": "http-custom-headers", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/http-invalid-tool-headers/codex-adapter", + "mode": "2026-07-28", + "scenario": "http-invalid-tool-headers", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/http-standard-headers/codex-adapter", + "mode": "2026-07-28", + "scenario": "http-standard-headers", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/json-schema-ref-no-deref/codex-adapter", + "mode": "2026-07-28", + "scenario": "json-schema-ref-no-deref", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/request-metadata/codex-adapter", + "mode": "2026-07-28", + "scenario": "request-metadata", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/sep-2322-client-request-state/codex-adapter", + "mode": "2026-07-28", + "scenario": "sep-2322-client-request-state", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "harness/tools_call/codex-adapter", + "mode": "2026-07-28", + "scenario": "tools_call", + "source": "harness", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/authorization-server-migration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/authorization-server-migration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/authorization-server-migration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/authorization-server-migration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/authorization-server-migration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/authorization-server-migration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/authorization-server-migration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/authorization-server-migration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2352-no-cross-as-credential-reuse", + "mode": "2026-07-28", + "scenario": "auth/authorization-server-migration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2352-no-reuse-on-as-change", + "mode": "2026-07-28", + "scenario": "auth/authorization-server-migration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2352-reregister-on-as-change", + "mode": "2026-07-28", + "scenario": "auth/authorization-server-migration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/authorization-server-migration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/authorization-server-migration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2026-07-28", + "scenario": "auth/authorization-server-migration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "cimd-client-id-used", + "mode": "2026-07-28", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2026-07-28", + "scenario": "auth/basic-cimd", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/iss-normalized", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/iss-normalized", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/iss-normalized", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/iss-normalized", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/iss-normalized", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/iss-normalized", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2468-client-no-normalization", + "mode": "2026-07-28", + "scenario": "auth/iss-normalized", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/iss-normalized", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/iss-not-advertised", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/iss-not-advertised", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/iss-not-advertised", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/iss-not-advertised", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/iss-not-advertised", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/iss-not-advertised", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/iss-not-advertised", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/iss-not-advertised", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2468-client-proceed-no-iss", + "mode": "2026-07-28", + "scenario": "auth/iss-not-advertised", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/iss-not-advertised", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/iss-not-advertised", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2026-07-28", + "scenario": "auth/iss-not-advertised", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/iss-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/iss-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/iss-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/iss-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/iss-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/iss-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/iss-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/iss-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2468-client-compare-iss-supported", + "mode": "2026-07-28", + "scenario": "auth/iss-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/iss-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/iss-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2026-07-28", + "scenario": "auth/iss-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/iss-supported-missing", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/iss-supported-missing", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/iss-supported-missing", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/iss-supported-missing", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/iss-supported-missing", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/iss-supported-missing", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2468-client-reject-missing-iss", + "mode": "2026-07-28", + "scenario": "auth/iss-supported-missing", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/iss-supported-missing", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/iss-unexpected", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/iss-unexpected", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/iss-unexpected", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/iss-unexpected", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/iss-unexpected", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/iss-unexpected", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2468-client-compare-iss-unadvertised", + "mode": "2026-07-28", + "scenario": "auth/iss-unexpected", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/iss-unexpected", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/iss-wrong-issuer", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/iss-wrong-issuer", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/iss-wrong-issuer", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/iss-wrong-issuer", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/iss-wrong-issuer", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/iss-wrong-issuer", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2468-client-compare-iss-supported", + "mode": "2026-07-28", + "scenario": "auth/iss-wrong-issuer", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/iss-wrong-issuer", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2026-07-28", + "scenario": "auth/metadata-default", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/metadata-issuer-mismatch", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/metadata-issuer-mismatch", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2468-client-validate-metadata-issuer", + "mode": "2026-07-28", + "scenario": "auth/metadata-issuer-mismatch", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2026-07-28", + "scenario": "auth/metadata-var1", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2026-07-28", + "scenario": "auth/metadata-var2", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/offline-access-not-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/offline-access-not-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/offline-access-not-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/offline-access-not-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/offline-access-not-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/offline-access-not-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/offline-access-not-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/offline-access-not-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2207-offline-access-not-requested", + "mode": "2026-07-28", + "scenario": "auth/offline-access-not-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/offline-access-not-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/offline-access-not-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2026-07-28", + "scenario": "auth/offline-access-not-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/offline-access-scope", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/offline-access-scope", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/offline-access-scope", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/offline-access-scope", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/offline-access-scope", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/offline-access-scope", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/offline-access-scope", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/offline-access-scope", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2207-client-metadata-grant-types", + "mode": "2026-07-28", + "scenario": "auth/offline-access-scope", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2207-offline-access-requested", + "mode": "2026-07-28", + "scenario": "auth/offline-access-scope", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/offline-access-scope", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/offline-access-scope", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2026-07-28", + "scenario": "auth/offline-access-scope", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/pre-registration", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/resource-mismatch", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-mismatch-rejected", + "mode": "2026-07-28", + "scenario": "auth/resource-mismatch", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "scope-from-scopes-supported", + "mode": "2026-07-28", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2026-07-28", + "scenario": "auth/scope-from-scopes-supported", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2026-07-28", + "scenario": "auth/scope-from-www-authenticate", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "scope-omitted-when-undefined", + "mode": "2026-07-28", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2026-07-28", + "scenario": "auth/scope-omitted-when-undefined", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "scope-retry-limit", + "mode": "2026-07-28", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/scope-retry-limit", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "scope-step-up-escalation", + "mode": "2026-07-28", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2026-07-28", + "scenario": "auth/scope-step-up", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-consistency", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-authorization", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-token", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-valid-uri", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-endpoint-auth-method", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-basic", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-consistency", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-authorization", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-token", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-valid-uri", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-endpoint-auth-method", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-none", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-request", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "authorization-server-metadata", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "client-registration", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-challenge-sent", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-code-verifier-sent", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-s256-method-used", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "pkce-verifier-matches-challenge", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "prm-pathbased-requested", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-consistency", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-authorization", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-in-token", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "resource-parameter-valid-uri", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-837-application-type-present", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-endpoint-auth-method", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "token-request", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "valid-bearer-token", + "mode": "2026-07-28", + "scenario": "auth/token-endpoint-auth-post", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2243-client-base64-unsafe", + "mode": "2026-07-28", + "scenario": "http-custom-headers", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2243-client-encode-values", + "mode": "2026-07-28", + "scenario": "http-custom-headers", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2243-client-mirrors-designated-params", + "mode": "2026-07-28", + "scenario": "http-custom-headers", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2243-client-omit-null", + "mode": "2026-07-28", + "scenario": "http-custom-headers", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2243-client-supports-custom-headers", + "mode": "2026-07-28", + "scenario": "http-custom-headers", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2243-client-reject-invalid-tool", + "mode": "2026-07-28", + "scenario": "http-invalid-tool-headers", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2243-x-mcp-header-charset", + "mode": "2026-07-28", + "scenario": "http-invalid-tool-headers", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2243-x-mcp-header-not-empty", + "mode": "2026-07-28", + "scenario": "http-invalid-tool-headers", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2243-x-mcp-header-primitive-only", + "mode": "2026-07-28", + "scenario": "http-invalid-tool-headers", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2243-x-mcp-header-unique", + "mode": "2026-07-28", + "scenario": "http-invalid-tool-headers", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2243-client-includes-standard-headers", + "mode": "2026-07-28", + "scenario": "http-standard-headers", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2106-no-network-ref-deref", + "mode": "2026-07-28", + "scenario": "json-schema-ref-no-deref", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2575-client-declares-elicitation-capability", + "mode": "2026-07-28", + "scenario": "request-metadata", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2575-client-populates-meta", + "mode": "2026-07-28", + "scenario": "request-metadata", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2575-client-retry-supported-version", + "mode": "2026-07-28", + "scenario": "request-metadata", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2575-client-sends-client-info", + "mode": "2026-07-28", + "scenario": "request-metadata", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2575-http-client-sends-version-header", + "mode": "2026-07-28", + "scenario": "request-metadata", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2575-http-version-header-matches-meta", + "mode": "2026-07-28", + "scenario": "request-metadata", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2322-client-jsonrpc-id-different", + "mode": "2026-07-28", + "scenario": "sep-2322-client-request-state", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2322-client-no-state-omitted", + "mode": "2026-07-28", + "scenario": "sep-2322-client-request-state", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2322-client-parallel-isolation", + "mode": "2026-07-28", + "scenario": "sep-2322-client-request-state", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2322-client-request-state-echoed", + "mode": "2026-07-28", + "scenario": "sep-2322-client-request-state", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "sep-2322-default-result-type-complete", + "mode": "2026-07-28", + "scenario": "sep-2322-client-request-state", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "tool-add-numbers", + "mode": "2026-07-28", + "scenario": "tools_call", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "wire-schema-valid", + "mode": "2026-07-28", + "scenario": "tools_call", + "source": "official", + "transport": "official-http" + }, + { + "check_id": "app_server_initialize", + "mode": "2026-07-28", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "echo_tool", + "mode": "2026-07-28", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "ephemeral_thread", + "mode": "2026-07-28", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "inventory", + "mode": "2026-07-28", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "isolated_config_cleanup", + "mode": "2026-07-28", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "mcp_add", + "mode": "2026-07-28", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "mcp_get", + "mode": "2026-07-28", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "mcp_remove", + "mode": "2026-07-28", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "modern_feature_enablement", + "mode": "2026-07-28", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "multi_round_trip_request", + "mode": "2026-07-28", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "per_request_metadata", + "mode": "2026-07-28", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "request_scoped_notifications", + "mode": "2026-07-28", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + }, + { + "check_id": "unicode_resource_read", + "mode": "2026-07-28", + "scenario": "", + "source": "supplemental", + "transport": "stdio" + } + ] + }, + "modernFeatureEnablement": true, + "officialConformance": { + "authenticationIncluded": true, + "gitRef": "49103de6ed70804e940637bf3e9e29e4a3f54e64", + "repository": "modelcontextprotocol/conformance", + "scenarios": { + "2025-06-18": [ + "initialize", + "tools_call", + "auth/token-endpoint-auth-basic", + "auth/token-endpoint-auth-post", + "auth/token-endpoint-auth-none" + ], + "2025-11-25": [ + "initialize", + "tools_call", + "elicitation-sep1034-client-defaults", + "sse-retry", + "auth/metadata-default", + "auth/metadata-var1", + "auth/metadata-var2", + "auth/metadata-var3", + "auth/basic-cimd", + "auth/scope-from-www-authenticate", + "auth/scope-from-scopes-supported", + "auth/scope-omitted-when-undefined", + "auth/scope-step-up", + "auth/scope-retry-limit", + "auth/token-endpoint-auth-basic", + "auth/token-endpoint-auth-post", + "auth/token-endpoint-auth-none", + "auth/pre-registration" + ], + "2026-07-28": [ + "tools_call", + "request-metadata", + "sep-2322-client-request-state", + "http-standard-headers", + "http-custom-headers", + "http-invalid-tool-headers", + "json-schema-ref-no-deref", + "auth/metadata-default", + "auth/metadata-var1", + "auth/metadata-var2", + "auth/metadata-var3", + "auth/basic-cimd", + "auth/scope-from-www-authenticate", + "auth/scope-from-scopes-supported", + "auth/scope-omitted-when-undefined", + "auth/scope-step-up", + "auth/scope-retry-limit", + "auth/token-endpoint-auth-basic", + "auth/token-endpoint-auth-post", + "auth/token-endpoint-auth-none", + "auth/pre-registration", + "auth/resource-mismatch", + "auth/offline-access-scope", + "auth/offline-access-not-supported", + "auth/authorization-server-migration", + "auth/iss-supported", + "auth/iss-not-advertised", + "auth/iss-supported-missing", + "auth/iss-wrong-issuer", + "auth/iss-unexpected", + "auth/iss-normalized", + "auth/metadata-issuer-mismatch" + ] + } + }, + "requiredModes": [ + "2025-06-18", + "2025-11-25", + "2026-07-28" + ], + "schemaVersion": 4, + "transports": [ + "stdio", + "official-http" + ], + "versionCheck": { + "success": true + } +} diff --git a/scripts/mcp_conformance/review-regression-baseline-v1.json b/scripts/mcp_conformance/review-regression-baseline-v1.json new file mode 100644 index 0000000000..5688eafc04 --- /dev/null +++ b/scripts/mcp_conformance/review-regression-baseline-v1.json @@ -0,0 +1,1039 @@ +{ + "baselineKind": "mcp-review-regression-baseline-v1", + "checks": { + "failing": [ + { + "check_id": "review/catalog-over-limit/exact-limit-error", + "mode": "2025-06-18", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/server-info", + "mode": "2025-06-18", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/startup-failed", + "mode": "2025-06-18", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/tool-count", + "mode": "2025-06-18", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/exact-limit-error", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/server-info", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/startup-failed", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/tool-count", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/exact-limit-error", + "mode": "2025-11-25", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/server-info", + "mode": "2025-11-25", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/startup-failed", + "mode": "2025-11-25", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/tool-count", + "mode": "2025-11-25", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/exact-limit-error", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/server-info", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/startup-failed", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/tool-count", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/exact-limit-error", + "mode": "2026-07-28", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/server-info", + "mode": "2026-07-28", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/startup-failed", + "mode": "2026-07-28", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/tool-count", + "mode": "2026-07-28", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/exact-large-integer-elicitation-round-trip", + "mode": "2026-07-28", + "transport": "review-http:review-regressions" + }, + { + "check_id": "review/exact-large-integer-elicitation-schema", + "mode": "2026-07-28", + "transport": "review-http:review-regressions" + }, + { + "check_id": "review/mrtr-input-requests-bounded-to-64", + "mode": "2026-07-28", + "transport": "review-http:review-regressions" + }, + { + "check_id": "review/catalog-over-limit/exact-limit-error", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/server-info", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/startup-failed", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/tool-count", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/exact-large-integer-elicitation-round-trip", + "mode": "2026-07-28", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/exact-large-integer-elicitation-schema", + "mode": "2026-07-28", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/mrtr-input-requests-bounded-to-64", + "mode": "2026-07-28", + "transport": "review-stdio:review-regressions" + } + ], + "passing": [ + { + "check_id": "review/app-server-initialize", + "mode": "2025-06-18", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/configured-server", + "mode": "2025-06-18", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/server-info", + "mode": "2025-06-18", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/startup-ready", + "mode": "2025-06-18", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/tool-count", + "mode": "2025-06-18", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2025-06-18", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2025-06-18", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/mcp-registration", + "mode": "2025-06-18", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2025-06-18", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/configured-server", + "mode": "2025-06-18", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2025-06-18", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2025-06-18", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/mcp-registration", + "mode": "2025-06-18", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/configured-server", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/server-info", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/startup-ready", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/tool-count", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/mcp-registration", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/configured-server", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/mcp-registration", + "mode": "2025-06-18", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2025-06-18", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2025-06-18", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/exact-large-integer-tool-round-trip", + "mode": "2025-06-18", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/exact-large-integer-tool-schema", + "mode": "2025-06-18", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2025-06-18", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/legacy-reserved-stdio-environment-preserved", + "mode": "2025-06-18", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/mcp-registration", + "mode": "2025-06-18", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/server-discovery", + "mode": "2025-06-18", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2025-11-25", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/configured-server", + "mode": "2025-11-25", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/server-info", + "mode": "2025-11-25", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/startup-ready", + "mode": "2025-11-25", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/tool-count", + "mode": "2025-11-25", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2025-11-25", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2025-11-25", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/mcp-registration", + "mode": "2025-11-25", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2025-11-25", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/configured-server", + "mode": "2025-11-25", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2025-11-25", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2025-11-25", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/mcp-registration", + "mode": "2025-11-25", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/configured-server", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/server-info", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/startup-ready", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/tool-count", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/mcp-registration", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/configured-server", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/mcp-registration", + "mode": "2025-11-25", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2025-11-25", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2025-11-25", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/exact-large-integer-tool-round-trip", + "mode": "2025-11-25", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/exact-large-integer-tool-schema", + "mode": "2025-11-25", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2025-11-25", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/legacy-reserved-stdio-environment-preserved", + "mode": "2025-11-25", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/mcp-registration", + "mode": "2025-11-25", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/server-discovery", + "mode": "2025-11-25", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2026-07-28", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/configured-server", + "mode": "2026-07-28", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/server-info", + "mode": "2026-07-28", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/startup-ready", + "mode": "2026-07-28", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/tool-count", + "mode": "2026-07-28", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2026-07-28", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2026-07-28", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/mcp-registration", + "mode": "2026-07-28", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/modern-feature-configuration", + "mode": "2026-07-28", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/modern-feature-enablement", + "mode": "2026-07-28", + "transport": "review-http:catalog-max" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2026-07-28", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/configured-server", + "mode": "2026-07-28", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2026-07-28", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2026-07-28", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/mcp-registration", + "mode": "2026-07-28", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/modern-feature-configuration", + "mode": "2026-07-28", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/modern-feature-enablement", + "mode": "2026-07-28", + "transport": "review-http:catalog-over-limit" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2026-07-28", + "transport": "review-http:discovery-mismatched-id" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2026-07-28", + "transport": "review-http:discovery-mismatched-id" + }, + { + "check_id": "review/mcp-registration", + "mode": "2026-07-28", + "transport": "review-http:discovery-mismatched-id" + }, + { + "check_id": "review/modern-feature-configuration", + "mode": "2026-07-28", + "transport": "review-http:discovery-mismatched-id" + }, + { + "check_id": "review/modern-feature-enablement", + "mode": "2026-07-28", + "transport": "review-http:discovery-mismatched-id" + }, + { + "check_id": "review/reject-mismatched-discovery-response-id", + "mode": "2026-07-28", + "transport": "review-http:discovery-mismatched-id" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2026-07-28", + "transport": "review-http:discovery-null-id" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2026-07-28", + "transport": "review-http:discovery-null-id" + }, + { + "check_id": "review/mcp-registration", + "mode": "2026-07-28", + "transport": "review-http:discovery-null-id" + }, + { + "check_id": "review/modern-feature-configuration", + "mode": "2026-07-28", + "transport": "review-http:discovery-null-id" + }, + { + "check_id": "review/modern-feature-enablement", + "mode": "2026-07-28", + "transport": "review-http:discovery-null-id" + }, + { + "check_id": "review/reject-null-discovery-response-id", + "mode": "2026-07-28", + "transport": "review-http:discovery-null-id" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2026-07-28", + "transport": "review-http:repeated-cursor" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2026-07-28", + "transport": "review-http:repeated-cursor" + }, + { + "check_id": "review/mcp-registration", + "mode": "2026-07-28", + "transport": "review-http:repeated-cursor" + }, + { + "check_id": "review/modern-feature-configuration", + "mode": "2026-07-28", + "transport": "review-http:repeated-cursor" + }, + { + "check_id": "review/modern-feature-enablement", + "mode": "2026-07-28", + "transport": "review-http:repeated-cursor" + }, + { + "check_id": "review/repeated-pagination-cursor-bounded", + "mode": "2026-07-28", + "transport": "review-http:repeated-cursor" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2026-07-28", + "transport": "review-http:review-regressions" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2026-07-28", + "transport": "review-http:review-regressions" + }, + { + "check_id": "review/exact-large-integer-tool-round-trip", + "mode": "2026-07-28", + "transport": "review-http:review-regressions" + }, + { + "check_id": "review/exact-large-integer-tool-schema", + "mode": "2026-07-28", + "transport": "review-http:review-regressions" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2026-07-28", + "transport": "review-http:review-regressions" + }, + { + "check_id": "review/mcp-registration", + "mode": "2026-07-28", + "transport": "review-http:review-regressions" + }, + { + "check_id": "review/modern-feature-configuration", + "mode": "2026-07-28", + "transport": "review-http:review-regressions" + }, + { + "check_id": "review/modern-feature-enablement", + "mode": "2026-07-28", + "transport": "review-http:review-regressions" + }, + { + "check_id": "review/modern-paginated-tools-page-two", + "mode": "2026-07-28", + "transport": "review-http:review-regressions" + }, + { + "check_id": "review/server-discovery", + "mode": "2026-07-28", + "transport": "review-http:review-regressions" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2026-07-28", + "transport": "review-http:sse-comment-flood" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2026-07-28", + "transport": "review-http:sse-comment-flood" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2026-07-28", + "transport": "review-http:sse-comment-flood" + }, + { + "check_id": "review/mcp-registration", + "mode": "2026-07-28", + "transport": "review-http:sse-comment-flood" + }, + { + "check_id": "review/modern-feature-configuration", + "mode": "2026-07-28", + "transport": "review-http:sse-comment-flood" + }, + { + "check_id": "review/modern-feature-enablement", + "mode": "2026-07-28", + "transport": "review-http:sse-comment-flood" + }, + { + "check_id": "review/sse-comment-flood-excluded-from-event-size-limit", + "mode": "2026-07-28", + "transport": "review-http:sse-comment-flood" + }, + { + "check_id": "review/sse-cr-discovery", + "mode": "2026-07-28", + "transport": "review-http:sse-comment-flood" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2026-07-28", + "transport": "review-http:sse-cr-comments" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2026-07-28", + "transport": "review-http:sse-cr-comments" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2026-07-28", + "transport": "review-http:sse-cr-comments" + }, + { + "check_id": "review/mcp-registration", + "mode": "2026-07-28", + "transport": "review-http:sse-cr-comments" + }, + { + "check_id": "review/modern-feature-configuration", + "mode": "2026-07-28", + "transport": "review-http:sse-cr-comments" + }, + { + "check_id": "review/modern-feature-enablement", + "mode": "2026-07-28", + "transport": "review-http:sse-cr-comments" + }, + { + "check_id": "review/sse-carriage-return-and-comment-framing", + "mode": "2026-07-28", + "transport": "review-http:sse-cr-comments" + }, + { + "check_id": "review/sse-cr-discovery", + "mode": "2026-07-28", + "transport": "review-http:sse-cr-comments" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/configured-server", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/server-info", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/startup-ready", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/catalog-at-limit/tool-count", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/mcp-registration", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/modern-feature-configuration", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/modern-feature-enablement", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-max" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/catalog-over-limit/configured-server", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/mcp-registration", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/modern-feature-configuration", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/modern-feature-enablement", + "mode": "2026-07-28", + "transport": "review-stdio:catalog-over-limit" + }, + { + "check_id": "review/app-server-initialize", + "mode": "2026-07-28", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/ephemeral-thread", + "mode": "2026-07-28", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/exact-large-integer-tool-round-trip", + "mode": "2026-07-28", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/exact-large-integer-tool-schema", + "mode": "2026-07-28", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/isolated-registration-cleanup", + "mode": "2026-07-28", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/mcp-registration", + "mode": "2026-07-28", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/modern-feature-configuration", + "mode": "2026-07-28", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/modern-feature-enablement", + "mode": "2026-07-28", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/modern-paginated-tools-page-two", + "mode": "2026-07-28", + "transport": "review-stdio:review-regressions" + }, + { + "check_id": "review/server-discovery", + "mode": "2026-07-28", + "transport": "review-stdio:review-regressions" + } + ] + }, + "requiredCases": [ + { + "mode": "2025-06-18", + "transport": "review-http:catalog-max" + }, + { + "mode": "2025-06-18", + "transport": "review-http:catalog-over-limit" + }, + { + "mode": "2025-06-18", + "transport": "review-stdio:catalog-max" + }, + { + "mode": "2025-06-18", + "transport": "review-stdio:catalog-over-limit" + }, + { + "mode": "2025-06-18", + "transport": "review-stdio:review-regressions" + }, + { + "mode": "2025-11-25", + "transport": "review-http:catalog-max" + }, + { + "mode": "2025-11-25", + "transport": "review-http:catalog-over-limit" + }, + { + "mode": "2025-11-25", + "transport": "review-stdio:catalog-max" + }, + { + "mode": "2025-11-25", + "transport": "review-stdio:catalog-over-limit" + }, + { + "mode": "2025-11-25", + "transport": "review-stdio:review-regressions" + }, + { + "mode": "2026-07-28", + "transport": "review-http:catalog-max" + }, + { + "mode": "2026-07-28", + "transport": "review-http:catalog-over-limit" + }, + { + "mode": "2026-07-28", + "transport": "review-http:discovery-mismatched-id" + }, + { + "mode": "2026-07-28", + "transport": "review-http:discovery-null-id" + }, + { + "mode": "2026-07-28", + "transport": "review-http:repeated-cursor" + }, + { + "mode": "2026-07-28", + "transport": "review-http:review-regressions" + }, + { + "mode": "2026-07-28", + "transport": "review-http:sse-comment-flood" + }, + { + "mode": "2026-07-28", + "transport": "review-http:sse-cr-comments" + }, + { + "mode": "2026-07-28", + "transport": "review-stdio:catalog-max" + }, + { + "mode": "2026-07-28", + "transport": "review-stdio:catalog-over-limit" + }, + { + "mode": "2026-07-28", + "transport": "review-stdio:review-regressions" + } + ], + "requiredModes": [ + "2025-06-18", + "2025-11-25", + "2026-07-28" + ], + "reviewer": "codex-mcp-regression-suite", + "schemaVersion": 1, + "summary": { + "casesPassed": 13, + "casesTotal": 21, + "failed": 30, + "passed": 156, + "total": 186 + } +} diff --git a/scripts/mcp_conformance/review_regressions.py b/scripts/mcp_conformance/review_regressions.py new file mode 100644 index 0000000000..7a9d059e45 --- /dev/null +++ b/scripts/mcp_conformance/review_regressions.py @@ -0,0 +1,1294 @@ +#!/usr/bin/env python3 +"""Run reviewer-derived MCP regressions against a real Codex app-server.""" + +import argparse +import json +import os +import shutil +import sys +import tempfile +import threading +import time +from contextlib import contextmanager, nullcontext +from dataclasses import asdict, dataclass +from pathlib import Path +from typing import Iterator, Mapping, Sequence + +_MODULE_DIR = Path(__file__).resolve().parent +if str(_MODULE_DIR) not in sys.path: + sys.path.insert(0, str(_MODULE_DIR)) + +from run_codex_compliance import ( # noqa: E402 - direct scripts must first add their sibling directory. + LEGACY_VERSION, + MISMATCHED_DISCOVERY_ID_PROFILE, + MODERN_VERSION, + NULL_DISCOVERY_ID_PROFILE, + REPEATED_CURSOR_PROFILE, + REVIEW_EXACT_INTEGER, + REVIEW_MRTR_INPUT_REQUEST_COUNT, + REVIEW_PROFILE, + SHIPPING_LEGACY_VERSION, + SSE_COMMENT_FLOOD_PROFILE, + SSE_CR_COMMENTS_PROFILE, + TEST_SERVER_NAME, + AppServerClient, + AppServerError, + CaseResult, + ProtocolServer, + _call_tool, + _command_detail, + _isolated_environment, + _response_result, + _run_command, + make_http_server, +) +from server import ( # noqa: E402 - direct scripts must first add their sibling directory. + CATALOG_MAX_PROFILE, + CATALOG_OVER_LIMIT_PROFILE, + MAX_CATALOG_ITEMS, +) + +REVIEW_MODES = (SHIPPING_LEGACY_VERSION, LEGACY_VERSION, MODERN_VERSION) +REVIEW_REPORT_SCHEMA_VERSION = 1 +REVIEW_BASELINE_KIND = "mcp-review-regression-baseline-v1" +REVIEWER = "codex-mcp-regression-suite" +LEGACY_ENVIRONMENT_SENTINEL = "review-legacy-protocol-environment" +MRTR_REQUEST_LIMIT = 64 +CATALOG_BOUNDARY_PROFILES = (CATALOG_MAX_PROFILE, CATALOG_OVER_LIMIT_PROFILE) +CATALOG_BOUNDARY_TRANSPORTS = ("stdio", "http") +CATALOG_LIMIT_ERROR = ( + f"tools/list exceeded the catalog limit of {MAX_CATALOG_ITEMS} items" +) + + +@dataclass(frozen=True, order=True) +class _ReviewCheckIdentity: + mode: str + transport: str + check_id: str + + +def _required_review_cases() -> set[tuple[str, str]]: + cases = {(mode, f"review-stdio:{REVIEW_PROFILE}") for mode in REVIEW_MODES} + cases.update( + (MODERN_VERSION, f"review-http:{profile}") + for profile in ( + REVIEW_PROFILE, + REPEATED_CURSOR_PROFILE, + MISMATCHED_DISCOVERY_ID_PROFILE, + NULL_DISCOVERY_ID_PROFILE, + SSE_CR_COMMENTS_PROFILE, + SSE_COMMENT_FLOOD_PROFILE, + ) + ) + cases.update( + (mode, f"review-{transport}:{profile}") + for mode in REVIEW_MODES + for transport in CATALOG_BOUNDARY_TRANSPORTS + for profile in CATALOG_BOUNDARY_PROFILES + ) + return cases + + +@contextmanager +def _running_review_http_fixture(mode: str, profile: str) -> Iterator[str]: + server = make_http_server( + ProtocolServer(mode, profile=profile), + "127.0.0.1", + 0, + log_requests=False, + ) + thread = threading.Thread( + target=server.serve_forever, + name=f"mcp-review-{profile}-{mode}", + daemon=True, + ) + thread.start() + try: + _, port = server.server_address + yield f"http://127.0.0.1:{port}/mcp" + finally: + server.shutdown() + server.server_close() + thread.join(timeout=3) + + +def _review_registration_command( + codex_binary: Path, + server_script: Path, + *, + transport: str, + mode: str, + profile: str, + http_url: str | None, +) -> list[str]: + command = [str(codex_binary), "mcp", "add", TEST_SERVER_NAME] + if transport == "stdio": + protocol_environment = ( + MODERN_VERSION if mode == MODERN_VERSION else LEGACY_ENVIRONMENT_SENTINEL + ) + command.extend( + [ + "--env", + f"CODEX_MCP_PROTOCOL_VERSION={protocol_environment}", + "--", + sys.executable, + str(server_script), + "--mode", + mode, + "--transport", + "stdio", + "--profile", + profile, + ] + ) + else: + if http_url is None: + raise ValueError("review HTTP transport requires a fixture URL") + command.extend(["--url", http_url]) + return command + + +def _review_inventory_entry(result: Mapping[str, object]) -> dict[str, object] | None: + entries = result.get("data") + if not isinstance(entries, list): + return None + for entry in entries: + if isinstance(entry, dict) and entry.get("name") == TEST_SERVER_NAME: + return entry + return None + + +def _tool_input_schema( + entry: Mapping[str, object], + tool_name: str, +) -> Mapping[str, object] | None: + tools = entry.get("tools") + if not isinstance(tools, dict): + return None + tool = tools.get(tool_name) + if not isinstance(tool, dict): + return None + schema = tool.get("inputSchema") + if not isinstance(schema, dict): + schema = tool.get("input_schema") + return schema if isinstance(schema, dict) else None + + +def _exact_integer_property(schema: Mapping[str, object] | None) -> bool: + if schema is None: + return False + properties = schema.get("properties") + if not isinstance(properties, dict): + return False + value = properties.get("value") + return ( + isinstance(value, dict) + and value.get("type") == "integer" + and isinstance(value.get("minimum"), int) + and not isinstance(value.get("minimum"), bool) + and value.get("minimum") == REVIEW_EXACT_INTEGER + and isinstance(value.get("default"), int) + and not isinstance(value.get("default"), bool) + and value.get("default") == REVIEW_EXACT_INTEGER + ) + + +def _elicitation_schema(request: Mapping[str, object]) -> Mapping[str, object] | None: + for key in ("requestedSchema", "requested_schema", "schema"): + schema = request.get(key) + if isinstance(schema, dict): + return schema + for key in ("request", "elicitation", "params"): + nested = request.get(key) + if isinstance(nested, dict): + found = _elicitation_schema(nested) + if found is not None: + return found + return None + + +def _review_elicitation_content(params: Mapping[str, object]) -> Mapping[str, object]: + schema = _elicitation_schema(params) + if _exact_integer_property(schema): + return {"value": REVIEW_EXACT_INTEGER} + return {"value": "review-confirmed", "confirmation": "confirmed"} + + +def _mrtr_budget_is_bounded( + request_count: int, + *, + elapsed_seconds: float, + timeout_seconds: float, +) -> bool: + return ( + 0 <= request_count <= MRTR_REQUEST_LIMIT + and elapsed_seconds <= timeout_seconds + 1 + ) + + +def _initialize_review_client( + case: CaseResult, + client: AppServerClient, + *, + mode: str, +) -> bool: + response, detail = _response_result( + client.request( + "initialize", + { + "clientInfo": { + "name": "mcp-review-regression-runner", + "title": "MCP review regression runner", + "version": "1.0.0", + }, + "capabilities": { + "experimentalApi": True, + "requestAttestation": False, + "mcpServerOpenaiFormElicitation": True, + }, + }, + ) + ) + if not case.check("review/app-server-initialize", response is not None, detail): + return False + client.notify("initialized") + if mode != MODERN_VERSION: + return True + + feature, feature_detail = _response_result( + client.request( + "experimentalFeature/enablement/set", + {"enablement": {"mcp_2026_07_28": True}}, + ) + ) + enabled = ( + feature is not None + and isinstance(feature.get("enablement"), dict) + and feature["enablement"].get("mcp_2026_07_28") is True + ) + return case.check( + "review/modern-feature-enablement", + enabled, + "enabled the production modern MCP feature" if enabled else feature_detail, + ) + + +def _review_thread_id( + case: CaseResult, + client: AppServerClient, + workspace: Path, +) -> str | None: + response, detail = _response_result( + client.request("thread/start", {"cwd": str(workspace), "ephemeral": True}) + ) + thread = response.get("thread") if response is not None else None + thread_id = thread.get("id") if isinstance(thread, dict) else None + if not case.check( + "review/ephemeral-thread", + isinstance(thread_id, str), + "created isolated review thread" if isinstance(thread_id, str) else detail, + ): + return None + return str(thread_id) + + +def _run_normal_review_checks( + case: CaseResult, + client: AppServerClient, + *, + mode: str, + workspace: Path, +) -> None: + inventory, inventory_detail = _response_result( + client.request("mcpServerStatus/list", {"detail": "full"}) + ) + entry = _review_inventory_entry(inventory) if inventory is not None else None + if not case.check( + "review/server-discovery", + entry is not None, + "discovered the adversarial review fixture" + if entry is not None + else inventory_detail, + ): + return + assert entry is not None + + tools = entry.get("tools") + if not isinstance(tools, dict): + tools = {} + if mode == MODERN_VERSION: + required_page_two = { + "review_integer_elicitation", + "review_large_integer", + "review_mrtr_cap", + "review_protocol_env", + } + missing = sorted(required_page_two - set(tools)) + case.check( + "review/modern-paginated-tools-page-two", + not missing, + "all protected second-page tools reached the app-server inventory" + if not missing + else "second-page tools missing: " + ", ".join(missing), + ) + + schema = _tool_input_schema(entry, "review_large_integer") + exact_schema = _exact_integer_property(schema) + case.check( + "review/exact-large-integer-tool-schema", + exact_schema, + "preserved integer minimum and default 9007199254740993" + if exact_schema + else "the app-server rounded or omitted the 2^53+1 tool schema", + ) + + thread_id = _review_thread_id(case, client, workspace) + if thread_id is None: + return + + integer, integer_detail = _call_tool( + client, + thread_id=thread_id, + tool="review_large_integer", + arguments={"value": REVIEW_EXACT_INTEGER}, + ) + content = integer.get("structuredContent") if integer is not None else None + exact_value = ( + isinstance(content, dict) + and isinstance(content.get("value"), int) + and not isinstance(content.get("value"), bool) + and content.get("value") == REVIEW_EXACT_INTEGER + ) + case.check( + "review/exact-large-integer-tool-round-trip", + exact_value, + "round-tripped 9007199254740993 without floating-point conversion" + if exact_value + else integer_detail, + ) + + if mode != MODERN_VERSION: + environment, environment_detail = _call_tool( + client, + thread_id=thread_id, + tool="review_protocol_env", + arguments={}, + ) + observed = ( + environment.get("structuredContent") if environment is not None else None + ) + preserved = ( + isinstance(observed, dict) + and observed.get("value") == LEGACY_ENVIRONMENT_SENTINEL + ) + case.check( + "review/legacy-reserved-stdio-environment-preserved", + preserved, + "forwarded the explicitly configured legacy protocol environment unchanged" + if preserved + else environment_detail, + ) + return + + before = len(client.elicitation_requests) + elicitation, elicitation_detail = _call_tool( + client, + thread_id=thread_id, + tool="review_integer_elicitation", + arguments={}, + ) + requests = client.elicitation_requests[before:] + exact_elicitation = any( + _exact_integer_property(_elicitation_schema(request)) for request in requests + ) + case.check( + "review/exact-large-integer-elicitation-schema", + exact_elicitation, + "app-server preserved 9007199254740993 in the actual elicitation schema" + if exact_elicitation + else "the production elicitation schema rounded or omitted the 2^53+1 integer", + ) + elicitation_content = ( + elicitation.get("structuredContent") if elicitation is not None else None + ) + completed = ( + isinstance(elicitation_content, dict) + and elicitation_content.get("value") == REVIEW_EXACT_INTEGER + ) + case.check( + "review/exact-large-integer-elicitation-round-trip", + completed, + "completed elicitation using the exact 2^53+1 integer" + if completed + else elicitation_detail, + ) + + mrtr_before = len(client.elicitation_requests) + started = time.monotonic() + mrtr_result = None + try: + mrtr_result, mrtr_detail = _call_tool( + client, + thread_id=thread_id, + tool="review_mrtr_cap", + arguments={}, + ) + except AppServerError as exc: + mrtr_detail = str(exc) + elapsed = time.monotonic() - started + observed = len(client.elicitation_requests) - mrtr_before + bounded = mrtr_result is None and _mrtr_budget_is_bounded( + observed, + elapsed_seconds=elapsed, + timeout_seconds=client.timeout_seconds, + ) + case.check( + "review/mrtr-input-requests-bounded-to-64", + bounded, + ( + f"rejected {REVIEW_MRTR_INPUT_REQUEST_COUNT} simultaneous input requests; " + f"observed {observed} elicitations in {elapsed:.2f}s" + if bounded + else ( + f"observed {observed} elicitation requests for a " + f"{REVIEW_MRTR_INPUT_REQUEST_COUNT}-request response in " + f"{elapsed:.2f}s; {mrtr_detail}" + ) + ), + ) + + +def _run_malformed_discovery_check( + case: CaseResult, + client: AppServerClient, + *, + profile: str, +) -> None: + name = ( + "review/reject-null-discovery-response-id" + if profile == NULL_DISCOVERY_ID_PROFILE + else "review/reject-mismatched-discovery-response-id" + ) + try: + inventory, detail = _response_result( + client.request("mcpServerStatus/list", {"detail": "full"}) + ) + except AppServerError as exc: + case.check(name, True, f"rejected malformed discovery response: {exc}") + return + entry = _review_inventory_entry(inventory) if inventory is not None else None + tools = entry.get("tools") if entry is not None else None + rejected = ( + inventory is None or entry is None or not isinstance(tools, dict) or not tools + ) + case.check( + name, + rejected, + "rejected the uncorrelated discovery response without silently downgrading" + if rejected + else "accepted an uncorrelated discovery response and exposed server tools", + ) + + +def _run_repeated_cursor_check(case: CaseResult, client: AppServerClient) -> None: + started = time.monotonic() + try: + inventory, _ = _response_result( + client.request("mcpServerStatus/list", {"detail": "full"}) + ) + entry = _review_inventory_entry(inventory) if inventory is not None else None + tools = entry.get("tools") if entry is not None else None + rejected = ( + inventory is None + or entry is None + or not isinstance(tools, dict) + or not tools + ) + detail = ( + "rejected the repeated pagination cursor within the startup deadline" + if rejected + else "accepted or exposed a catalog with a repeated pagination cursor" + ) + except AppServerError as exc: + rejected = True + detail = f"terminated repeated-cursor pagination: {exc}" + elapsed = time.monotonic() - started + case.check( + "review/repeated-pagination-cursor-bounded", + rejected and elapsed <= client.timeout_seconds + 2, + f"{detail}; elapsed={elapsed:.2f}s", + ) + + +def _run_catalog_boundary_checks( + case: CaseResult, + client: AppServerClient, + *, + profile: str, + workspace: Path, +) -> None: + if profile not in CATALOG_BOUNDARY_PROFILES: + raise ValueError(f"unknown catalog boundary profile: {profile}") + + at_limit = profile == CATALOG_MAX_PROFILE + check_prefix = ( + "review/catalog-at-limit" if at_limit else "review/catalog-over-limit" + ) + expected_status = "ready" if at_limit else "failed" + event_index = len(client.events) + if _review_thread_id(case, client, workspace) is None: + return + + try: + event = client.wait_for_notification( + "mcpServer/startupStatus/updated", + predicate=lambda params: ( + params.get("name") == TEST_SERVER_NAME + and params.get("status") in {"ready", "failed"} + ), + after_event_index=event_index, + ) + except AppServerError as exc: + case.check( + f"{check_prefix}/startup-{expected_status}", + False, + f"did not observe the required {expected_status} startup: {exc}", + ) + return + + params = event.get("params") + case.check( + f"{check_prefix}/startup-{expected_status}", + isinstance(params, dict) and params.get("status") == expected_status, + f"observed the expected {expected_status} MCP server startup" + if isinstance(params, dict) and params.get("status") == expected_status + else ( + "observed MCP server startup " + f"{params.get('status')!r}; expected {expected_status!r}" + if isinstance(params, dict) + else "MCP server startup notification did not contain parameters" + ), + ) + if not isinstance(params, dict): + return + + if not at_limit: + error = params.get("error") + case.check( + f"{check_prefix}/exact-limit-error", + isinstance(error, str) and CATALOG_LIMIT_ERROR in error, + error + if isinstance(error, str) + else f"startup failure did not report {CATALOG_LIMIT_ERROR!r}", + ) + + inventory, detail = _response_result( + client.request("mcpServerStatus/list", {"detail": "full"}) + ) + entry = _review_inventory_entry(inventory) if inventory is not None else None + if not case.check( + f"{check_prefix}/configured-server", + entry is not None, + "retained the configured catalog-boundary server" + if entry is not None + else detail, + ): + return + assert entry is not None + + server_info = entry.get("serverInfo") + expected_server_info = ( + isinstance(server_info, dict) if at_limit else server_info is None + ) + case.check( + f"{check_prefix}/server-info", + expected_server_info, + "exposed initialized server information" + if at_limit and expected_server_info + else "left rejected server information null" + if not at_limit and expected_server_info + else "catalog server information did not match its startup status", + ) + + tools = entry.get("tools") + actual_count = len(tools) if isinstance(tools, dict) else None + expected_count = MAX_CATALOG_ITEMS if at_limit else 0 + case.check( + f"{check_prefix}/tool-count", + isinstance(tools, dict) and actual_count == expected_count, + f"discovered exactly {expected_count} tools" + if actual_count == expected_count + else f"expected {expected_count} discovered tools, observed {actual_count}", + ) + + +def _run_sse_check( + case: CaseResult, + client: AppServerClient, + workspace: Path, + *, + profile: str, +) -> None: + inventory, detail = _response_result( + client.request("mcpServerStatus/list", {"detail": "full"}) + ) + entry = _review_inventory_entry(inventory) if inventory is not None else None + if not case.check( + "review/sse-cr-discovery", + entry is not None, + "discovered the CR-only SSE fixture" if entry is not None else detail, + ): + return + thread_id = _review_thread_id(case, client, workspace) + if thread_id is None: + return + result, detail = _call_tool( + client, + thread_id=thread_id, + tool="progress", + arguments={}, + meta={"progressToken": "review-cr-only-sse"}, + ) + content = result.get("content") if result is not None else None + accepted = isinstance(content, list) and bool(content) + flood = profile == SSE_COMMENT_FLOOD_PROFILE + case.check( + "review/sse-comment-flood-excluded-from-event-size-limit" + if flood + else "review/sse-carriage-return-and-comment-framing", + accepted, + "accepted more than 8 MiB of valid SSE comment keepalives" + if accepted and flood + else "accepted valid CR-only SSE lines and ignored comment keepalives" + if accepted + else detail, + ) + + +def _run_review_case( + codex_binary: Path, + server_script: Path, + *, + mode: str, + profile: str, + transport: str, + case_home: Path, + timeout_seconds: float, +) -> CaseResult: + started = time.monotonic() + case = CaseResult(transport=f"review-{transport}:{profile}", mode=mode) + case_home.mkdir(parents=True, exist_ok=True) + workspace = case_home / "workspace" + workspace.mkdir(exist_ok=True) + env = _isolated_environment(case_home) + registered = False + fixture = ( + _running_review_http_fixture(mode, profile) + if transport == "http" + else nullcontext(None) + ) + + try: + if mode == MODERN_VERSION: + feature = _run_command( + [str(codex_binary), "features", "enable", "mcp_2026_07_28"], + env=env, + cwd=workspace, + timeout_seconds=timeout_seconds, + ) + if not case.check( + "review/modern-feature-configuration", + feature.returncode == 0, + _command_detail(feature), + ): + return case + + with fixture as http_url: + registration = _run_command( + _review_registration_command( + codex_binary, + server_script, + transport=transport, + mode=mode, + profile=profile, + http_url=http_url, + ), + env=env, + cwd=workspace, + timeout_seconds=timeout_seconds, + ) + registered = case.check( + "review/mcp-registration", + registration.returncode == 0, + _command_detail(registration), + ) + if not registered: + return case + + with AppServerClient( + codex_binary, + env=env, + cwd=workspace, + timeout_seconds=timeout_seconds, + elicitation_content=_review_elicitation_content, + ) as client: + if not _initialize_review_client(case, client, mode=mode): + return case + if profile in ( + MISMATCHED_DISCOVERY_ID_PROFILE, + NULL_DISCOVERY_ID_PROFILE, + ): + _run_malformed_discovery_check(case, client, profile=profile) + elif profile == REPEATED_CURSOR_PROFILE: + _run_repeated_cursor_check(case, client) + elif profile in CATALOG_BOUNDARY_PROFILES: + _run_catalog_boundary_checks( + case, client, profile=profile, workspace=workspace + ) + elif profile in (SSE_CR_COMMENTS_PROFILE, SSE_COMMENT_FLOOD_PROFILE): + _run_sse_check(case, client, workspace, profile=profile) + else: + _run_normal_review_checks( + case, client, mode=mode, workspace=workspace + ) + except (AppServerError, OSError, ValueError) as exc: + case.check("review/client-runtime", False, str(exc)) + finally: + if registered: + removal = _run_command( + [str(codex_binary), "mcp", "remove", TEST_SERVER_NAME], + env=env, + cwd=workspace, + timeout_seconds=timeout_seconds, + ) + case.check( + "review/isolated-registration-cleanup", + removal.returncode == 0, + _command_detail(removal), + ) + case.finish(started) + return case + + +def run_review_regressions( + codex_binary: Path, + *, + modes: Sequence[str] = REVIEW_MODES, + server_script: Path | None = None, + timeout_seconds: float = 8, + artifact_parent: Path | None = None, + keep_artifacts: bool = False, +) -> dict[str, object]: + server_script = server_script or _MODULE_DIR / "server.py" + root = Path(tempfile.mkdtemp(prefix="codex-mcp-review-", dir=artifact_parent)) + cases: list[CaseResult] = [] + try: + for mode in modes: + cases.append( + _run_review_case( + codex_binary, + server_script, + mode=mode, + profile=REVIEW_PROFILE, + transport="stdio", + case_home=root / f"review-stdio-{mode}", + timeout_seconds=timeout_seconds, + ) + ) + if mode != MODERN_VERSION: + continue + for profile in ( + REVIEW_PROFILE, + REPEATED_CURSOR_PROFILE, + MISMATCHED_DISCOVERY_ID_PROFILE, + NULL_DISCOVERY_ID_PROFILE, + SSE_CR_COMMENTS_PROFILE, + SSE_COMMENT_FLOOD_PROFILE, + ): + cases.append( + _run_review_case( + codex_binary, + server_script, + mode=mode, + profile=profile, + transport="http", + case_home=root / f"review-http-{mode}-{profile}", + timeout_seconds=timeout_seconds, + ) + ) + + for mode in modes: + for transport in CATALOG_BOUNDARY_TRANSPORTS: + for profile in CATALOG_BOUNDARY_PROFILES: + cases.append( + _run_review_case( + codex_binary, + server_script, + mode=mode, + profile=profile, + transport=transport, + case_home=root / f"review-{transport}-{mode}-{profile}", + timeout_seconds=timeout_seconds, + ) + ) + + checks = [check for case in cases for check in case.checks] + failures = [check for check in checks if not check.success] + return { + "schemaVersion": REVIEW_REPORT_SCHEMA_VERSION, + "success": bool(cases) and all(case.success for case in cases), + "codexBinary": str(codex_binary), + "reviewer": REVIEWER, + "modes": list(modes), + "summary": { + "passed": len(checks) - len(failures), + "failed": len(failures), + "total": len(checks), + "casesPassed": sum(case.success for case in cases), + "casesTotal": len(cases), + }, + "cases": [asdict(case) for case in cases], + "artifacts": str(root) if keep_artifacts else None, + } + finally: + if not keep_artifacts: + shutil.rmtree(root, ignore_errors=True) + + +def _required_review_checks( + report: Mapping[str, object], + *, + label: str, + errors: list[str], +) -> dict[_ReviewCheckIdentity, bool]: + if report.get("schemaVersion") != REVIEW_REPORT_SCHEMA_VERSION: + errors.append( + f"{label} does not use reviewer report schema " + f"{REVIEW_REPORT_SCHEMA_VERSION}" + ) + if report.get("reviewer") != REVIEWER: + errors.append(f"{label} does not identify the production reviewer probes") + + compact = report.get("baselineKind") is not None + modes = report.get("requiredModes" if compact else "modes") + if modes != list(REVIEW_MODES): + errors.append(f"{label} does not run all three required MCP protocol modes") + + expected_cases = _required_review_cases() + observed_cases: set[tuple[str, str]] = set() + identities: dict[_ReviewCheckIdentity, bool] = {} + + if compact: + if report.get("baselineKind") != REVIEW_BASELINE_KIND: + errors.append(f"{label} uses an unsupported reviewer baseline format") + + raw_cases = report.get("requiredCases") + if not isinstance(raw_cases, list): + errors.append(f"{label} does not contain required reviewer cases") + raw_cases = [] + for case in raw_cases: + if not isinstance(case, dict): + errors.append(f"{label} contains a malformed reviewer case") + continue + mode = case.get("mode") + transport = case.get("transport") + if not isinstance(mode, str) or not isinstance(transport, str): + errors.append(f"{label} contains a malformed reviewer case") + continue + identity = (mode, transport) + if identity in observed_cases: + errors.append(f"{label} contains a duplicate reviewer case") + continue + observed_cases.add(identity) + + raw_checks = report.get("checks") + if not isinstance(raw_checks, dict): + errors.append(f"{label} does not contain reviewer check identities") + raw_checks = {} + for bucket, success in (("passing", True), ("failing", False)): + records = raw_checks.get(bucket) + if not isinstance(records, list): + errors.append(f"{label} does not contain {bucket} reviewer checks") + continue + for record in records: + if not isinstance(record, dict): + errors.append( + f"{label} contains a malformed {bucket} reviewer check" + ) + continue + mode = record.get("mode") + transport = record.get("transport") + check_id = record.get("check_id") + if ( + not isinstance(mode, str) + or not isinstance(transport, str) + or not isinstance(check_id, str) + or not check_id.startswith("review/") + or (mode, transport) not in expected_cases + ): + errors.append( + f"{label} contains a malformed {bucket} reviewer check" + ) + continue + identity = _ReviewCheckIdentity(mode, transport, check_id) + if identity in identities: + errors.append(f"{label} contains a duplicate reviewer check") + continue + identities[identity] = success + else: + raw_cases = report.get("cases") + if not isinstance(raw_cases, list): + errors.append(f"{label} does not contain production reviewer cases") + raw_cases = [] + for case in raw_cases: + if not isinstance(case, dict): + errors.append(f"{label} contains a malformed reviewer case") + continue + mode = case.get("mode") + transport = case.get("transport") + if ( + not isinstance(mode, str) + or not isinstance(transport, str) + or (mode, transport) not in expected_cases + ): + errors.append(f"{label} contains an unexpected reviewer case") + continue + case_identity = (mode, transport) + if case_identity in observed_cases: + errors.append(f"{label} contains a duplicate reviewer case") + continue + observed_cases.add(case_identity) + + checks = case.get("checks") + if not isinstance(checks, list) or not checks: + errors.append( + f"{label} has no reviewer checks for {transport} / {mode}" + ) + continue + case_success = True + for check in checks: + if not isinstance(check, dict): + errors.append(f"{label} contains a malformed reviewer check") + continue + check_id = check.get("name") + success = check.get("success") + if ( + not isinstance(check_id, str) + or not check_id.startswith("review/") + or not isinstance(success, bool) + ): + errors.append(f"{label} contains a malformed reviewer check") + continue + identity = _ReviewCheckIdentity(mode, transport, check_id) + if identity in identities: + errors.append(f"{label} contains a duplicate reviewer check") + continue + identities[identity] = success + case_success = case_success and success + if case.get("success") is not case_success: + errors.append( + f"{label} reports inconsistent reviewer case success " + f"for {transport} / {mode}" + ) + + missing_cases = sorted(expected_cases - observed_cases) + unexpected_cases = sorted(observed_cases - expected_cases) + for mode, transport in missing_cases: + errors.append(f"{label} is missing required reviewer case {transport} / {mode}") + for mode, transport in unexpected_cases: + errors.append(f"{label} contains unexpected reviewer case {transport} / {mode}") + + summary = report.get("summary") + passing = sum(identities.values()) + failed = len(identities) - passing + cases_passed = sum( + all( + success + for identity, success in identities.items() + if (identity.mode, identity.transport) == case + ) + and any((identity.mode, identity.transport) == case for identity in identities) + for case in observed_cases + ) + expected_summary = { + "passed": passing, + "failed": failed, + "total": len(identities), + "casesPassed": cases_passed, + "casesTotal": len(observed_cases), + } + if not isinstance(summary, dict) or any( + summary.get(name) != value for name, value in expected_summary.items() + ): + errors.append(f"{label} reports inconsistent reviewer check totals") + if not compact and report.get("success") is not ( + bool(observed_cases) and failed == 0 + ): + errors.append(f"{label} reports inconsistent overall reviewer success") + return identities + + +def _compact_review_regression_baseline( + report: Mapping[str, object], +) -> dict[str, object]: + errors: list[str] = [] + checks = _required_review_checks(report, label="baseline", errors=errors) + if errors: + raise ValueError("; ".join(errors)) + + summary = report.get("summary") + assert isinstance(summary, dict) + return { + "baselineKind": REVIEW_BASELINE_KIND, + "schemaVersion": REVIEW_REPORT_SCHEMA_VERSION, + "reviewer": REVIEWER, + "requiredModes": list(REVIEW_MODES), + "requiredCases": [ + {"mode": mode, "transport": transport} + for mode, transport in sorted(_required_review_cases()) + ], + "summary": { + key: summary[key] + for key in ("passed", "failed", "total", "casesPassed", "casesTotal") + }, + "checks": { + "passing": [ + asdict(identity) + for identity, success in sorted(checks.items()) + if success + ], + "failing": [ + asdict(identity) + for identity, success in sorted(checks.items()) + if not success + ], + }, + } + + +def _evaluate_review_regression_gate( + report: Mapping[str, object], + baseline_report: Mapping[str, object], + *, + baseline_path: Path | None = None, +) -> dict[str, object]: + errors: list[str] = [] + baseline_checks = _required_review_checks( + baseline_report, label="baseline", errors=errors + ) + candidate_checks = _required_review_checks(report, label="candidate", errors=errors) + known_failures: list[dict[str, object]] = [] + new_failures: list[dict[str, object]] = [] + missing_checks: list[dict[str, object]] = [] + fixed_checks: list[dict[str, object]] = [] + + for identity, success in sorted(candidate_checks.items()): + if success: + continue + if baseline_checks.get(identity) is False: + known_failures.append(asdict(identity)) + else: + new_failures.append(asdict(identity)) + + for identity, baseline_success in sorted(baseline_checks.items()): + candidate_success = candidate_checks.get(identity) + if candidate_success is None: + missing_checks.append(asdict(identity)) + elif not baseline_success and candidate_success: + fixed_checks.append(asdict(identity)) + + return { + "success": not errors and not new_failures and not missing_checks, + "requiredModes": list(REVIEW_MODES), + "baselineReport": str(baseline_path) if baseline_path is not None else None, + "configurationErrors": errors, + "knownFailures": known_failures, + "newFailures": new_failures, + "missingChecks": missing_checks, + "fixedChecks": fixed_checks, + } + + +def _write_review_json(path: Path, value: Mapping[str, object]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text( + json.dumps(value, ensure_ascii=False, indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + + +def _parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: + parser = argparse.ArgumentParser( + description="Run reviewer-derived MCP client regression tests against Codex." + ) + parser.add_argument("codex_binary", type=Path) + parser.add_argument("--mode", choices=("all", *REVIEW_MODES), default="all") + parser.add_argument("--timeout", type=float, default=8) + parser.add_argument("--server-script", type=Path, default=_MODULE_DIR / "server.py") + parser.add_argument("--report", type=Path) + parser.add_argument( + "--baseline-report", + type=Path, + help="Compare every production reviewer check with a reviewed baseline.", + ) + parser.add_argument( + "--write-baseline", + type=Path, + help="Write a compact reviewer baseline from the completed full matrix.", + ) + parser.add_argument( + "--extract-baseline", + type=Path, + help="Extract a compact reviewer baseline from --baseline-report and exit.", + ) + parser.add_argument("--artifact-parent", type=Path) + parser.add_argument("--keep-artifacts", action="store_true") + parser.add_argument("--json", action="store_true") + return parser.parse_args(argv) + + +def _print_report(report: Mapping[str, object]) -> None: + print( + "Codex MCP reviewer regressions: " + + ("PASS" if report.get("success") is True else "FAIL") + ) + print(f"Binary: {report.get('codexBinary')}") + cases = report.get("cases") + if isinstance(cases, list): + for case in cases: + if not isinstance(case, dict): + continue + print(f"\n{case.get('transport')} / {case.get('mode')}") + for check in case.get("checks", []): + if isinstance(check, dict): + status = "PASS" if check.get("success") else "FAIL" + print(f" {status} {check.get('name')}: {check.get('detail')}") + summary = report.get("summary") + if isinstance(summary, dict): + print( + f"\nSummary: {summary.get('passed')}/{summary.get('total')} checks; " + f"{summary.get('casesPassed')}/{summary.get('casesTotal')} cases" + ) + gate = report.get("regressionGate") + if isinstance(gate, dict): + status = "PASS" if gate.get("success") is True else "FAIL" + known = gate.get("knownFailures") + new = gate.get("newFailures") + missing = gate.get("missingChecks") + fixed = gate.get("fixedChecks") + print( + f"Regression gate: {status}; " + f"{len(known) if isinstance(known, list) else 0} known, " + f"{len(new) if isinstance(new, list) else 0} new, " + f"{len(missing) if isinstance(missing, list) else 0} missing, " + f"{len(fixed) if isinstance(fixed, list) else 0} fixed" + ) + errors = gate.get("configurationErrors") + if isinstance(errors, list): + for error in errors: + print(f" Configuration error: {error}") + + +def main(argv: Sequence[str] | None = None) -> int: + args = _parse_args(argv) + codex_binary = args.codex_binary.expanduser().resolve() + server_script = args.server_script.expanduser().resolve() + if not codex_binary.is_file() or not os.access(codex_binary, os.X_OK): + print(f"error: Codex binary is not executable: {codex_binary}", file=sys.stderr) + return 2 + if not server_script.is_file(): + print(f"error: review fixture does not exist: {server_script}", file=sys.stderr) + return 2 + if args.timeout <= 0: + print("error: --timeout must be positive", file=sys.stderr) + return 2 + if args.extract_baseline is not None and args.baseline_report is None: + print("error: --extract-baseline requires --baseline-report", file=sys.stderr) + return 2 + if args.extract_baseline is not None and args.write_baseline is not None: + print( + "error: --extract-baseline cannot be combined with --write-baseline", + file=sys.stderr, + ) + return 2 + + baseline_report: dict[str, object] | None = None + baseline_path: Path | None = None + if args.baseline_report is not None: + baseline_path = args.baseline_report.expanduser().resolve() + try: + loaded_baseline = json.loads(baseline_path.read_text(encoding="utf-8")) + except (OSError, UnicodeError, json.JSONDecodeError) as exc: + print( + f"error: cannot read reviewer baseline {baseline_path}: {exc}", + file=sys.stderr, + ) + return 2 + if not isinstance(loaded_baseline, dict): + print( + f"error: reviewer baseline must be a JSON object: {baseline_path}", + file=sys.stderr, + ) + return 2 + baseline_report = loaded_baseline + + if args.extract_baseline is not None: + assert baseline_report is not None + output_path = args.extract_baseline.expanduser().resolve() + try: + _write_review_json( + output_path, _compact_review_regression_baseline(baseline_report) + ) + except (OSError, ValueError) as exc: + print(f"error: cannot extract reviewer baseline: {exc}", file=sys.stderr) + return 2 + print(f"Reviewer regression baseline: {output_path}") + return 0 + + parent = None + if args.artifact_parent is not None: + parent = args.artifact_parent.expanduser().resolve() + parent.mkdir(parents=True, exist_ok=True) + + modes = REVIEW_MODES if args.mode == "all" else (args.mode,) + report = run_review_regressions( + codex_binary, + modes=modes, + server_script=server_script, + timeout_seconds=args.timeout, + artifact_parent=parent, + keep_artifacts=args.keep_artifacts, + ) + if baseline_report is not None: + report["regressionGate"] = _evaluate_review_regression_gate( + report, + baseline_report, + baseline_path=baseline_path, + ) + if args.write_baseline is not None: + output_path = args.write_baseline.expanduser().resolve() + try: + _write_review_json(output_path, _compact_review_regression_baseline(report)) + except (OSError, ValueError) as exc: + print(f"error: cannot write reviewer baseline: {exc}", file=sys.stderr) + return 2 + if args.report is not None: + path = args.report.expanduser().resolve() + _write_review_json(path, report) + if args.json: + print(json.dumps(report, ensure_ascii=False, indent=2, sort_keys=True)) + else: + _print_report(report) + if args.report is not None: + print(f"JSON report: {args.report.expanduser().resolve()}") + if args.write_baseline is not None: + print(f"Reviewer regression baseline: {args.write_baseline.resolve()}") + if baseline_report is not None: + gate = report.get("regressionGate") + return 0 if isinstance(gate, dict) and gate.get("success") is True else 1 + return 0 if report.get("success") is True else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/mcp_conformance/run_codex_compliance.py b/scripts/mcp_conformance/run_codex_compliance.py new file mode 100644 index 0000000000..bfb31cc6c5 --- /dev/null +++ b/scripts/mcp_conformance/run_codex_compliance.py @@ -0,0 +1,2201 @@ +#!/usr/bin/env python3 +"""Black-box MCP compliance runner for a supplied Codex binary.""" + +import argparse +import importlib.util +import json +import os +import queue +import shutil +import subprocess +import sys +import tempfile +import threading +import time +from collections import deque +from contextlib import contextmanager, nullcontext +from dataclasses import asdict, dataclass, field +from datetime import datetime, timezone +from pathlib import Path +from typing import Callable, Iterator, Mapping, Sequence + +_MODULE_DIR = Path(__file__).resolve().parent +if str(_MODULE_DIR) not in sys.path: + # Some execution environments set PYTHONSAFEPATH. Add only this + # trusted package directory so direct `python path/to/script.py` usage + # continues to work alongside the Bazel entry point. + sys.path.insert(0, str(_MODULE_DIR)) + +from official_conformance import ( # noqa: E402 - direct scripts must first add their sibling directory. + OFFICIAL_CONFORMANCE_GIT_REF, + OFFICIAL_CONFORMANCE_REPOSITORY, + OfficialScenarioResult, + default_conformance_command, + run_official_mode, + scenarios_for_mode, +) + +_FIXTURE_MODULE_PATH = Path(__file__).resolve().with_name("server.py") +_FIXTURE_SPEC = importlib.util.spec_from_file_location( + "_mcp_spec_test_fixture_server", + _FIXTURE_MODULE_PATH, +) +if _FIXTURE_SPEC is None or _FIXTURE_SPEC.loader is None: + raise RuntimeError(f"could not load MCP fixture module: {_FIXTURE_MODULE_PATH}") +_FIXTURE_MODULE = importlib.util.module_from_spec(_FIXTURE_SPEC) +sys.modules[_FIXTURE_SPEC.name] = _FIXTURE_MODULE +_FIXTURE_SPEC.loader.exec_module(_FIXTURE_MODULE) + +LEGACY_VERSION = _FIXTURE_MODULE.LEGACY_VERSION +SHIPPING_LEGACY_VERSION = _FIXTURE_MODULE.SHIPPING_LEGACY_VERSION +MODERN_VERSION = _FIXTURE_MODULE.MODERN_VERSION +MISMATCHED_DISCOVERY_ID_PROFILE = _FIXTURE_MODULE.MISMATCHED_DISCOVERY_ID_PROFILE +NULL_DISCOVERY_ID_PROFILE = _FIXTURE_MODULE.NULL_DISCOVERY_ID_PROFILE +REPEATED_CURSOR_PROFILE = _FIXTURE_MODULE.REPEATED_CURSOR_PROFILE +REVIEW_EXACT_INTEGER = _FIXTURE_MODULE.REVIEW_EXACT_INTEGER +REVIEW_MRTR_INPUT_REQUEST_COUNT = _FIXTURE_MODULE.REVIEW_MRTR_INPUT_REQUEST_COUNT +REVIEW_PROFILE = _FIXTURE_MODULE.REVIEW_PROFILE +RESOURCE_URIS = _FIXTURE_MODULE.RESOURCE_URIS +SERVER_NAME = _FIXTURE_MODULE.SERVER_NAME +SERVER_VERSION = _FIXTURE_MODULE.SERVER_VERSION +SSE_COMMENT_FLOOD_PROFILE = _FIXTURE_MODULE.SSE_COMMENT_FLOOD_PROFILE +SSE_CR_COMMENTS_PROFILE = _FIXTURE_MODULE.SSE_CR_COMMENTS_PROFILE +ProtocolServer = _FIXTURE_MODULE.ProtocolServer +make_http_server = _FIXTURE_MODULE.make_http_server + +REPORT_SCHEMA_VERSION = 4 +COMPACT_REGRESSION_BASELINE_KIND = "mcp-conformance-regression-baseline-v1" +REQUIRED_REGRESSION_MODES = ( + SHIPPING_LEGACY_VERSION, + LEGACY_VERSION, + MODERN_VERSION, +) +TEST_SERVER_NAME = "mcp_spec_fixture" +TEST_META_KEY = "com.openai/mcp-spec-test" +CHECK_ORDER = ( + "mcp_add", + "mcp_get", + "app_server_initialize", + "modern_feature_enablement", + "inventory", + "ephemeral_thread", + "echo_tool", + "unicode_resource_read", + "per_request_metadata", + "request_scoped_notifications", + "multi_round_trip_request", + "http_header_mirroring", + "app_server_protocol", + "case_runtime", + "mcp_remove", + "isolated_config_cleanup", +) + + +@dataclass +class CommandResult: + returncode: int + stdout: str + stderr: str + + +@dataclass +class CheckResult: + name: str + success: bool + detail: str + status: str | None = None + source: str = "supplemental" + scenario: str | None = None + check_id: str | None = None + category: str | None = None + + def __post_init__(self) -> None: + if self.status is None: + self.status = "PASS" if self.success else "FAIL" + + +@dataclass +class CaseResult: + transport: str + mode: str + success: bool = False + duration_seconds: float = 0.0 + checks: list[CheckResult] = field(default_factory=list) + diagnostics: str | None = None + + def check(self, name: str, success: bool, detail: str) -> bool: + self.checks.append( + CheckResult( + name=name, + success=success, + detail=detail, + status="PASS" if success else "FAIL", + ) + ) + return success + + def finish(self, started_at: float) -> None: + self.duration_seconds = round(time.monotonic() - started_at, 3) + self.success = bool(self.checks) and all(check.success for check in self.checks) + + +@dataclass(frozen=True, order=True) +class _RegressionCheckIdentity: + mode: str + transport: str + source: str + scenario: str + check_id: str + + +class AppServerError(RuntimeError): + pass + + +class AppServerClient: + """Minimal JSONL client for the Codex app-server protocol.""" + + def __init__( + self, + codex_binary: Path, + *, + env: Mapping[str, str], + cwd: Path, + timeout_seconds: float, + elicitation_content: ( + Callable[[Mapping[str, object]], Mapping[str, object]] | None + ) = None, + ) -> None: + self.timeout_seconds = timeout_seconds + self._elicitation_content = elicitation_content + self._next_id = 1 + self._messages: queue.Queue[dict[str, object] | None] = queue.Queue() + self._write_lock = threading.Lock() + self._stderr: deque[str] = deque(maxlen=200) + self._parse_errors: deque[str] = deque(maxlen=20) + self.events: list[dict[str, object]] = [] + self.elicitation_requests: list[dict[str, object]] = [] + self.process = subprocess.Popen( + [str(codex_binary), "app-server"], + cwd=cwd, + env=dict(env), + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + bufsize=1, + ) + self._stdout_thread = threading.Thread( + target=self._read_stdout, + name="codex-app-server-stdout", + daemon=True, + ) + self._stderr_thread = threading.Thread( + target=self._read_stderr, + name="codex-app-server-stderr", + daemon=True, + ) + self._stdout_thread.start() + self._stderr_thread.start() + + def __enter__(self) -> "AppServerClient": + return self + + def __exit__(self, *_: object) -> None: + self.close() + + def _read_stdout(self) -> None: + assert self.process.stdout is not None + for line in self.process.stdout: + stripped = line.strip() + if not stripped: + continue + try: + message = json.loads(stripped) + except json.JSONDecodeError: + self._parse_errors.append(stripped[:500]) + continue + if isinstance(message, dict): + self._messages.put(message) + self._messages.put(None) + + def _read_stderr(self) -> None: + assert self.process.stderr is not None + for line in self.process.stderr: + self._stderr.append(line.rstrip()) + + def _send(self, message: Mapping[str, object]) -> None: + if self.process.poll() is not None: + raise AppServerError( + f"Codex app-server exited with code {self.process.returncode}" + ) + assert self.process.stdin is not None + encoded = json.dumps(message, ensure_ascii=False, separators=(",", ":")) + with self._write_lock: + self.process.stdin.write(encoded + "\n") + self.process.stdin.flush() + + def notify(self, method: str) -> None: + self._send({"method": method}) + + def request( + self, + method: str, + params: Mapping[str, object] | None, + ) -> dict[str, object]: + request_id = self._next_id + self._next_id += 1 + message: dict[str, object] = {"id": request_id, "method": method} + if params is not None: + message["params"] = dict(params) + self._send(message) + + deadline = time.monotonic() + self.timeout_seconds + while True: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise AppServerError( + f"timed out waiting for app-server response to {method}" + ) + try: + message = self._messages.get(timeout=remaining) + except queue.Empty as exc: + raise AppServerError( + f"timed out waiting for app-server response to {method}" + ) from exc + if message is None: + raise AppServerError( + f"Codex app-server closed stdout while handling {method}" + ) + if message.get("id") == request_id and "method" not in message: + return message + if "id" in message and isinstance(message.get("method"), str): + self._handle_server_request(message) + else: + self.events.append(message) + + def wait_for_notification( + self, + method: str, + *, + predicate: Callable[[Mapping[str, object]], bool] | None = None, + after_event_index: int = 0, + ) -> dict[str, object]: + for event in self.events[after_event_index:]: + if event.get("method") != method: + continue + params = event.get("params") + if isinstance(params, dict) and (predicate is None or predicate(params)): + return event + + deadline = time.monotonic() + self.timeout_seconds + while True: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise AppServerError( + f"timed out waiting for app-server notification {method}" + ) + try: + message = self._messages.get(timeout=remaining) + except queue.Empty as exc: + raise AppServerError( + f"timed out waiting for app-server notification {method}" + ) from exc + if message is None: + raise AppServerError( + f"Codex app-server closed stdout while waiting for {method}" + ) + if "id" in message and isinstance(message.get("method"), str): + self._handle_server_request(message) + continue + self.events.append(message) + if message.get("method") != method: + continue + params = message.get("params") + if isinstance(params, dict) and (predicate is None or predicate(params)): + return message + + def _handle_server_request(self, message: dict[str, object]) -> None: + self.events.append(message) + request_id = message.get("id") + method = message.get("method") + if method == "mcpServer/elicitation/request": + params = message.get("params") + if isinstance(params, dict): + self.elicitation_requests.append(params) + else: + params = {} + content = ( + dict(self._elicitation_content(params)) + if self._elicitation_content is not None + else {"confirmation": "confirmed"} + ) + self._send( + { + "id": request_id, + "result": { + "action": "accept", + "content": content, + "_meta": None, + }, + } + ) + return + self._send( + { + "id": request_id, + "error": { + "code": -32601, + "message": f"unsupported test client request: {method}", + }, + } + ) + + def diagnostic_text(self) -> str: + parts: list[str] = [] + if self._parse_errors: + parts.append("non-JSON stdout: " + " | ".join(self._parse_errors)) + startup_failures = [ + event + for event in self.events + if event.get("method") == "mcpServer/startupStatus/updated" + and isinstance(event.get("params"), dict) + and event["params"].get("status") not in ("starting", "ready") + ] + if startup_failures: + parts.append( + "startup events: " + + json.dumps(startup_failures[-3:], ensure_ascii=False) + ) + if self._stderr: + parts.append("stderr:\n" + "\n".join(self._stderr)) + return "\n".join(parts)[-8_000:] + + def close(self) -> None: + if self.process.poll() is None: + self.process.terminate() + try: + self.process.wait(timeout=3) + except subprocess.TimeoutExpired: + self.process.kill() + self.process.wait(timeout=3) + for stream in (self.process.stdin, self.process.stdout, self.process.stderr): + if stream is not None: + stream.close() + self._stdout_thread.join(timeout=1) + self._stderr_thread.join(timeout=1) + + +def _run_command( + command: Sequence[str], + *, + env: Mapping[str, str], + cwd: Path, + timeout_seconds: float, +) -> CommandResult: + try: + completed = subprocess.run( + list(command), + cwd=cwd, + env=dict(env), + text=True, + capture_output=True, + timeout=timeout_seconds, + check=False, + ) + return CommandResult( + completed.returncode, + completed.stdout.strip(), + completed.stderr.strip(), + ) + except subprocess.TimeoutExpired as exc: + stdout = ( + exc.stdout.decode() if isinstance(exc.stdout, bytes) else exc.stdout or "" + ) + stderr = ( + exc.stderr.decode() if isinstance(exc.stderr, bytes) else exc.stderr or "" + ) + return CommandResult( + 124, + stdout.strip(), + (stderr + f"\ncommand timed out after {timeout_seconds:g}s").strip(), + ) + + +def _response_result( + response: Mapping[str, object], +) -> tuple[dict[str, object] | None, str]: + error = response.get("error") + if isinstance(error, dict): + return None, json.dumps(error, ensure_ascii=False, sort_keys=True) + result = response.get("result") + if not isinstance(result, dict): + return None, "response did not contain an object result" + return result, "ok" + + +def _command_detail(result: CommandResult) -> str: + detail = result.stdout or result.stderr or f"exit code {result.returncode}" + return detail[-2_000:] + + +def _result_detail( + result: Mapping[str, object] | None, + fallback: str, +) -> str: + if result is None: + return fallback + return ( + "unexpected result: " + + json.dumps(result, ensure_ascii=False, sort_keys=True, default=str) + )[-2_000:] + + +def _isolated_environment(codex_home: Path) -> dict[str, str]: + env = dict(os.environ) + env["CODEX_HOME"] = str(codex_home) + # Direct app-server MCP calls do not need a model or credentials. + for name in ("CODEX_API_KEY", "CODEX_ACCESS_TOKEN", "OPENAI_API_KEY"): + env.pop(name, None) + return env + + +@contextmanager +def _running_http_fixture(mode: str) -> Iterator[str]: + httpd = make_http_server( + ProtocolServer(mode), + "127.0.0.1", + 0, + log_requests=False, + ) + thread = threading.Thread( + target=httpd.serve_forever, + name=f"mcp-fixture-{mode}", + daemon=True, + ) + thread.start() + try: + _, port = httpd.server_address + yield f"http://127.0.0.1:{port}/mcp" + finally: + httpd.shutdown() + httpd.server_close() + thread.join(timeout=3) + + +def _registration_command( + codex_binary: Path, + server_script: Path, + *, + transport: str, + mode: str, + http_url: str | None, +) -> list[str]: + command = [str(codex_binary), "mcp", "add", TEST_SERVER_NAME] + if transport == "stdio": + if mode == MODERN_VERSION: + command.extend(["--env", f"CODEX_MCP_PROTOCOL_VERSION={MODERN_VERSION}"]) + return [ + *command, + "--", + sys.executable, + str(server_script), + "--mode", + mode, + "--transport", + "stdio", + ] + assert http_url is not None + return [*command, "--url", http_url] + + +def _validate_registration( + config: Mapping[str, object], + *, + transport: str, + mode: str, + http_url: str | None, +) -> tuple[bool, str]: + if config.get("name") != TEST_SERVER_NAME or config.get("enabled") is not True: + return False, "registered server name or enabled state was incorrect" + value = config.get("transport") + if not isinstance(value, dict): + return False, "registered server did not contain a transport object" + if transport == "stdio": + args = value.get("args") + if value.get("type") != "stdio" or not isinstance(args, list): + return False, f"unexpected stdio transport: {value!r}" + if mode not in args or "stdio" not in args: + return False, f"stdio registration omitted mode or transport: {args!r}" + env = value.get("env") + modern_opt_in = ( + isinstance(env, dict) + and env.get("CODEX_MCP_PROTOCOL_VERSION") == MODERN_VERSION + ) + if mode == MODERN_VERSION and not modern_opt_in: + return False, "modern stdio registration omitted its protocol opt-in" + if mode in (SHIPPING_LEGACY_VERSION, LEGACY_VERSION) and modern_opt_in: + return ( + False, + "legacy stdio registration unexpectedly enabled the modern protocol", + ) + elif value.get("type") != "streamable_http" or value.get("url") != http_url: + return False, f"unexpected HTTP transport: {value!r}" + return True, f"registered {value.get('type')} transport" + + +def _validate_inventory( + result: Mapping[str, object], + *, + mode: str, +) -> tuple[bool, str]: + entries = result.get("data") + if not isinstance(entries, list): + return False, "mcpServerStatus/list result did not contain data" + entry = next( + ( + item + for item in entries + if isinstance(item, dict) and item.get("name") == TEST_SERVER_NAME + ), + None, + ) + if not isinstance(entry, dict): + return False, f"{TEST_SERVER_NAME!r} was absent from MCP status" + + missing: list[str] = [] + server_info = entry.get("serverInfo") + if ( + not isinstance(server_info, dict) + or server_info.get("name") != SERVER_NAME + or server_info.get("version") != SERVER_VERSION + ): + missing.append("server identity") + tools = entry.get("tools") + expected_tools = {"echo", "client_metadata", "progress"} + if mode == MODERN_VERSION: + expected_tools.add("request_input") + if not isinstance(tools, dict): + missing.append("tool inventory") + else: + absent_tools = sorted(expected_tools - set(tools)) + if absent_tools: + missing.append("tools " + ", ".join(absent_tools)) + resources = entry.get("resources") + resource_uris = ( + { + item.get("uri") + for item in resources + if isinstance(item, dict) and isinstance(item.get("uri"), str) + } + if isinstance(resources, list) + else set() + ) + absent_resources = sorted(set(RESOURCE_URIS) - resource_uris) + if absent_resources: + missing.append("paginated resources " + ", ".join(absent_resources)) + if missing: + return False, "missing " + "; ".join(missing) + return True, "identity, tools, and both paginated resources discovered" + + +def _validate_echo( + result: Mapping[str, object], + sentinel: str, +) -> tuple[bool, str]: + structured = result.get("structuredContent") + if isinstance(structured, dict) and structured.get("text") == sentinel: + return True, "echo returned the exact sentinel" + return False, f"unexpected echo result: {dict(result)!r}" + + +def _ratio(passed: int, total: int) -> dict[str, int | float]: + percentage = round((passed / total) * 100, 1) if total else 0.0 + return { + "passed": passed, + "total": total, + "percentage": percentage, + } + + +def _check_ratio(checks: Sequence[CheckResult]) -> dict[str, int | float]: + applicable = [check for check in checks if check.status != "SKIP"] + return _ratio( + sum(check.success for check in applicable), + len(applicable), + ) + + +def _scenario_ratio( + checks: Sequence[CheckResult], + *, + category: str, +) -> dict[str, int | float]: + by_scenario: dict[str, list[CheckResult]] = {} + for check in checks: + if ( + check.scenario is None + or check.category != category + or check.source not in {"official", "harness"} + ): + continue + by_scenario.setdefault(check.scenario, []).append(check) + return _ratio( + sum( + all(check.success for check in scenario_checks) + for scenario_checks in by_scenario.values() + ), + len(by_scenario), + ) + + +def _summarize_modes(cases: Sequence[CaseResult]) -> list[dict[str, object]]: + summaries: list[dict[str, object]] = [] + modes = dict.fromkeys(case.mode for case in cases) + for mode in modes: + mode_cases = [case for case in cases if case.mode == mode] + checks = [check for case in mode_cases for check in case.checks] + official_checks = [check for check in checks if check.source == "official"] + official_non_auth_checks = [ + check for check in official_checks if check.category == "non-auth" + ] + official_auth_checks = [ + check for check in official_checks if check.category == "auth" + ] + harness_checks = [check for check in checks if check.source == "harness"] + supplemental_checks = [ + check for check in checks if check.source == "supplemental" + ] + summaries.append( + { + "mode": mode, + "checks": _check_ratio(checks), + "officialChecks": _check_ratio(official_checks), + "officialNonAuthChecks": _check_ratio(official_non_auth_checks), + "officialAuthChecks": _check_ratio(official_auth_checks), + "officialNonAuthScenarios": _scenario_ratio( + checks, + category="non-auth", + ), + "officialAuthScenarios": _scenario_ratio( + checks, + category="auth", + ), + "harnessChecks": _check_ratio(harness_checks), + "supplementalChecks": _check_ratio(supplemental_checks), + "transportCases": _ratio( + sum(case.success for case in mode_cases), + len(mode_cases), + ), + } + ) + return summaries + + +def _build_test_matrix(cases: Sequence[CaseResult]) -> dict[str, object]: + columns = [ + { + "key": f"{case.transport}:{case.mode}", + "transport": case.transport, + "mode": case.mode, + } + for case in cases + ] + checks_by_case = { + f"{case.transport}:{case.mode}": {check.name: check for check in case.checks} + for case in cases + } + observed = {check.name for case in cases for check in case.checks} + ordered_names = [name for name in CHECK_ORDER if name in observed] + ordered_names.extend(sorted(observed - set(ordered_names))) + rows: list[dict[str, object]] = [] + for name in ordered_names: + results: dict[str, str] = {} + for column in columns: + key = str(column["key"]) + check = checks_by_case[key].get(name) + results[key] = "N/A" if check is None else check.status + rows.append({"test": name, "results": results}) + return {"columns": columns, "rows": rows} + + +def _call_tool( + client: AppServerClient, + *, + thread_id: str, + tool: str, + arguments: Mapping[str, object], + meta: Mapping[str, object] | None = None, +) -> tuple[dict[str, object] | None, str]: + params: dict[str, object] = { + "threadId": thread_id, + "server": TEST_SERVER_NAME, + "tool": tool, + "arguments": dict(arguments), + } + if meta is not None: + params["_meta"] = dict(meta) + return _response_result(client.request("mcpServer/tool/call", params)) + + +def _exercise_app_server( + case: CaseResult, + client: AppServerClient, + *, + transport: str, + mode: str, + workspace: Path, + enable_modern_feature: bool, +) -> None: + initialize, initialize_detail = _response_result( + client.request( + "initialize", + { + "clientInfo": { + "name": "mcp-spec-compliance-runner", + "title": "MCP spec compliance runner", + "version": "1.0.0", + }, + "capabilities": { + "experimentalApi": True, + "requestAttestation": False, + "mcpServerOpenaiFormElicitation": True, + }, + }, + ) + ) + if not case.check( + "app_server_initialize", + initialize is not None, + initialize_detail, + ): + return + client.notify("initialized") + + if mode == MODERN_VERSION and enable_modern_feature: + feature_result, feature_detail = _response_result( + client.request( + "experimentalFeature/enablement/set", + {"enablement": {"mcp_2026_07_28": True}}, + ) + ) + enabled = ( + feature_result is not None + and isinstance(feature_result.get("enablement"), dict) + and feature_result["enablement"].get("mcp_2026_07_28") is True + ) + if not case.check( + "modern_feature_enablement", + enabled, + ( + "runtime feature mcp_2026_07_28 enabled" + if enabled + else _result_detail(feature_result, feature_detail) + ), + ): + return + + inventory, inventory_detail = _response_result( + client.request("mcpServerStatus/list", {}) + ) + if inventory is None: + case.check("inventory", False, inventory_detail) + else: + valid, detail = _validate_inventory(inventory, mode=mode) + case.check("inventory", valid, detail) + + thread_result, thread_detail = _response_result( + client.request( + "thread/start", + {"cwd": str(workspace), "ephemeral": True}, + ) + ) + thread_id: str | None = None + if thread_result is not None: + thread_value = thread_result.get("thread") + if isinstance(thread_value, dict) and isinstance(thread_value.get("id"), str): + thread_id = str(thread_value["id"]) + if not case.check( + "ephemeral_thread", + thread_id is not None, + "ephemeral thread created" if thread_id is not None else thread_detail, + ): + return + assert thread_id is not None + + sentinel = f"codex-mcp-{transport}-{mode}" + echo, echo_detail = _call_tool( + client, + thread_id=thread_id, + tool="echo", + arguments={"text": sentinel}, + meta={TEST_META_KEY: sentinel}, + ) + if echo is None: + case.check("echo_tool", False, echo_detail) + else: + valid, detail = _validate_echo(echo, sentinel) + case.check("echo_tool", valid, detail) + + resource, resource_detail = _response_result( + client.request( + "mcpServer/resource/read", + { + "threadId": thread_id, + "server": TEST_SERVER_NAME, + "uri": RESOURCE_URIS[1], + }, + ) + ) + expected_resource_text = f"fixture contents for {RESOURCE_URIS[1]}" + contents = resource.get("contents") if resource is not None else None + resource_ok = ( + isinstance(contents, list) + and len(contents) == 1 + and isinstance(contents[0], dict) + and contents[0].get("text") == expected_resource_text + ) + case.check( + "unicode_resource_read", + resource_ok, + ( + "read the Unicode resource" + if resource_ok + else _result_detail(resource, resource_detail) + ), + ) + + if mode != MODERN_VERSION: + return + + metadata, metadata_detail = _call_tool( + client, + thread_id=thread_id, + tool="client_metadata", + arguments={}, + meta={TEST_META_KEY: sentinel}, + ) + structured = metadata.get("structuredContent") if metadata is not None else None + metadata_ok = ( + isinstance(structured, dict) + and structured.get("io.modelcontextprotocol/protocolVersion") == MODERN_VERSION + and isinstance( + structured.get("io.modelcontextprotocol/clientCapabilities"), + dict, + ) + and structured.get(TEST_META_KEY) == sentinel + ) + case.check( + "per_request_metadata", + metadata_ok, + ( + "required and caller metadata were preserved" + if metadata_ok + else _result_detail(metadata, metadata_detail) + ), + ) + + progress, progress_detail = _call_tool( + client, + thread_id=thread_id, + tool="progress", + arguments={}, + meta={ + "progressToken": "fixture-progress", + "io.modelcontextprotocol/logLevel": "info", + }, + ) + progress_content = progress.get("content") if progress is not None else None + progress_ok = isinstance(progress_content, list) and bool(progress_content) + case.check( + "request_scoped_notifications", + progress_ok, + ( + "progress and log notifications were consumed before the result" + if progress_ok + else _result_detail(progress, progress_detail) + ), + ) + + elicitations_before = len(client.elicitation_requests) + mrtr, mrtr_detail = _call_tool( + client, + thread_id=thread_id, + tool="request_input", + arguments={}, + ) + mrtr_structured = mrtr.get("structuredContent") if mrtr is not None else None + new_elicitations = client.elicitation_requests[elicitations_before:] + mrtr_ok = ( + isinstance(mrtr_structured, dict) + and mrtr_structured.get("confirmation") == "confirmed" + and any( + item.get("serverName") == TEST_SERVER_NAME and item.get("mode") == "form" + for item in new_elicitations + ) + ) + case.check( + "multi_round_trip_request", + mrtr_ok, + ( + "elicitation was surfaced, answered, retried, and completed" + if mrtr_ok + else _result_detail(mrtr, mrtr_detail) + ), + ) + + if transport == "http": + arguments = { + "region": "us-west1", + "attempt": 3, + "enabled": True, + "greeting": "Hello, 世界", + } + mirrored, mirrored_detail = _call_tool( + client, + thread_id=thread_id, + tool="header_echo", + arguments=arguments, + ) + mirrored_structured = ( + mirrored.get("structuredContent") if mirrored is not None else None + ) + mirrored_ok = mirrored_structured == arguments + case.check( + "http_header_mirroring", + mirrored_ok, + ( + "method, name, scalar, and Base64 headers matched" + if mirrored_ok + else _result_detail(mirrored, mirrored_detail) + ), + ) + + +def _run_case( + codex_binary: Path, + server_script: Path, + *, + transport: str, + mode: str, + case_home: Path, + timeout_seconds: float, + enable_modern_feature: bool, +) -> CaseResult: + started_at = time.monotonic() + case = CaseResult(transport=transport, mode=mode) + case_home.mkdir(parents=True) + workspace = case_home / "workspace" + workspace.mkdir() + env = _isolated_environment(case_home) + client: AppServerClient | None = None + registered = False + + http_context = ( + _running_http_fixture(mode) if transport == "http" else nullcontext(None) + ) + try: + with http_context as http_url: + add = _run_command( + _registration_command( + codex_binary, + server_script, + transport=transport, + mode=mode, + http_url=http_url, + ), + env=env, + cwd=workspace, + timeout_seconds=timeout_seconds, + ) + registered = case.check( + "mcp_add", + add.returncode == 0, + _command_detail(add), + ) + if registered: + get = _run_command( + [ + str(codex_binary), + "mcp", + "get", + TEST_SERVER_NAME, + "--json", + ], + env=env, + cwd=workspace, + timeout_seconds=timeout_seconds, + ) + config: dict[str, object] | None = None + if get.returncode == 0: + try: + decoded = json.loads(get.stdout) + if isinstance(decoded, dict): + config = decoded + except json.JSONDecodeError: + pass + if config is None: + case.check("mcp_get", False, _command_detail(get)) + else: + valid, detail = _validate_registration( + config, + transport=transport, + mode=mode, + http_url=http_url, + ) + case.check("mcp_get", valid, detail) + + client = AppServerClient( + codex_binary, + env=env, + cwd=workspace, + timeout_seconds=timeout_seconds, + ) + try: + _exercise_app_server( + case, + client, + transport=transport, + mode=mode, + workspace=workspace, + enable_modern_feature=enable_modern_feature, + ) + except (AppServerError, OSError) as exc: + case.check("app_server_protocol", False, str(exc)) + finally: + client.close() + diagnostic_text = client.diagnostic_text() + if diagnostic_text and any( + not check.success for check in case.checks + ): + case.diagnostics = diagnostic_text + except OSError as exc: + case.check("case_runtime", False, str(exc)) + finally: + if client is not None and client.process.poll() is None: + client.close() + if registered: + remove = _run_command( + [str(codex_binary), "mcp", "remove", TEST_SERVER_NAME], + env=env, + cwd=workspace, + timeout_seconds=timeout_seconds, + ) + case.check( + "mcp_remove", + remove.returncode == 0, + _command_detail(remove), + ) + listed = _run_command( + [str(codex_binary), "mcp", "list", "--json"], + env=env, + cwd=workspace, + timeout_seconds=timeout_seconds, + ) + empty = False + if listed.returncode == 0: + try: + empty = json.loads(listed.stdout) == [] + except json.JSONDecodeError: + pass + case.check( + "isolated_config_cleanup", + empty, + "no MCP servers remained" if empty else _command_detail(listed), + ) + case.finish(started_at) + return case + + +def _official_check_result( + result: OfficialScenarioResult, + *, + index: int, + duplicate: bool, +) -> CheckResult: + check = result.checks[index] + if check.status == "SUCCESS": + status = "PASS" + success = True + elif check.status in {"SKIPPED", "INFO"}: + status = "SKIP" + success = True + else: + status = "FAIL" + success = False + suffix = f"{check.name}#{index + 1}" if duplicate else check.name + name = f"official/{result.scenario}/{suffix}" + detail = check.error_message or check.description + return CheckResult( + name=name, + success=success, + detail=detail, + status=status, + source="official", + scenario=result.scenario, + check_id=check.check_id, + category="auth" if result.scenario.startswith("auth/") else "non-auth", + ) + + +def _run_official_case( + codex_binary: Path, + adapter_script: Path, + *, + conformance_command: Sequence[str], + mode: str, + scenarios: Sequence[str], + case_home: Path, + timeout_seconds: float, + enable_modern_feature: bool, + require_automatic_auth: bool = False, +) -> CaseResult: + started_at = time.monotonic() + case = CaseResult(transport="official-http", mode=mode) + case_home.mkdir(parents=True) + env = _isolated_environment(case_home) + env["CODEX_CONFORMANCE_TIMEOUT"] = str(timeout_seconds) + env["CODEX_CONFORMANCE_ENABLE_MODERN_FEATURE"] = ( + "1" if enable_modern_feature else "0" + ) + env["CODEX_CONFORMANCE_REQUIRE_AUTOMATIC_AUTH"] = ( + "1" if require_automatic_auth else "0" + ) + results = run_official_mode( + conformance_command=conformance_command, + adapter_script=adapter_script, + codex_binary=codex_binary, + mode=mode, + scenarios=scenarios, + output_dir=case_home / "official-results", + timeout_seconds=timeout_seconds, + base_env=env, + ) + + diagnostics: list[str] = [] + for result in results: + counts: dict[str, int] = {} + for check in result.checks: + counts[check.name] = counts.get(check.name, 0) + 1 + for index, check in enumerate(result.checks): + case.checks.append( + _official_check_result( + result, + index=index, + duplicate=counts[check.name] > 1, + ) + ) + + case.checks.append( + CheckResult( + name=f"harness/{result.scenario}/codex-adapter", + success=result.adapter_success, + detail=result.adapter_detail, + status="PASS" if result.adapter_success else "FAIL", + source="harness", + scenario=result.scenario, + category=( + "auth" if result.scenario.startswith("auth/") else "non-auth" + ), + ) + ) + if ( + result.adapter_success + and not result.success + and not any( + check.status in {"FAILURE", "WARNING"} for check in result.checks + ) + ): + case.checks.append( + CheckResult( + name=f"harness/{result.scenario}/official-runner", + success=False, + detail=result.runner_detail or "official runner failed", + status="FAIL", + source="harness", + scenario=result.scenario, + category=( + "auth" if result.scenario.startswith("auth/") else "non-auth" + ), + ) + ) + if not result.success and result.runner_detail: + diagnostics.append(f"{result.scenario}:\n{result.runner_detail}") + + if not results: + case.checks.append( + CheckResult( + name="harness/official-scenario-selection", + success=False, + detail=f"no official scenarios selected for {mode}", + status="FAIL", + source="harness", + ) + ) + if diagnostics: + case.diagnostics = "\n\n".join(diagnostics)[-16_000:] + case.finish(started_at) + return case + + +def run_compliance( + codex_binary: Path, + *, + server_script: Path, + adapter_script: Path, + conformance_command: Sequence[str], + official_scenarios: Sequence[str] | None, + modes: Sequence[str], + transports: Sequence[str], + timeout_seconds: float, + artifact_parent: Path | None, + keep_artifacts: bool, + enable_modern_feature: bool = True, + include_auth: bool = True, + require_automatic_auth: bool = False, +) -> tuple[dict[str, object], Path | None]: + started_at = datetime.now(timezone.utc) + run_root = Path( + tempfile.mkdtemp( + prefix="codex-mcp-compliance-", + dir=artifact_parent, + ) + ) + version_env = _isolated_environment(run_root) + version = _run_command( + [str(codex_binary), "--version"], + env=version_env, + cwd=run_root, + timeout_seconds=timeout_seconds, + ) + + cases: list[CaseResult] = [] + selected_scenarios: dict[str, tuple[str, ...]] = { + mode: scenarios_for_mode( + mode, + official_scenarios, + include_auth=include_auth, + ) + for mode in modes + } + + # Transport order is intentional: supplemental local stdio first, then the + # official suite's loopback HTTP server. + for transport in transports: + for mode in modes: + if transport == "stdio": + case_home = run_root / f"stdio-{mode}" + cases.append( + _run_case( + codex_binary, + server_script, + transport="stdio", + mode=mode, + case_home=case_home, + timeout_seconds=timeout_seconds, + enable_modern_feature=enable_modern_feature, + ) + ) + else: + if not selected_scenarios[mode]: + continue + case_home = run_root / f"official-http-{mode}" + cases.append( + _run_official_case( + codex_binary, + adapter_script, + conformance_command=conformance_command, + mode=mode, + scenarios=selected_scenarios[mode], + case_home=case_home, + timeout_seconds=timeout_seconds, + enable_modern_feature=enable_modern_feature, + require_automatic_auth=require_automatic_auth, + ) + ) + + passed = sum(case.success for case in cases) + finished_at = datetime.now(timezone.utc) + report: dict[str, object] = { + "schemaVersion": REPORT_SCHEMA_VERSION, + "success": version.returncode == 0 and passed == len(cases), + "startedAt": started_at.isoformat(), + "finishedAt": finished_at.isoformat(), + "codexBinary": str(codex_binary), + "codexVersion": version.stdout or None, + "modernFeatureEnablement": enable_modern_feature, + "automaticAuthRequired": require_automatic_auth, + "officialConformance": { + "repository": OFFICIAL_CONFORMANCE_REPOSITORY, + "gitRef": OFFICIAL_CONFORMANCE_GIT_REF, + "scope": ( + "all versioned client scenarios" + if include_auth + else "versioned non-auth client scenarios" + ), + "authenticationIncluded": include_auth, + "scenarios": selected_scenarios, + }, + "versionCheck": { + "success": version.returncode == 0, + "detail": _command_detail(version), + }, + "summary": { + "passed": passed, + "failed": len(cases) - passed, + "total": len(cases), + }, + "modeSummaries": _summarize_modes(cases), + "testMatrix": _build_test_matrix(cases), + "cases": [asdict(case) for case in cases], + "artifacts": str(run_root) if keep_artifacts else None, + } + + retained: Path | None = run_root if keep_artifacts else None + if not keep_artifacts: + shutil.rmtree(run_root, ignore_errors=True) + return report, retained + + +def _regression_check_identity( + mode: str, + transport: str, + check: Mapping[str, object], +) -> _RegressionCheckIdentity | None: + source = check.get("source") + scenario = check.get("scenario") + name = check.get("name") + check_id = check.get("check_id") + if source not in {"official", "harness", "supplemental"}: + return None + if not isinstance(name, str) or not name: + return None + if scenario is not None and (not isinstance(scenario, str) or not scenario): + return None + if check_id is not None and (not isinstance(check_id, str) or not check_id): + return None + + # The upstream runner numbers repeated request observations by arrival + # order. Neither that number nor the number of passing retries is a stable + # protocol assertion, so prefer the upstream check ID when it exists. + stable_name = name + stem, separator, suffix = name.rpartition("#") + if separator and stem and suffix.isdecimal(): + stable_name = stem + return _RegressionCheckIdentity( + mode=mode, + transport=transport, + source=source, + scenario=scenario if isinstance(scenario, str) else "", + check_id=check_id if isinstance(check_id, str) else stable_name, + ) + + +def _required_regression_checks( + report: Mapping[str, object], + *, + label: str, + errors: list[str], +) -> dict[_RegressionCheckIdentity, set[str]]: + if report.get("schemaVersion") != REPORT_SCHEMA_VERSION: + errors.append(f"{label} does not use report schema {REPORT_SCHEMA_VERSION}") + + version_check = report.get("versionCheck") + if not isinstance(version_check, dict) or version_check.get("success") is not True: + errors.append(f"{label} does not contain a successful Codex version check") + if report.get("modernFeatureEnablement") is not True: + errors.append(f"{label} did not enable the modern MCP feature") + + official = report.get("officialConformance") + selected: Mapping[str, object] = {} + if not isinstance(official, dict): + errors.append(f"{label} does not contain official conformance metadata") + else: + if official.get("repository") != OFFICIAL_CONFORMANCE_REPOSITORY: + errors.append(f"{label} uses a different upstream conformance repository") + if official.get("gitRef") != OFFICIAL_CONFORMANCE_GIT_REF: + errors.append( + f"{label} uses a different pinned upstream conformance revision" + ) + if official.get("authenticationIncluded") is not True: + errors.append(f"{label} does not include the required OAuth scenarios") + scenarios = official.get("scenarios") + if isinstance(scenarios, dict): + selected = scenarios + else: + errors.append(f"{label} does not contain a versioned scenario catalog") + + baseline_kind = report.get("baselineKind") + if baseline_kind is not None: + if baseline_kind != COMPACT_REGRESSION_BASELINE_KIND: + errors.append(f"{label} uses an unsupported compact baseline format") + return {} + if report.get("requiredModes") != list(REQUIRED_REGRESSION_MODES): + errors.append( + f"{label} does not require the shipping, intermediate, and modern MCP versions" + ) + transports = report.get("transports") + if ( + not isinstance(transports, list) + or len(transports) != 2 + or set(transports) != {"stdio", "official-http"} + ): + errors.append(f"{label} does not require both stdio and official HTTP") + + raw_identities = report.get("checks") + if not isinstance(raw_identities, dict): + errors.append(f"{label} does not contain compact baseline check identities") + return {} + + identities: dict[_RegressionCheckIdentity, set[str]] = {} + for bucket, status in (("passing", "PASS"), ("failing", "FAIL")): + records = raw_identities.get(bucket) + if not isinstance(records, list): + errors.append(f"{label} does not contain {bucket} baseline identities") + continue + seen: set[_RegressionCheckIdentity] = set() + for record in records: + if not isinstance(record, dict): + errors.append( + f"{label} contains a malformed {bucket} baseline identity" + ) + continue + mode = record.get("mode") + transport = record.get("transport") + source = record.get("source") + scenario = record.get("scenario") + check_id = record.get("check_id") + if ( + mode not in REQUIRED_REGRESSION_MODES + or transport not in {"stdio", "official-http"} + or source not in {"official", "harness", "supplemental"} + or not isinstance(scenario, str) + or not isinstance(check_id, str) + or not check_id + ): + errors.append( + f"{label} contains a malformed {bucket} baseline identity" + ) + continue + identity = _RegressionCheckIdentity( + mode, transport, source, scenario, check_id + ) + if identity in seen: + errors.append( + f"{label} contains a duplicate {bucket} baseline identity" + ) + continue + seen.add(identity) + identities.setdefault(identity, set()).add(status) + + for mode in REQUIRED_REGRESSION_MODES: + expected_scenarios = scenarios_for_mode(mode, include_auth=True) + actual_scenarios = selected.get(mode) + if ( + not isinstance(actual_scenarios, (list, tuple)) + or tuple(actual_scenarios) != expected_scenarios + ): + errors.append( + f"{label} does not run the complete authenticated {mode} scenario catalog" + ) + for transport in ("stdio", "official-http"): + if not any( + identity.mode == mode and identity.transport == transport + for identity in identities + ): + errors.append( + f"{label} is missing the required {transport} case for {mode}" + ) + observed = { + identity.scenario + for identity in identities + if identity.mode == mode + and identity.transport == "official-http" + and identity.source in {"official", "harness"} + } + missing = sorted(set(expected_scenarios) - observed) + unexpected = sorted(observed - set(expected_scenarios)) + if missing: + errors.append( + f"{label} did not observe {mode} scenarios: {', '.join(missing)}" + ) + if unexpected: + errors.append( + f"{label} observed unexpected {mode} scenarios: {', '.join(unexpected)}" + ) + return identities + + case_list = report.get("cases") + cases: dict[tuple[str, str], Mapping[str, object]] = {} + if not isinstance(case_list, list): + errors.append(f"{label} does not contain transport cases") + case_list = [] + for case in case_list: + if not isinstance(case, dict): + errors.append(f"{label} contains a malformed transport case") + continue + mode = case.get("mode") + transport = case.get("transport") + if mode not in REQUIRED_REGRESSION_MODES: + continue + if not isinstance(mode, str) or not isinstance(transport, str): + errors.append(f"{label} contains a malformed required transport case") + continue + key = (mode, transport) + if key in cases: + errors.append(f"{label} contains a duplicate {transport} case for {mode}") + continue + cases[key] = case + + identities: dict[_RegressionCheckIdentity, set[str]] = {} + for mode in REQUIRED_REGRESSION_MODES: + expected_scenarios = scenarios_for_mode(mode, include_auth=True) + actual_scenarios = selected.get(mode) + if ( + not isinstance(actual_scenarios, (list, tuple)) + or tuple(actual_scenarios) != expected_scenarios + ): + errors.append( + f"{label} does not run the complete authenticated {mode} scenario catalog" + ) + + for transport in ("stdio", "official-http"): + case = cases.get((mode, transport)) + if case is None: + errors.append( + f"{label} is missing the required {transport} case for {mode}" + ) + continue + raw_checks = case.get("checks") + if not isinstance(raw_checks, list) or not raw_checks: + errors.append(f"{label} has no {transport} checks for {mode}") + continue + + observed_scenarios: set[str] = set() + for check in raw_checks: + if not isinstance(check, dict): + errors.append( + f"{label} contains a malformed {transport} check for {mode}" + ) + continue + identity = _regression_check_identity(mode, transport, check) + status = check.get("status") + success = check.get("success") + if ( + identity is None + or status not in {"PASS", "FAIL", "SKIP"} + or not isinstance(success, bool) + or success != (status != "FAIL") + ): + errors.append( + f"{label} contains a malformed {transport} check for {mode}" + ) + continue + identities.setdefault(identity, set()).add(status) + if transport == "official-http" and identity.source in { + "official", + "harness", + }: + observed_scenarios.add(identity.scenario) + + if transport == "official-http": + expected = set(expected_scenarios) + missing = sorted(expected - observed_scenarios) + unexpected = sorted(observed_scenarios - expected) + if missing: + errors.append( + f"{label} did not observe {mode} scenarios: {', '.join(missing)}" + ) + if unexpected: + errors.append( + f"{label} observed unexpected {mode} scenarios: {', '.join(unexpected)}" + ) + + return identities + + +def _compact_regression_baseline(report: Mapping[str, object]) -> dict[str, object]: + errors: list[str] = [] + identities = _required_regression_checks(report, label="baseline", errors=errors) + automatic_auth = report.get("automaticAuthRequired") + if not isinstance(automatic_auth, bool): + errors.append("baseline does not record its production OAuth policy") + if any( + identity.mode == SHIPPING_LEGACY_VERSION and "FAIL" in statuses + for identity, statuses in identities.items() + ): + errors.append("baseline contains a shipping MCP regression") + if errors: + raise ValueError("; ".join(errors)) + + official = report.get("officialConformance") + assert isinstance(official, dict) + scenarios = official.get("scenarios") + assert isinstance(scenarios, dict) + return { + "baselineKind": COMPACT_REGRESSION_BASELINE_KIND, + "schemaVersion": REPORT_SCHEMA_VERSION, + "requiredModes": list(REQUIRED_REGRESSION_MODES), + "transports": ["stdio", "official-http"], + "versionCheck": {"success": True}, + "modernFeatureEnablement": True, + "automaticAuthRequired": automatic_auth, + "officialConformance": { + "repository": OFFICIAL_CONFORMANCE_REPOSITORY, + "gitRef": OFFICIAL_CONFORMANCE_GIT_REF, + "authenticationIncluded": True, + "scenarios": {mode: scenarios[mode] for mode in REQUIRED_REGRESSION_MODES}, + }, + "checks": { + "passing": [ + asdict(identity) + for identity, statuses in sorted(identities.items()) + if "PASS" in statuses + ], + "failing": [ + asdict(identity) + for identity, statuses in sorted(identities.items()) + if "FAIL" in statuses + ], + }, + } + + +def _write_json_file(path: Path, value: Mapping[str, object]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text( + json.dumps(value, ensure_ascii=False, indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + + +def _write_compact_regression_baseline( + report: Mapping[str, object], + path: Path, +) -> None: + _write_json_file(path, _compact_regression_baseline(report)) + + +def _evaluate_regression_gate( + report: Mapping[str, object], + baseline_report: Mapping[str, object], + *, + baseline_path: Path | None = None, +) -> dict[str, object]: + errors: list[str] = [] + baseline_checks = _required_regression_checks( + baseline_report, label="baseline", errors=errors + ) + candidate_checks = _required_regression_checks( + report, label="candidate", errors=errors + ) + + baseline_auth = baseline_report.get("automaticAuthRequired") + candidate_auth = report.get("automaticAuthRequired") + if not isinstance(baseline_auth, bool) or not isinstance(candidate_auth, bool): + errors.append( + "baseline and candidate must record their production OAuth policy" + ) + elif baseline_auth != candidate_auth: + errors.append("baseline and candidate use different production OAuth policies") + + new_failures: list[dict[str, object]] = [] + missing_checks: list[dict[str, object]] = [] + known_failures: list[dict[str, object]] = [] + fixed_checks: list[dict[str, object]] = [] + + for identity, statuses in sorted(candidate_checks.items()): + baseline_statuses = baseline_checks.get(identity, set()) + if "FAIL" not in statuses: + continue + if identity.mode == SHIPPING_LEGACY_VERSION: + errors.append(f"shipping MCP check failed: {identity.check_id}") + if "FAIL" in baseline_statuses: + known_failures.append(asdict(identity)) + else: + new_failures.append(asdict(identity)) + + for identity, baseline_statuses in sorted(baseline_checks.items()): + candidate_statuses = candidate_checks.get(identity, set()) + if identity.mode == SHIPPING_LEGACY_VERSION and "FAIL" in baseline_statuses: + errors.append( + f"baseline shipping MCP check is not passing: {identity.check_id}" + ) + if "PASS" in baseline_statuses and "PASS" not in candidate_statuses: + missing_checks.append(asdict(identity)) + elif ( + "FAIL" in baseline_statuses + and "FAIL" not in candidate_statuses + and "PASS" not in candidate_statuses + ): + missing_checks.append(asdict(identity)) + elif "FAIL" in baseline_statuses and "FAIL" not in candidate_statuses: + fixed_checks.append(asdict(identity)) + + return { + "success": not errors and not new_failures and not missing_checks, + "requiredModes": list(REQUIRED_REGRESSION_MODES), + "baselineReport": str(baseline_path) if baseline_path is not None else None, + "configurationErrors": errors, + "newFailures": new_failures, + "missingChecks": missing_checks, + "knownFailures": known_failures, + "fixedChecks": fixed_checks, + } + + +def _format_ratio(value: Mapping[str, object]) -> str: + passed = value.get("passed", 0) + total = value.get("total", 0) + if total == 0: + return "0/0 (N/A)" + percentage = value.get("percentage", 0) + return f"{passed}/{total} ({percentage:g}%)" + + +def _print_table(headers: Sequence[str], rows: Sequence[Sequence[str]]) -> None: + widths = [ + max(len(headers[index]), *(len(row[index]) for row in rows)) + for index in range(len(headers)) + ] + + def print_row(row: Sequence[str], separator: str = " | ") -> None: + print( + separator.join( + value.ljust(widths[index]) for index, value in enumerate(row) + ) + ) + + print_row(headers) + print("-+-".join("-" * width for width in widths)) + for row in rows: + print_row(row) + + +def _print_human_report(report: Mapping[str, object]) -> None: + success = report.get("success") is True + print(f"Codex MCP compliance: {'PASS' if success else 'FAIL'}") + print(f"Binary: {report.get('codexBinary')}") + print(f"Version: {report.get('codexVersion') or 'unknown'}") + failure_details: list[tuple[str, str, str]] = [] + cases = report.get("cases") + if isinstance(cases, list): + for case in cases: + if not isinstance(case, dict): + continue + label = f"{case.get('transport')} / {case.get('mode')}" + print(f" {'PASS' if case.get('success') else 'FAIL'} {label}") + checks = case.get("checks") + if not isinstance(checks, list): + continue + for check in checks: + if isinstance(check, dict) and check.get("success") is not True: + failure_details.append( + ( + label, + str(check.get("name")), + str(check.get("detail")), + ) + ) + + mode_summaries = report.get("modeSummaries") + if isinstance(mode_summaries, list) and mode_summaries: + print("\nPass rates by protocol mode") + summary_rows: list[list[str]] = [] + for item in mode_summaries: + if not isinstance(item, dict): + continue + checks = item.get("checks") + official_checks = item.get("officialChecks") + official_non_auth_checks = item.get("officialNonAuthChecks") + official_auth_checks = item.get("officialAuthChecks") + official_non_auth_scenarios = item.get("officialNonAuthScenarios") + official_auth_scenarios = item.get("officialAuthScenarios") + harness_checks = item.get("harnessChecks") + supplemental_checks = item.get("supplementalChecks") + transport_cases = item.get("transportCases") + if not all( + isinstance(value, dict) + for value in ( + checks, + official_checks, + official_non_auth_checks, + official_auth_checks, + official_non_auth_scenarios, + official_auth_scenarios, + harness_checks, + supplemental_checks, + transport_cases, + ) + ): + continue + summary_rows.append( + [ + str(item.get("mode")), + _format_ratio(official_non_auth_scenarios), + _format_ratio(official_auth_scenarios), + _format_ratio(official_checks), + _format_ratio(harness_checks), + _format_ratio(supplemental_checks), + _format_ratio(checks), + _format_ratio(transport_cases), + ] + ) + _print_table( + [ + "Mode", + "Non-auth scenarios", + "Auth scenarios", + "Official assertions", + "Harness", + "Supplemental", + "All checks", + "Transport cases", + ], + summary_rows, + ) + + test_matrix = report.get("testMatrix") + if isinstance(test_matrix, dict): + columns = test_matrix.get("columns") + rows = test_matrix.get("rows") + if isinstance(columns, list) and isinstance(rows, list) and columns: + matrix_headers = ["Test"] + column_keys: list[str] = [] + for column in columns: + if not isinstance(column, dict): + continue + key = str(column.get("key")) + column_keys.append(key) + matrix_headers.append(f"{column.get('transport')} {column.get('mode')}") + matrix_rows: list[list[str]] = [] + for row in rows: + if not isinstance(row, dict) or not isinstance( + row.get("results"), dict + ): + continue + results = row["results"] + matrix_rows.append( + [ + str(row.get("test")), + *(str(results.get(key, "N/A")) for key in column_keys), + ] + ) + print("\nPer-test results") + _print_table(matrix_headers, matrix_rows) + + if failure_details: + print("\nFailure details") + for label, name, detail in failure_details: + print(f" {label} / {name}: {detail}") + + summary = report.get("summary") + if isinstance(summary, dict): + print(f"Summary: {summary.get('passed')}/{summary.get('total')} cases passed") + gate = report.get("regressionGate") + if isinstance(gate, dict): + passed = gate.get("success") is True + known = gate.get("knownFailures") + known_count = len(known) if isinstance(known, list) else 0 + print( + f"Regression gate: {'PASS' if passed else 'FAIL'} " + f"({known_count} acknowledged baseline failures)" + ) + for category in ("configurationErrors", "newFailures", "missingChecks"): + values = gate.get(category) + if isinstance(values, list): + for value in values: + print(f" {category}: {value}") + artifacts = report.get("artifacts") + if isinstance(artifacts, str): + print(f"Artifacts: {artifacts}") + + +def _parse_args(argv: Sequence[str] | None) -> argparse.Namespace: + parser = argparse.ArgumentParser( + description=( + "Run the full versioned official MCP client conformance suite against a " + "supplied Codex binary, plus the supplemental local stdio suite." + ) + ) + parser.add_argument("codex_binary", type=Path) + parser.add_argument( + "--mode", + choices=("all", SHIPPING_LEGACY_VERSION, LEGACY_VERSION, MODERN_VERSION), + default="all", + help="Protocol era to test (default: both).", + ) + parser.add_argument( + "--transport", + choices=("all", "stdio", "http"), + default="all", + help=( + "Transport to test: supplemental stdio, official localhost HTTP, " + "or both (default: both)." + ), + ) + parser.add_argument( + "--server-script", + type=Path, + default=Path(__file__).resolve().with_name("server.py"), + help="Path to the MCP fixture server.py.", + ) + parser.add_argument( + "--adapter-script", + type=Path, + default=Path(__file__).resolve().with_name("codex_conformance_adapter.py"), + help="Path to the Codex adapter used by the official client suite.", + ) + parser.add_argument( + "--conformance-cli", + type=Path, + help=( + "Use an existing official conformance executable. By default the " + "runner bootstraps the pinned upstream Git commit with npx." + ), + ) + parser.add_argument( + "--official-scenario", + action="append", + help=( + "Run only this official versioned scenario. Repeat to select more " + "(default: every applicable scenario for the selected mode)." + ), + ) + parser.add_argument( + "--auth", + action=argparse.BooleanOptionalAction, + default=True, + help=( + "Include official OAuth authorization scenarios in HTTP coverage " + "(default: enabled). Use --no-auth for a faster non-auth slice." + ), + ) + parser.add_argument( + "--require-automatic-auth", + action="store_true", + help=( + "Require Codex to recover from OAuth scope escalation and " + "authorization-server migration without harness-injected re-login; " + "also exercise production client-metadata selection." + ), + ) + parser.add_argument( + "--timeout", + type=float, + default=45.0, + help="Per-command and per-request timeout in seconds.", + ) + parser.add_argument( + "--report", + type=Path, + help="Also write the complete JSON report to this path.", + ) + parser.add_argument( + "--baseline-report", + type=Path, + help=( + "Require the complete shipping, intermediate, and modern authenticated " + "HTTP and stdio matrices to preserve every passing check from this report. " + "Known baseline failures remain visible and new failures fail the gate." + ), + ) + parser.add_argument( + "--write-baseline", + type=Path, + help="Write a compact, deterministic regression baseline from the completed run.", + ) + parser.add_argument( + "--extract-baseline", + type=Path, + help=( + "Write a compact, deterministic baseline from --baseline-report " + "and exit without rerunning conformance." + ), + ) + parser.add_argument( + "--json", + action="store_true", + help="Print the report as JSON instead of the human summary.", + ) + parser.add_argument( + "--artifact-parent", + type=Path, + help="Parent directory for isolated temporary Codex homes.", + ) + parser.add_argument( + "--keep-artifacts", + action="store_true", + help="Keep isolated Codex homes and include their path in the report.", + ) + parser.add_argument( + "--enable-modern-feature", + action=argparse.BooleanOptionalAction, + default=True, + help=( + "Configure mcp_2026_07_28 before MCP startup and verify it through " + "the app-server runtime setter for modern cases (default: enabled)." + ), + ) + return parser.parse_args(argv) + + +def main(argv: Sequence[str] | None = None) -> int: + args = _parse_args(argv) + codex_binary = args.codex_binary.expanduser().resolve() + server_script = args.server_script.expanduser().resolve() + adapter_script = args.adapter_script.expanduser().resolve() + modes = ( + (SHIPPING_LEGACY_VERSION, LEGACY_VERSION, MODERN_VERSION) + if args.mode == "all" + else (args.mode,) + ) + transports = ("stdio", "http") if args.transport == "all" else (args.transport,) + if not codex_binary.is_file() or not os.access(codex_binary, os.X_OK): + print(f"error: Codex binary is not executable: {codex_binary}", file=sys.stderr) + return 2 + if "stdio" in transports and not server_script.is_file(): + print(f"error: fixture server does not exist: {server_script}", file=sys.stderr) + return 2 + if "http" in transports and not adapter_script.is_file(): + print( + f"error: official conformance adapter does not exist: {adapter_script}", + file=sys.stderr, + ) + return 2 + if args.timeout <= 0: + print("error: --timeout must be positive", file=sys.stderr) + return 2 + if args.extract_baseline is not None and args.baseline_report is None: + print("error: --extract-baseline requires --baseline-report", file=sys.stderr) + return 2 + if args.extract_baseline is not None and args.write_baseline is not None: + print( + "error: --extract-baseline cannot be combined with --write-baseline", + file=sys.stderr, + ) + return 2 + + baseline_report: dict[str, object] | None = None + baseline_path: Path | None = None + if args.baseline_report is not None: + baseline_path = args.baseline_report.expanduser().resolve() + try: + loaded_baseline = json.loads(baseline_path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as exc: + print( + f"error: cannot read baseline report {baseline_path}: {exc}", + file=sys.stderr, + ) + return 2 + if not isinstance(loaded_baseline, dict): + print( + f"error: baseline report must be a JSON object: {baseline_path}", + file=sys.stderr, + ) + return 2 + baseline_report = loaded_baseline + + if args.extract_baseline is not None: + assert baseline_report is not None + output_path = args.extract_baseline.expanduser().resolve() + try: + _write_compact_regression_baseline(baseline_report, output_path) + except (OSError, ValueError) as exc: + print( + f"error: cannot extract regression baseline {output_path}: {exc}", + file=sys.stderr, + ) + return 2 + print(f"Regression baseline: {output_path}") + return 0 + + conformance_command: list[str] + if args.conformance_cli is None: + conformance_command = default_conformance_command() + else: + conformance_cli = args.conformance_cli.expanduser().resolve() + if not conformance_cli.is_file(): + print( + f"error: official conformance CLI does not exist: {conformance_cli}", + file=sys.stderr, + ) + return 2 + if os.access(conformance_cli, os.X_OK): + conformance_command = [str(conformance_cli)] + elif conformance_cli.suffix in {".js", ".mjs"} and shutil.which("node"): + conformance_command = [str(shutil.which("node")), str(conformance_cli)] + else: + print( + "error: --conformance-cli must be executable (or a JavaScript " + f"file runnable with node): {conformance_cli}", + file=sys.stderr, + ) + return 2 + + try: + for mode in modes: + scenarios_for_mode( + mode, + args.official_scenario, + include_auth=args.auth, + ) + except ValueError as exc: + print(f"error: {exc}", file=sys.stderr) + return 2 + + artifact_parent: Path | None = None + if args.artifact_parent is not None: + artifact_parent = args.artifact_parent.expanduser().resolve() + artifact_parent.mkdir(parents=True, exist_ok=True) + + report, _ = run_compliance( + codex_binary, + server_script=server_script, + adapter_script=adapter_script, + conformance_command=conformance_command, + official_scenarios=args.official_scenario, + modes=modes, + transports=transports, + timeout_seconds=args.timeout, + artifact_parent=artifact_parent, + keep_artifacts=args.keep_artifacts, + enable_modern_feature=args.enable_modern_feature, + include_auth=args.auth, + require_automatic_auth=args.require_automatic_auth, + ) + if baseline_report is not None: + report["regressionGate"] = _evaluate_regression_gate( + report, + baseline_report, + baseline_path=baseline_path, + ) + if args.write_baseline is not None: + output_path = args.write_baseline.expanduser().resolve() + try: + _write_compact_regression_baseline(report, output_path) + except (OSError, ValueError) as exc: + print( + f"error: cannot write regression baseline {output_path}: {exc}", + file=sys.stderr, + ) + return 2 + + if args.report is not None: + report_path = args.report.expanduser().resolve() + _write_json_file(report_path, report) + if args.json: + print(json.dumps(report, ensure_ascii=False, indent=2, sort_keys=True)) + else: + _print_human_report(report) + if args.report is not None: + print(f"JSON report: {args.report.expanduser().resolve()}") + if baseline_report is not None: + gate = report.get("regressionGate") + return 0 if isinstance(gate, dict) and gate.get("success") is True else 1 + return 0 if report["success"] else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/mcp_conformance/server.py b/scripts/mcp_conformance/server.py new file mode 100644 index 0000000000..294fcb4c48 --- /dev/null +++ b/scripts/mcp_conformance/server.py @@ -0,0 +1,1744 @@ +#!/usr/bin/env python3 +"""MCP test server for shipping and draft legacy protocols and MCP 2026-07-28.""" + +import argparse +import base64 +import binascii +import json +import os +import sys +import threading +import uuid +from dataclasses import dataclass, field +from http import HTTPStatus +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from typing import IO, Mapping, Sequence +from urllib.parse import urlsplit + +SHIPPING_LEGACY_VERSION = "2025-06-18" +LEGACY_VERSION = "2025-11-25" +MODERN_VERSION = "2026-07-28" +SERVER_NAME = "openai-mcp-spec-test-server" +SERVER_VERSION = "0.1.0" + +DEFAULT_PROFILE = "default" +REVIEW_PROFILE = "review-regressions" +REPEATED_CURSOR_PROFILE = "repeated-cursor" +MISMATCHED_DISCOVERY_ID_PROFILE = "discovery-mismatched-id" +NULL_DISCOVERY_ID_PROFILE = "discovery-null-id" +SSE_CR_COMMENTS_PROFILE = "sse-cr-comments" +SSE_COMMENT_FLOOD_PROFILE = "sse-comment-flood" +CATALOG_MAX_PROFILE = "catalog-max" +CATALOG_OVER_LIMIT_PROFILE = "catalog-over-limit" +FIXTURE_PROFILES = ( + DEFAULT_PROFILE, + REVIEW_PROFILE, + REPEATED_CURSOR_PROFILE, + MISMATCHED_DISCOVERY_ID_PROFILE, + NULL_DISCOVERY_ID_PROFILE, + SSE_CR_COMMENTS_PROFILE, + SSE_COMMENT_FLOOD_PROFILE, + CATALOG_MAX_PROFILE, + CATALOG_OVER_LIMIT_PROFILE, +) +MAX_CATALOG_ITEMS = 1_024 +REVIEW_EXACT_INTEGER = 9_007_199_254_740_993 +REVIEW_MAX_INTEGER = 9_223_372_036_854_775_807 +REVIEW_MRTR_INPUT_REQUEST_COUNT = 65 +REVIEW_PAGE_CURSOR = "review-page-2" +REVIEW_SSE_EVENT_LIMIT_BYTES = 8 * 1_024 * 1_024 +REVIEW_SSE_COMMENT_LINE_BYTES = 2_048 +REVIEW_SSE_COMMENT_LINE_COUNT = ( + REVIEW_SSE_EVENT_LIMIT_BYTES // REVIEW_SSE_COMMENT_LINE_BYTES + 1 +) + +PARSE_ERROR = -32700 +INVALID_REQUEST = -32600 +METHOD_NOT_FOUND = -32601 +INVALID_PARAMS = -32602 +INTERNAL_ERROR = -32603 +HEADER_MISMATCH = -32020 +MISSING_REQUIRED_CLIENT_CAPABILITY = -32021 +UNSUPPORTED_PROTOCOL_VERSION = -32022 + +SERVER_INFO_META = { + "io.modelcontextprotocol/serverInfo": { + "name": SERVER_NAME, + "version": SERVER_VERSION, + } +} + +RESOURCE_URIS = ( + "test://fixture/readme", + "test://fixture/こんにちは", +) + + +@dataclass +class ConnectionState: + initialize_seen: bool = False + initialized: bool = False + + +@dataclass +class ResponsePlan: + response: dict[str, object] | None + notifications: list[dict[str, object]] = field(default_factory=list) + http_status: int = HTTPStatus.OK + force_sse: bool = False + + +class HeaderValidationError(ValueError): + pass + + +def _error_response( + request_id: object, + code: int, + message: str, + *, + data: dict[str, object] | None = None, +) -> dict[str, object]: + error: dict[str, object] = {"code": code, "message": message} + if data is not None: + error["data"] = data + return {"jsonrpc": "2.0", "id": request_id, "error": error} + + +def _response(request_id: object, result: dict[str, object]) -> dict[str, object]: + return {"jsonrpc": "2.0", "id": request_id, "result": result} + + +def _modern_result( + fields: Mapping[str, object] | None = None, + *, + cache_scope: str | None = None, + ttl_ms: int | None = None, + meta: Mapping[str, object] | None = None, +) -> dict[str, object]: + result: dict[str, object] = { + "resultType": "complete", + "_meta": dict(meta or SERVER_INFO_META), + } + if fields is not None: + result.update(fields) + if cache_scope is not None: + result["cacheScope"] = cache_scope + if ttl_ms is not None: + result["ttlMs"] = ttl_ms + return result + + +def _tool_definitions( + modern: bool, + *, + profile: str = DEFAULT_PROFILE, +) -> list[dict[str, object]]: + definitions: list[dict[str, object]] = [ + { + "name": "client_metadata", + "description": "Return the per-request MCP client metadata observed by the server.", + "inputSchema": {"type": "object", "additionalProperties": False}, + "outputSchema": {"type": "object"}, + "annotations": {"readOnlyHint": True}, + }, + { + "name": "echo", + "description": "Echo text through a normal MCP tool result.", + "inputSchema": { + "type": "object", + "properties": {"text": {"type": "string"}}, + "required": ["text"], + "additionalProperties": False, + }, + "outputSchema": { + "type": "object", + "properties": {"text": {"type": "string"}}, + "required": ["text"], + "additionalProperties": False, + }, + "annotations": {"readOnlyHint": True}, + }, + { + "name": "fail", + "description": "Return a tool execution error without using a JSON-RPC error.", + "inputSchema": {"type": "object", "additionalProperties": False}, + "annotations": {"readOnlyHint": True}, + }, + { + "name": "header_echo", + "description": "Echo values that modern HTTP clients must mirror into headers.", + "inputSchema": { + "type": "object", + "properties": { + "region": {"type": "string", "x-mcp-header": "Region"}, + "attempt": { + "type": "integer", + "minimum": -9_007_199_254_740_991, + "maximum": 9_007_199_254_740_991, + "x-mcp-header": "Attempt", + }, + "enabled": {"type": "boolean", "x-mcp-header": "Enabled"}, + "greeting": {"type": "string", "x-mcp-header": "Greeting"}, + }, + "required": ["region", "attempt", "enabled", "greeting"], + "additionalProperties": False, + }, + "outputSchema": {"type": "object"}, + "annotations": {"readOnlyHint": True}, + }, + { + "name": "progress", + "description": "Emit request-scoped progress and opt-in log notifications.", + "inputSchema": {"type": "object", "additionalProperties": False}, + "annotations": {"readOnlyHint": True}, + }, + ] + if modern: + # This tool first returns InputRequiredResult, not CallToolResult. Keep + # output-schema coverage on echo so the MRTR fixture tests one concern. + definitions.append( + { + "name": "request_input", + "description": "Exercise the 2026-07-28 multi round-trip request pattern.", + "inputSchema": {"type": "object", "additionalProperties": False}, + "annotations": {"readOnlyHint": True}, + } + ) + if profile in (REVIEW_PROFILE, REPEATED_CURSOR_PROFILE): + definitions.extend( + [ + { + "name": "review_large_integer", + "description": "Echo an integer without losing JSON-schema precision.", + "inputSchema": { + "type": "object", + "properties": { + "value": { + "type": "integer", + "minimum": REVIEW_EXACT_INTEGER, + "maximum": REVIEW_MAX_INTEGER, + "default": REVIEW_EXACT_INTEGER, + } + }, + "required": ["value"], + "additionalProperties": False, + }, + "outputSchema": { + "type": "object", + "properties": {"value": {"type": "integer"}}, + "required": ["value"], + "additionalProperties": False, + }, + "annotations": {"readOnlyHint": True}, + }, + { + "name": "review_protocol_env", + "description": "Report the MCP protocol-version environment variable.", + "inputSchema": {"type": "object", "additionalProperties": False}, + "outputSchema": { + "type": "object", + "properties": {"value": {"type": ["string", "null"]}}, + "required": ["value"], + "additionalProperties": False, + }, + "annotations": {"readOnlyHint": True}, + }, + ] + ) + if modern: + definitions.extend( + [ + { + "name": "review_integer_elicitation", + "description": "Request an integer form without losing schema precision.", + "inputSchema": { + "type": "object", + "additionalProperties": False, + }, + "annotations": {"readOnlyHint": True}, + }, + { + "name": "review_mrtr_cap", + "description": ( + "Return more simultaneous input requests than the client cap." + ), + "inputSchema": { + "type": "object", + "additionalProperties": False, + }, + "annotations": {"readOnlyHint": True}, + }, + ] + ) + if profile in (CATALOG_MAX_PROFILE, CATALOG_OVER_LIMIT_PROFILE): + catalog_size = MAX_CATALOG_ITEMS + if profile == CATALOG_OVER_LIMIT_PROFILE: + catalog_size += 1 + definitions.extend( + { + "name": f"catalog_boundary_{index:04d}", + "description": "Exercise the MCP catalog item boundary.", + "inputSchema": { + "type": "object", + "additionalProperties": False, + }, + "annotations": {"readOnlyHint": True}, + } + for index in range(catalog_size - len(definitions)) + ) + return sorted(definitions, key=lambda tool: str(tool["name"])) + + +class ProtocolServer: + def __init__(self, mode: str, *, profile: str = DEFAULT_PROFILE) -> None: + if mode not in (SHIPPING_LEGACY_VERSION, LEGACY_VERSION, MODERN_VERSION): + raise ValueError(f"unsupported mode: {mode}") + if profile not in FIXTURE_PROFILES: + raise ValueError(f"unsupported profile: {profile}") + self.mode = mode + self.profile = profile + + @property + def modern(self) -> bool: + return self.mode == MODERN_VERSION + + def handle( + self, + message: object, + state: ConnectionState, + ) -> ResponsePlan: + if not isinstance(message, dict): + return ResponsePlan( + _error_response(None, INVALID_REQUEST, "Request must be a JSON object"), + http_status=HTTPStatus.BAD_REQUEST, + ) + + if message.get("jsonrpc") != "2.0" or not isinstance( + message.get("method"), str + ): + return ResponsePlan( + _error_response( + message.get("id"), + INVALID_REQUEST, + "Invalid JSON-RPC 2.0 request", + ), + http_status=HTTPStatus.BAD_REQUEST, + ) + + method = str(message["method"]) + is_notification = "id" not in message + request_id = message.get("id") + params = message.get("params", {}) + if not isinstance(params, dict): + return ResponsePlan( + None + if is_notification + else _error_response( + request_id, INVALID_PARAMS, "params must be an object" + ), + http_status=HTTPStatus.BAD_REQUEST, + ) + + if is_notification: + return self._handle_notification(method, params, state) + + if self.modern: + return self._handle_modern(request_id, method, params) + return self._handle_legacy(request_id, method, params, state) + + def _handle_notification( + self, + method: str, + params: dict[str, object], + state: ConnectionState, + ) -> ResponsePlan: + if not self.modern and method == "notifications/initialized": + if state.initialize_seen: + state.initialized = True + return ResponsePlan(None, http_status=HTTPStatus.ACCEPTED) + if method == "notifications/cancelled": + return ResponsePlan(None, http_status=HTTPStatus.ACCEPTED) + return ResponsePlan(None, http_status=HTTPStatus.ACCEPTED) + + def _handle_modern( + self, + request_id: object, + method: str, + params: dict[str, object], + ) -> ResponsePlan: + if method == "initialize": + return ResponsePlan( + _error_response( + request_id, + METHOD_NOT_FOUND, + f"initialize is not supported; this server speaks MCP {MODERN_VERSION}", + data={"supported": [MODERN_VERSION]}, + ), + http_status=HTTPStatus.NOT_FOUND, + ) + + meta = params.get("_meta") + if not isinstance(meta, dict): + return ResponsePlan( + _error_response( + request_id, + INVALID_PARAMS, + "params._meta is required on every 2026-07-28 request", + ), + http_status=HTTPStatus.BAD_REQUEST, + ) + + version = meta.get("io.modelcontextprotocol/protocolVersion") + capabilities = meta.get("io.modelcontextprotocol/clientCapabilities") + if not isinstance(version, str) or not isinstance(capabilities, dict): + return ResponsePlan( + _error_response( + request_id, + INVALID_PARAMS, + "_meta must include protocolVersion and clientCapabilities", + ), + http_status=HTTPStatus.BAD_REQUEST, + ) + if version != MODERN_VERSION: + return ResponsePlan( + _error_response( + request_id, + UNSUPPORTED_PROTOCOL_VERSION, + "Unsupported protocol version", + data={"supported": [MODERN_VERSION], "requested": version}, + ), + http_status=HTTPStatus.BAD_REQUEST, + ) + + if method == "server/discover": + response_id = request_id + if self.profile == MISMATCHED_DISCOVERY_ID_PROFILE: + response_id = f"review-mismatched-{request_id}" + elif self.profile == NULL_DISCOVERY_ID_PROFILE: + response_id = None + return ResponsePlan( + _response( + response_id, + _modern_result( + { + "supportedVersions": [MODERN_VERSION], + "capabilities": self._modern_capabilities(), + "instructions": ( + "Use echo for a basic call, header_echo for HTTP " + "header mirroring, request_input for MRTR, and " + "progress for request-scoped notifications." + ), + }, + cache_scope="public", + ttl_ms=60_000, + ), + ) + ) + if method == "tools/list": + return self._list_tools(request_id, params) + if method == "tools/call": + return self._call_tool(request_id, params, meta) + if method == "resources/list": + return self._list_resources(request_id, params) + if method == "resources/templates/list": + return ResponsePlan( + _response( + request_id, + _modern_result( + { + "resourceTemplates": [ + { + "name": "fixture-by-name", + "uriTemplate": "test://fixture/{name}", + "description": "Read a named fixture resource.", + "mimeType": "text/plain", + } + ] + }, + cache_scope="public", + ttl_ms=5_000, + ), + ) + ) + if method == "resources/read": + return self._read_resource(request_id, params) + if method == "prompts/list": + return ResponsePlan( + _response( + request_id, + _modern_result( + {"prompts": self._prompts()}, + cache_scope="public", + ttl_ms=5_000, + ), + ) + ) + if method == "prompts/get": + return self._get_prompt(request_id, params) + if method == "subscriptions/listen": + return self._listen(request_id, params) + + return ResponsePlan( + _error_response(request_id, METHOD_NOT_FOUND, f"Unknown method: {method}"), + http_status=HTTPStatus.NOT_FOUND, + ) + + def _handle_legacy( + self, + request_id: object, + method: str, + params: dict[str, object], + state: ConnectionState, + ) -> ResponsePlan: + if method == "initialize": + requested_version = params.get("protocolVersion") + if not isinstance(requested_version, str): + return ResponsePlan( + _error_response( + request_id, + INVALID_PARAMS, + "initialize requires protocolVersion", + ) + ) + state.initialize_seen = True + return ResponsePlan( + _response( + request_id, + { + "protocolVersion": self.mode, + "capabilities": self._legacy_capabilities(), + "serverInfo": { + "name": SERVER_NAME, + "version": SERVER_VERSION, + }, + "instructions": f"Legacy {self.mode} compatibility fixture.", + }, + ) + ) + + if method == "server/discover": + return ResponsePlan( + _error_response( + request_id, + METHOD_NOT_FOUND, + "Unknown method: server/discover", + ), + http_status=HTTPStatus.NOT_FOUND, + ) + if not state.initialized: + return ResponsePlan( + _error_response( + request_id, + INVALID_REQUEST, + "Server has not received notifications/initialized", + ) + ) + if method == "ping": + return ResponsePlan(_response(request_id, {})) + if method == "logging/setLevel": + return ResponsePlan(_response(request_id, {})) + if method == "tools/list": + return self._list_tools(request_id, params) + if method == "tools/call": + return self._call_tool(request_id, params, {}) + if method == "resources/list": + return self._list_resources(request_id, params) + if method == "resources/templates/list": + return ResponsePlan( + _response( + request_id, + { + "resourceTemplates": [ + { + "name": "fixture-by-name", + "uriTemplate": "test://fixture/{name}", + "description": "Read a named fixture resource.", + "mimeType": "text/plain", + } + ] + }, + ) + ) + if method == "resources/read": + return self._read_resource(request_id, params) + if method == "prompts/list": + return ResponsePlan(_response(request_id, {"prompts": self._prompts()})) + if method == "prompts/get": + return self._get_prompt(request_id, params) + return ResponsePlan( + _error_response(request_id, METHOD_NOT_FOUND, f"Unknown method: {method}"), + http_status=HTTPStatus.NOT_FOUND, + ) + + def _list_tools( + self, + request_id: object, + params: dict[str, object], + ) -> ResponsePlan: + definitions = _tool_definitions(self.modern, profile=self.profile) + fields: dict[str, object] + if self.modern and self.profile in (REVIEW_PROFILE, REPEATED_CURSOR_PROFILE): + cursor = params.get("cursor") + if cursor is None: + fields = { + "tools": definitions[:2], + "nextCursor": REVIEW_PAGE_CURSOR, + } + elif cursor == REVIEW_PAGE_CURSOR: + fields = {"tools": definitions[2:]} + if self.profile == REPEATED_CURSOR_PROFILE: + fields["nextCursor"] = REVIEW_PAGE_CURSOR + else: + return ResponsePlan( + _error_response(request_id, INVALID_PARAMS, "Invalid tool cursor") + ) + else: + fields = {"tools": definitions} + + result = ( + _modern_result(fields, cache_scope="public", ttl_ms=1_000) + if self.modern + else fields + ) + return ResponsePlan(_response(request_id, result)) + + @staticmethod + def _modern_capabilities() -> dict[str, object]: + return { + "tools": {"listChanged": True}, + "resources": {"subscribe": True, "listChanged": True}, + "prompts": {"listChanged": True}, + "logging": {}, + "extensions": {"com.openai/mcp-spec-test": {}}, + } + + @staticmethod + def _legacy_capabilities() -> dict[str, object]: + return { + "tools": {"listChanged": True}, + "resources": {"subscribe": True, "listChanged": True}, + "prompts": {"listChanged": True}, + "logging": {}, + } + + def _call_tool( + self, + request_id: object, + params: dict[str, object], + request_meta: dict[str, object], + ) -> ResponsePlan: + name = params.get("name") + arguments = params.get("arguments", {}) + if not isinstance(name, str) or not isinstance(arguments, dict): + return ResponsePlan( + _error_response( + request_id, + INVALID_PARAMS, + "tools/call requires a string name and object arguments", + ) + ) + known = { + str(tool["name"]) + for tool in _tool_definitions(self.modern, profile=self.profile) + } + if name not in known: + return ResponsePlan( + _error_response(request_id, INVALID_PARAMS, f"Unknown tool: {name}") + ) + + if name == "request_input": + return self._request_input(request_id, params, request_meta) + if name == "review_integer_elicitation": + return self._review_integer_elicitation(request_id, params, request_meta) + if name == "review_mrtr_cap": + return self._review_mrtr_cap(request_id, request_meta) + + notifications: list[dict[str, object]] = [] + if name == "progress": + progress_token = request_meta.get("progressToken") + if isinstance(progress_token, (str, int)) and not isinstance( + progress_token, bool + ): + notifications.append( + { + "jsonrpc": "2.0", + "method": "notifications/progress", + "params": { + "progressToken": progress_token, + "progress": 1, + "total": 1, + "message": "fixture complete", + }, + } + ) + if isinstance(request_meta.get("io.modelcontextprotocol/logLevel"), str): + notifications.append( + { + "jsonrpc": "2.0", + "method": "notifications/message", + "params": { + "level": "info", + "logger": SERVER_NAME, + "data": "request-scoped fixture log", + }, + } + ) + + if name == "review_large_integer": + value = arguments.get("value") + if ( + not isinstance(value, int) + or isinstance(value, bool) + or not REVIEW_EXACT_INTEGER <= value <= REVIEW_MAX_INTEGER + ): + return ResponsePlan( + _error_response( + request_id, + INVALID_PARAMS, + "review_large_integer requires an exactly representable JSON integer", + ) + ) + structured = {"value": value} + fields: dict[str, object] = { + "content": [{"type": "text", "text": json.dumps(structured)}], + "structuredContent": structured, + } + elif name == "review_protocol_env": + structured = {"value": os.environ.get("CODEX_MCP_PROTOCOL_VERSION")} + fields = { + "content": [{"type": "text", "text": json.dumps(structured)}], + "structuredContent": structured, + } + elif name == "echo": + text = arguments.get("text") + if not isinstance(text, str): + return ResponsePlan( + _error_response( + request_id, + INVALID_PARAMS, + "echo requires a string text argument", + ) + ) + fields = { + "content": [{"type": "text", "text": text}], + "structuredContent": {"text": text}, + } + elif name == "client_metadata": + fields = { + "content": [ + { + "type": "text", + "text": json.dumps(request_meta, sort_keys=True), + } + ], + "structuredContent": request_meta, + } + elif name == "header_echo": + required = { + "region": str, + "attempt": int, + "enabled": bool, + "greeting": str, + } + if any( + key not in arguments + or not isinstance(arguments[key], expected_type) + or (expected_type is int and isinstance(arguments[key], bool)) + for key, expected_type in required.items() + ) or not ( + -9_007_199_254_740_991 + <= int(arguments["attempt"]) + <= 9_007_199_254_740_991 + ): + return ResponsePlan( + _error_response( + request_id, + INVALID_PARAMS, + "header_echo arguments have invalid types", + ) + ) + fields = { + "content": [ + { + "type": "text", + "text": json.dumps( + arguments, ensure_ascii=False, sort_keys=True + ), + } + ], + "structuredContent": arguments, + } + elif name == "fail": + fields = { + "content": [ + { + "type": "text", + "text": "Intentional tool execution failure", + } + ], + "isError": True, + } + else: + fields = { + "content": [{"type": "text", "text": "progress complete"}], + } + + result = _modern_result(fields) if self.modern else fields + return ResponsePlan( + _response(request_id, result), + notifications=notifications, + force_sse=bool(notifications), + ) + + def _review_integer_elicitation( + self, + request_id: object, + params: dict[str, object], + request_meta: dict[str, object], + ) -> ResponsePlan: + capabilities = request_meta.get( + "io.modelcontextprotocol/clientCapabilities", {} + ) + if not isinstance(capabilities, dict) or not isinstance( + capabilities.get("elicitation"), dict + ): + return ResponsePlan( + _error_response( + request_id, + MISSING_REQUIRED_CLIENT_CAPABILITY, + "review_integer_elicitation requires elicitation support", + data={"requiredCapabilities": {"elicitation": {"form": {}}}}, + ), + http_status=HTTPStatus.BAD_REQUEST, + ) + + input_responses = params.get("inputResponses") + request_state = params.get("requestState") + expected_state = "opaque:review_integer_elicitation:v1" + if input_responses is None and request_state is None: + return ResponsePlan( + _response( + request_id, + { + "resultType": "input_required", + "_meta": dict(SERVER_INFO_META), + "inputRequests": { + "large_integer": { + "method": "elicitation/create", + "params": { + "mode": "form", + "message": "Return the exact large integer.", + "requestedSchema": { + "type": "object", + "properties": { + "value": { + "type": "integer", + "minimum": REVIEW_EXACT_INTEGER, + "maximum": REVIEW_MAX_INTEGER, + "default": REVIEW_EXACT_INTEGER, + } + }, + "required": ["value"], + }, + }, + } + }, + "requestState": expected_state, + }, + ) + ) + + if request_state != expected_state or not isinstance(input_responses, dict): + return ResponsePlan( + _error_response( + request_id, + INVALID_PARAMS, + "integer elicitation retry must echo requestState and inputResponses", + ) + ) + + response = input_responses.get("large_integer") + if not isinstance(response, dict) or response.get("action") != "accept": + return ResponsePlan( + _error_response( + request_id, INVALID_PARAMS, "integer input must be accepted" + ) + ) + + content = response.get("content") + value = content.get("value") if isinstance(content, dict) else None + if ( + not isinstance(value, int) + or isinstance(value, bool) + or not REVIEW_EXACT_INTEGER <= value <= REVIEW_MAX_INTEGER + ): + return ResponsePlan( + _error_response( + request_id, + INVALID_PARAMS, + "integer elicitation requires the exact integer form response", + ) + ) + + return ResponsePlan( + _response( + request_id, + _modern_result( + { + "content": [{"type": "text", "text": str(value)}], + "structuredContent": {"value": value}, + } + ), + ) + ) + + def _review_mrtr_cap( + self, + request_id: object, + request_meta: dict[str, object], + ) -> ResponsePlan: + capabilities = request_meta.get( + "io.modelcontextprotocol/clientCapabilities", {} + ) + if not isinstance(capabilities, dict) or not isinstance( + capabilities.get("elicitation"), dict + ): + return ResponsePlan( + _error_response( + request_id, + MISSING_REQUIRED_CLIENT_CAPABILITY, + "review_mrtr_cap requires elicitation support", + data={"requiredCapabilities": {"elicitation": {"form": {}}}}, + ), + http_status=HTTPStatus.BAD_REQUEST, + ) + + return ResponsePlan( + _response( + request_id, + { + "resultType": "input_required", + "_meta": dict(SERVER_INFO_META), + "inputRequests": { + f"review-input-{index:02d}": { + "method": "elicitation/create", + "params": { + "mode": "form", + "message": f"Review MRTR input {index}.", + "requestedSchema": { + "type": "object", + "properties": {"value": {"type": "string"}}, + "required": ["value"], + }, + }, + } + for index in range(REVIEW_MRTR_INPUT_REQUEST_COUNT) + }, + "requestState": "opaque:review_mrtr_cap:v1", + }, + ) + ) + + def _request_input( + self, + request_id: object, + params: dict[str, object], + request_meta: dict[str, object], + ) -> ResponsePlan: + capabilities = request_meta.get( + "io.modelcontextprotocol/clientCapabilities", {} + ) + if not isinstance(capabilities, dict) or not isinstance( + capabilities.get("elicitation"), dict + ): + return ResponsePlan( + _error_response( + request_id, + MISSING_REQUIRED_CLIENT_CAPABILITY, + "request_input requires elicitation support", + data={"requiredCapabilities": {"elicitation": {"form": {}}}}, + ), + http_status=HTTPStatus.BAD_REQUEST, + ) + + input_responses = params.get("inputResponses") + request_state = params.get("requestState") + expected_state = "opaque:request_input:v1" + if input_responses is None and request_state is None: + return ResponsePlan( + _response( + request_id, + { + "resultType": "input_required", + "_meta": dict(SERVER_INFO_META), + "inputRequests": { + "confirmation": { + "method": "elicitation/create", + "params": { + "mode": "form", + "message": "Confirm the MCP 2026-07-28 MRTR test.", + "requestedSchema": { + "type": "object", + "properties": { + "confirmation": { + "type": "string", + "default": "confirmed", + } + }, + "required": ["confirmation"], + }, + }, + } + }, + "requestState": expected_state, + }, + ) + ) + + if request_state != expected_state or not isinstance(input_responses, dict): + return ResponsePlan( + _error_response( + request_id, + INVALID_PARAMS, + "MRTR retry must echo requestState and provide inputResponses", + ) + ) + confirmation = input_responses.get("confirmation") + if not isinstance(confirmation, dict) or confirmation.get("action") != "accept": + return ResponsePlan( + _error_response( + request_id, + INVALID_PARAMS, + "confirmation input must be accepted", + ) + ) + content = confirmation.get("content") + if not isinstance(content, dict) or not isinstance( + content.get("confirmation"), str + ): + return ResponsePlan( + _error_response( + request_id, + INVALID_PARAMS, + "confirmation content is required", + ) + ) + value = str(content["confirmation"]) + return ResponsePlan( + _response( + request_id, + _modern_result( + { + "content": [{"type": "text", "text": value}], + "structuredContent": {"confirmation": value}, + } + ), + ) + ) + + def _list_resources( + self, + request_id: object, + params: dict[str, object], + ) -> ResponsePlan: + cursor = params.get("cursor") + if cursor is None: + resources = [self._resource(RESOURCE_URIS[0])] + next_cursor: str | None = "page-2" + elif cursor == "page-2": + resources = [self._resource(RESOURCE_URIS[1])] + next_cursor = None + else: + return ResponsePlan( + _error_response(request_id, INVALID_PARAMS, "Invalid resource cursor") + ) + + fields: dict[str, object] = {"resources": resources} + if next_cursor is not None: + fields["nextCursor"] = next_cursor + result = ( + _modern_result(fields, cache_scope="public", ttl_ms=2_000) + if self.modern + else fields + ) + return ResponsePlan(_response(request_id, result)) + + @staticmethod + def _resource(uri: str) -> dict[str, object]: + return { + "name": uri.rsplit("/", 1)[-1], + "title": f"Fixture resource: {uri.rsplit('/', 1)[-1]}", + "uri": uri, + "description": "A deterministic text resource from the MCP spec fixture.", + "mimeType": "text/plain", + } + + def _read_resource( + self, + request_id: object, + params: dict[str, object], + ) -> ResponsePlan: + uri = params.get("uri") + if uri not in RESOURCE_URIS: + return ResponsePlan( + _error_response(request_id, INVALID_PARAMS, "Resource not found") + ) + fields: dict[str, object] = { + "contents": [ + { + "uri": uri, + "mimeType": "text/plain", + "text": f"fixture contents for {uri}", + } + ] + } + result = ( + _modern_result(fields, cache_scope="private", ttl_ms=500) + if self.modern + else fields + ) + return ResponsePlan(_response(request_id, result)) + + @staticmethod + def _prompts() -> list[dict[str, object]]: + return [ + { + "name": "fixture.greeting", + "title": "Fixture greeting", + "description": "Create a deterministic greeting.", + "arguments": [ + { + "name": "name", + "description": "Name to greet.", + "required": True, + } + ], + } + ] + + def _get_prompt( + self, + request_id: object, + params: dict[str, object], + ) -> ResponsePlan: + if params.get("name") != "fixture.greeting": + return ResponsePlan( + _error_response(request_id, INVALID_PARAMS, "Unknown prompt") + ) + arguments = params.get("arguments", {}) + if not isinstance(arguments, dict) or not isinstance( + arguments.get("name"), str + ): + return ResponsePlan( + _error_response( + request_id, + INVALID_PARAMS, + "fixture.greeting requires the name argument", + ) + ) + fields: dict[str, object] = { + "description": "A deterministic fixture greeting.", + "messages": [ + { + "role": "user", + "content": { + "type": "text", + "text": f"Hello, {arguments['name']}!", + }, + } + ], + } + result = _modern_result(fields) if self.modern else fields + return ResponsePlan(_response(request_id, result)) + + def _listen( + self, + request_id: object, + params: dict[str, object], + ) -> ResponsePlan: + requested = params.get("notifications") + if not isinstance(requested, dict): + return ResponsePlan( + _error_response( + request_id, + INVALID_PARAMS, + "subscriptions/listen requires a notifications filter", + ) + ) + + acknowledged: dict[str, object] = {} + if requested.get("toolsListChanged") is True: + acknowledged["toolsListChanged"] = True + if requested.get("promptsListChanged") is True: + acknowledged["promptsListChanged"] = True + if requested.get("resourcesListChanged") is True: + acknowledged["resourcesListChanged"] = True + subscriptions = requested.get("resourceSubscriptions") + valid_subscriptions = ( + [uri for uri in subscriptions if uri in RESOURCE_URIS] + if isinstance(subscriptions, list) + and all(isinstance(uri, str) for uri in subscriptions) + else [] + ) + if valid_subscriptions: + acknowledged["resourceSubscriptions"] = valid_subscriptions + + subscription_meta = { + "io.modelcontextprotocol/subscriptionId": request_id, + } + notifications: list[dict[str, object]] = [ + { + "jsonrpc": "2.0", + "method": "notifications/subscriptions/acknowledged", + "params": { + "_meta": subscription_meta, + "notifications": acknowledged, + }, + } + ] + notification_methods = ( + ("toolsListChanged", "notifications/tools/list_changed"), + ("promptsListChanged", "notifications/prompts/list_changed"), + ("resourcesListChanged", "notifications/resources/list_changed"), + ) + for filter_name, method in notification_methods: + if acknowledged.get(filter_name) is True: + notifications.append( + { + "jsonrpc": "2.0", + "method": method, + "params": {"_meta": subscription_meta}, + } + ) + for uri in valid_subscriptions: + notifications.append( + { + "jsonrpc": "2.0", + "method": "notifications/resources/updated", + "params": {"_meta": subscription_meta, "uri": uri}, + } + ) + + closing_meta = { + **SERVER_INFO_META, + "io.modelcontextprotocol/subscriptionId": request_id, + } + return ResponsePlan( + _response( + request_id, + _modern_result(meta=closing_meta), + ), + notifications=notifications, + force_sse=True, + ) + + +def _decode_mirrored_header(value: str) -> str: + if value.startswith("=?base64?") and value.endswith("?="): + encoded = value[len("=?base64?") : -len("?=")] + try: + return base64.b64decode(encoded, validate=True).decode("utf-8") + except (binascii.Error, UnicodeDecodeError) as exc: + raise HeaderValidationError("invalid Base64 mirrored header") from exc + if value.startswith("=?base64?") or value.endswith("?="): + raise HeaderValidationError("malformed Base64 sentinel") + if value != value.strip() or any( + char != "\t" and (ord(char) < 0x20 or ord(char) > 0x7E) for char in value + ): + raise HeaderValidationError("unsafe plain mirrored header") + return value + + +def validate_modern_http_headers( + message: dict[str, object], + headers: Mapping[str, str], +) -> None: + normalized = {name.lower(): value for name, value in headers.items()} + accept = { + value.strip().split(";", 1)[0].lower() + for value in normalized.get("accept", "").split(",") + } + if not {"application/json", "text/event-stream"}.issubset(accept): + raise HeaderValidationError( + "Accept must include application/json and text/event-stream" + ) + + method = message.get("method") + params = message.get("params") + if not isinstance(method, str) or not isinstance(params, dict): + return + meta = params.get("_meta") + + header_version = normalized.get("mcp-protocol-version") + if header_version is None: + raise HeaderValidationError("missing MCP-Protocol-Version header") + if isinstance(meta, dict): + body_version = meta.get("io.modelcontextprotocol/protocolVersion") + if isinstance(body_version, str) and body_version != header_version: + raise HeaderValidationError( + "MCP-Protocol-Version header does not match request _meta" + ) + + header_method = normalized.get("mcp-method") + if header_method is None: + raise HeaderValidationError("missing Mcp-Method header") + if header_method != method: + raise HeaderValidationError("Mcp-Method header does not match request method") + + name_source: object | None = None + if method in ("tools/call", "prompts/get"): + name_source = params.get("name") + elif method == "resources/read": + name_source = params.get("uri") + if name_source is not None: + header_name = normalized.get("mcp-name") + if header_name is None: + raise HeaderValidationError("missing Mcp-Name header") + if ( + not isinstance(name_source, str) + or _decode_mirrored_header(header_name) != name_source + ): + raise HeaderValidationError("Mcp-Name header does not match request body") + + if method != "tools/call" or params.get("name") != "header_echo": + return + arguments = params.get("arguments", {}) + if not isinstance(arguments, dict): + return + mirrored = { + "region": "region", + "attempt": "attempt", + "enabled": "enabled", + "greeting": "greeting", + } + for argument_name, header_suffix in mirrored.items(): + header_key = f"mcp-param-{header_suffix}" + argument = arguments.get(argument_name) + header_value = normalized.get(header_key) + if argument is None: + if header_value is not None: + raise HeaderValidationError( + f"Mcp-Param-{header_suffix} must be omitted" + ) + continue + if header_value is None: + raise HeaderValidationError(f"missing Mcp-Param-{header_suffix} header") + decoded = _decode_mirrored_header(header_value) + if isinstance(argument, bool): + expected = "true" if argument else "false" + elif isinstance(argument, int): + expected = str(argument) + elif isinstance(argument, str): + expected = argument + else: + raise HeaderValidationError( + f"unsupported mirrored argument type for {argument_name}" + ) + if decoded != expected: + raise HeaderValidationError( + f"Mcp-Param-{header_suffix} header does not match request body" + ) + + +def _encode_sse_messages( + messages: Sequence[Mapping[str, object]], + *, + profile: str, +) -> bytes: + if profile == SSE_COMMENT_FLOOD_PROFILE: + comment_prefix = b": reviewer keepalive " + comment = ( + comment_prefix + + b"x" * (REVIEW_SSE_COMMENT_LINE_BYTES - len(comment_prefix) - 1) + + b"\r" + ) + prefix = comment * REVIEW_SSE_COMMENT_LINE_COUNT + separator = "\r\r" + elif profile == SSE_CR_COMMENTS_PROFILE: + prefix = b": reviewer keepalive\r" + separator = "\r\r" + else: + prefix = b"" + separator = "\n\n" + + return prefix + b"".join( + ( + f"data: {json.dumps(message, ensure_ascii=False, separators=(',', ':'))}" + f"{separator}" + ).encode() + for message in messages + ) + + +class FixtureHTTPServer(ThreadingHTTPServer): + daemon_threads = True + + def __init__( + self, + address: tuple[str, int], + fixture: ProtocolServer, + *, + endpoint: str, + allowed_origins: Sequence[str], + log_requests: bool, + ) -> None: + self.fixture = fixture + self.endpoint = endpoint + self.allowed_origins = set(allowed_origins) + self.log_requests = log_requests + self.sessions: dict[str, ConnectionState] = {} + self.sessions_lock = threading.Lock() + super().__init__(address, FixtureRequestHandler) + + +class FixtureRequestHandler(BaseHTTPRequestHandler): + server: FixtureHTTPServer + + def do_GET(self) -> None: + if self.path == "/healthz": + body = json.dumps( + {"status": "ok", "mode": self.server.fixture.mode} + ).encode() + self._send_bytes(HTTPStatus.OK, "application/json", body) + return + self._send_bytes( + HTTPStatus.METHOD_NOT_ALLOWED, + "application/json", + b"", + extra_headers={"Allow": "POST"}, + ) + + def do_DELETE(self) -> None: + if self.path != self.server.endpoint or self.server.fixture.modern: + self._send_bytes( + HTTPStatus.METHOD_NOT_ALLOWED, + "application/json", + b"", + extra_headers={"Allow": "POST"}, + ) + return + session_id = self.headers.get("Mcp-Session-Id") + with self.server.sessions_lock: + existed = ( + session_id is not None + and self.server.sessions.pop(session_id, None) is not None + ) + self._send_bytes( + HTTPStatus.OK if existed else HTTPStatus.NOT_FOUND, + "application/json", + b"", + ) + + def do_POST(self) -> None: + if self.path != self.server.endpoint: + self._send_json_error( + None, + HTTPStatus.NOT_FOUND, + METHOD_NOT_FOUND, + f"Unknown MCP endpoint: {self.path}", + ) + return + if not self._origin_is_allowed(): + self._send_json_error( + None, + HTTPStatus.FORBIDDEN, + INVALID_REQUEST, + "Origin is not allowed", + ) + return + if self.headers.get_content_type() != "application/json": + self._send_json_error( + None, + HTTPStatus.UNSUPPORTED_MEDIA_TYPE, + INVALID_REQUEST, + "Content-Type must be application/json", + ) + return + try: + content_length = int(self.headers.get("Content-Length", "")) + except ValueError: + content_length = -1 + if content_length < 0 or content_length > 1_048_576: + self._send_json_error( + None, + HTTPStatus.BAD_REQUEST, + INVALID_REQUEST, + "Invalid Content-Length", + ) + return + try: + message = json.loads(self.rfile.read(content_length)) + except (json.JSONDecodeError, UnicodeDecodeError): + self._send_json_error( + None, + HTTPStatus.BAD_REQUEST, + PARSE_ERROR, + "Parse error", + ) + return + if not isinstance(message, dict): + self._send_json_error( + None, + HTTPStatus.BAD_REQUEST, + INVALID_REQUEST, + "Request must be a JSON object", + ) + return + + request_id = message.get("id") + if self.server.fixture.modern: + if "id" in message: + try: + validate_modern_http_headers(message, self.headers) + except HeaderValidationError as exc: + self._send_json_error( + request_id, + HTTPStatus.BAD_REQUEST, + HEADER_MISMATCH, + f"Header mismatch: {exc}", + ) + return + state = ConnectionState() + response_headers: dict[str, str] = {} + else: + state, response_headers = self._legacy_state(message) + if state is None: + return + + plan = self.server.fixture.handle(message, state) + if plan.response is None: + self._send_bytes( + plan.http_status, + "application/json", + b"", + extra_headers=response_headers, + ) + return + if plan.force_sse: + messages = [*plan.notifications, plan.response] + body = _encode_sse_messages(messages, profile=self.server.fixture.profile) + self._send_bytes( + plan.http_status, + "text/event-stream", + body, + extra_headers={ + **response_headers, + "Cache-Control": "no-cache", + "X-Accel-Buffering": "no", + }, + ) + return + body = json.dumps( + plan.response, ensure_ascii=False, separators=(",", ":") + ).encode() + self._send_bytes( + plan.http_status, + "application/json", + body, + extra_headers=response_headers, + ) + + def _legacy_state( + self, + message: dict[str, object], + ) -> tuple[ConnectionState | None, dict[str, str]]: + method = message.get("method") + if method == "initialize": + state = ConnectionState() + session_id = uuid.uuid4().hex + with self.server.sessions_lock: + self.server.sessions[session_id] = state + return state, {"Mcp-Session-Id": session_id} + if method == "server/discover": + return ConnectionState(), {} + session_id = self.headers.get("Mcp-Session-Id") + with self.server.sessions_lock: + state = ( + self.server.sessions.get(session_id) if session_id is not None else None + ) + if state is None: + self._send_json_error( + message.get("id"), + HTTPStatus.BAD_REQUEST, + INVALID_REQUEST, + "Missing or unknown Mcp-Session-Id", + ) + return None, {} + return state, {"Mcp-Session-Id": session_id} + + def _origin_is_allowed(self) -> bool: + origin = self.headers.get("Origin") + if origin is None: + return True + if origin in self.server.allowed_origins: + return True + parsed = urlsplit(origin) + return parsed.scheme in ("http", "https") and parsed.hostname in ( + "localhost", + "127.0.0.1", + "::1", + ) + + def _send_json_error( + self, + request_id: object, + status: int, + code: int, + message: str, + ) -> None: + body = json.dumps( + _error_response(request_id, code, message), separators=(",", ":") + ).encode() + self._send_bytes(status, "application/json", body) + + def _send_bytes( + self, + status: int, + content_type: str, + body: bytes, + *, + extra_headers: Mapping[str, str] | None = None, + ) -> None: + self.send_response(status) + self.send_header("Content-Type", content_type) + self.send_header("Content-Length", str(len(body))) + if extra_headers is not None: + for name, value in extra_headers.items(): + self.send_header(name, value) + self.end_headers() + if body: + self.wfile.write(body) + + def log_message(self, format: str, *args: object) -> None: + if not self.server.log_requests: + return + print( + f"{self.address_string()} - {format % args}", + file=sys.stderr, + flush=True, + ) + + +def make_http_server( + fixture: ProtocolServer, + host: str, + port: int, + *, + endpoint: str = "/mcp", + allowed_origins: Sequence[str] = (), + log_requests: bool = True, +) -> FixtureHTTPServer: + return FixtureHTTPServer( + (host, port), + fixture, + endpoint=endpoint, + allowed_origins=allowed_origins, + log_requests=log_requests, + ) + + +def run_stdio( + fixture: ProtocolServer, + stdin: IO[str], + stdout: IO[str], +) -> None: + state = ConnectionState() + for line in stdin: + try: + message = json.loads(line) + except json.JSONDecodeError: + stdout.write( + json.dumps( + _error_response(None, PARSE_ERROR, "Parse error"), + separators=(",", ":"), + ) + + "\n" + ) + stdout.flush() + continue + plan = fixture.handle(message, state) + for notification in plan.notifications: + stdout.write( + json.dumps(notification, ensure_ascii=False, separators=(",", ":")) + + "\n" + ) + if plan.response is not None: + stdout.write( + json.dumps(plan.response, ensure_ascii=False, separators=(",", ":")) + + "\n" + ) + stdout.flush() + + +def _parse_args(argv: Sequence[str] | None) -> argparse.Namespace: + parser = argparse.ArgumentParser( + description=( + "Run a deterministic MCP server for testing shipping 2025-06-18, " + "2025-11-25 compatibility, or 2026-07-28 draft compliance." + ) + ) + parser.add_argument( + "--mode", + choices=(SHIPPING_LEGACY_VERSION, LEGACY_VERSION, MODERN_VERSION), + default=MODERN_VERSION, + ) + parser.add_argument( + "--transport", + choices=("stdio", "http"), + default="stdio", + ) + parser.add_argument( + "--profile", + choices=FIXTURE_PROFILES, + default=DEFAULT_PROFILE, + help="Select an optional deterministic reviewer-regression fixture profile.", + ) + parser.add_argument("--host", default="127.0.0.1") + parser.add_argument("--port", type=int, default=8765) + parser.add_argument("--endpoint", default="/mcp") + parser.add_argument( + "--allowed-origin", + action="append", + default=[], + help="Additional exact Origin value accepted by the HTTP transport.", + ) + return parser.parse_args(argv) + + +def main(argv: Sequence[str] | None = None) -> int: + args = _parse_args(argv) + fixture = ProtocolServer(args.mode, profile=args.profile) + if args.transport == "stdio": + print( + f"{SERVER_NAME} starting stdio mode={args.mode}", + file=sys.stderr, + flush=True, + ) + run_stdio(fixture, sys.stdin, sys.stdout) + return 0 + + server = make_http_server( + fixture, + args.host, + args.port, + endpoint=args.endpoint, + allowed_origins=args.allowed_origin, + ) + print( + f"{SERVER_NAME} listening on http://{args.host}:{args.port}{args.endpoint} " + f"mode={args.mode}", + file=sys.stderr, + flush=True, + ) + try: + server.serve_forever() + except KeyboardInterrupt: + pass + finally: + server.server_close() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/mcp_conformance/test_codex_compliance.py b/scripts/mcp_conformance/test_codex_compliance.py new file mode 100644 index 0000000000..e97374d515 --- /dev/null +++ b/scripts/mcp_conformance/test_codex_compliance.py @@ -0,0 +1,1202 @@ +import os +import sys +import json +from copy import deepcopy +from pathlib import Path + +import pytest +from run_codex_compliance import ( + COMPACT_REGRESSION_BASELINE_KIND, + LEGACY_VERSION, + MODERN_VERSION, + OFFICIAL_CONFORMANCE_GIT_REF, + OFFICIAL_CONFORMANCE_REPOSITORY, + REPORT_SCHEMA_VERSION, + REQUIRED_REGRESSION_MODES, + RESOURCE_URIS, + SERVER_NAME, + SERVER_VERSION, + SHIPPING_LEGACY_VERSION, + TEST_SERVER_NAME, + CaseResult, + CheckResult, + OfficialScenarioResult, + _build_test_matrix, + _compact_regression_baseline, + _evaluate_regression_gate, + _isolated_environment, + _parse_args, + _registration_command, + _run_official_case, + _scenario_ratio, + _summarize_modes, + _validate_inventory, + _validate_registration, + _write_compact_regression_baseline, + _write_json_file, + main, + scenarios_for_mode, +) + + +def test_registration_commands_select_stdio_then_http_shapes() -> None: + codex = Path("/opt/codex") + server = Path("/src/server.py") + + stdio = _registration_command( + codex, + server, + transport="stdio", + mode=MODERN_VERSION, + http_url=None, + ) + http = _registration_command( + codex, + server, + transport="http", + mode=MODERN_VERSION, + http_url="http://127.0.0.1:8765/mcp", + ) + + assert stdio == [ + "/opt/codex", + "mcp", + "add", + TEST_SERVER_NAME, + "--env", + f"CODEX_MCP_PROTOCOL_VERSION={MODERN_VERSION}", + "--", + sys.executable, + "/src/server.py", + "--mode", + MODERN_VERSION, + "--transport", + "stdio", + ] + assert http == [ + "/opt/codex", + "mcp", + "add", + TEST_SERVER_NAME, + "--url", + "http://127.0.0.1:8765/mcp", + ] + + +def test_registration_validation_checks_transport_and_mode() -> None: + valid, _ = _validate_registration( + { + "name": TEST_SERVER_NAME, + "enabled": True, + "transport": { + "type": "stdio", + "command": sys.executable, + "args": [ + "/src/server.py", + "--mode", + LEGACY_VERSION, + "--transport", + "stdio", + ], + }, + }, + transport="stdio", + mode=LEGACY_VERSION, + http_url=None, + ) + assert valid + + valid, _ = _validate_registration( + { + "name": TEST_SERVER_NAME, + "enabled": True, + "transport": { + "type": "stdio", + "command": sys.executable, + "args": [ + "/src/server.py", + "--mode", + MODERN_VERSION, + "--transport", + "stdio", + ], + "env": {"CODEX_MCP_PROTOCOL_VERSION": MODERN_VERSION}, + }, + }, + transport="stdio", + mode=MODERN_VERSION, + http_url=None, + ) + assert valid + + valid, detail = _validate_registration( + { + "name": TEST_SERVER_NAME, + "enabled": True, + "transport": { + "type": "stdio", + "command": sys.executable, + "args": [ + "/src/server.py", + "--mode", + MODERN_VERSION, + "--transport", + "stdio", + ], + }, + }, + transport="stdio", + mode=MODERN_VERSION, + http_url=None, + ) + assert not valid + assert "protocol opt-in" in detail + + valid, detail = _validate_registration( + { + "name": TEST_SERVER_NAME, + "enabled": True, + "transport": { + "type": "streamable_http", + "url": "http://127.0.0.1:2/mcp", + }, + }, + transport="http", + mode=MODERN_VERSION, + http_url="http://127.0.0.1:1/mcp", + ) + assert not valid + assert "unexpected HTTP transport" in detail + + +def test_inventory_validation_requires_modern_tool_and_all_pages() -> None: + inventory = { + "data": [ + { + "name": TEST_SERVER_NAME, + "serverInfo": { + "name": SERVER_NAME, + "version": SERVER_VERSION, + }, + "tools": { + name: {"name": name} + for name in ( + "echo", + "client_metadata", + "progress", + "request_input", + ) + }, + "resources": [{"uri": uri} for uri in RESOURCE_URIS], + } + ] + } + + valid, _ = _validate_inventory(inventory, mode=MODERN_VERSION) + assert valid + + inventory["data"][0]["tools"].pop("request_input") + inventory["data"][0]["resources"].pop() + valid, detail = _validate_inventory(inventory, mode=MODERN_VERSION) + assert not valid + assert "request_input" in detail + assert RESOURCE_URIS[1] in detail + + +def test_isolated_environment_drops_model_credentials( + tmp_path: Path, + monkeypatch, +) -> None: + monkeypatch.setenv("CODEX_API_KEY", "secret") + monkeypatch.setenv("CODEX_ACCESS_TOKEN", "secret") + monkeypatch.setenv("OPENAI_API_KEY", "secret") + + env = _isolated_environment(tmp_path) + + assert env["CODEX_HOME"] == str(tmp_path) + assert "CODEX_API_KEY" not in env + assert "CODEX_ACCESS_TOKEN" not in env + assert "OPENAI_API_KEY" not in env + assert env.get("PATH") == os.environ.get("PATH") + + +def test_mode_summaries_separate_official_and_supplemental_percentages() -> None: + legacy = CaseResult( + transport="stdio", + mode=LEGACY_VERSION, + success=True, + checks=[ + CheckResult("mcp_add", True, "ok"), + CheckResult("inventory", True, "ok"), + CheckResult("echo_tool", True, "ok"), + ], + ) + modern = CaseResult( + transport="http", + mode=MODERN_VERSION, + success=False, + checks=[ + CheckResult("mcp_add", True, "ok"), + CheckResult("modern_feature_enablement", True, "ok"), + CheckResult("inventory", False, "failed"), + CheckResult("http_header_mirroring", False, "failed"), + CheckResult( + "official/example/pass", + True, + "ok", + source="official", + category="non-auth", + scenario="example-pass", + ), + CheckResult( + "official/example/skipped", + True, + "not exercised", + status="SKIP", + source="official", + category="auth", + scenario="example-skipped", + ), + ], + ) + + summaries = _summarize_modes([legacy, modern]) + + assert summaries == [ + { + "mode": LEGACY_VERSION, + "checks": {"passed": 3, "total": 3, "percentage": 100.0}, + "officialChecks": {"passed": 0, "total": 0, "percentage": 0.0}, + "officialNonAuthChecks": { + "passed": 0, + "total": 0, + "percentage": 0.0, + }, + "officialAuthChecks": { + "passed": 0, + "total": 0, + "percentage": 0.0, + }, + "officialNonAuthScenarios": { + "passed": 0, + "total": 0, + "percentage": 0.0, + }, + "officialAuthScenarios": { + "passed": 0, + "total": 0, + "percentage": 0.0, + }, + "harnessChecks": {"passed": 0, "total": 0, "percentage": 0.0}, + "supplementalChecks": { + "passed": 3, + "total": 3, + "percentage": 100.0, + }, + "transportCases": {"passed": 1, "total": 1, "percentage": 100.0}, + }, + { + "mode": MODERN_VERSION, + "checks": {"passed": 3, "total": 5, "percentage": 60.0}, + "officialChecks": { + "passed": 1, + "total": 1, + "percentage": 100.0, + }, + "officialNonAuthChecks": { + "passed": 1, + "total": 1, + "percentage": 100.0, + }, + "officialAuthChecks": { + "passed": 0, + "total": 0, + "percentage": 0.0, + }, + "officialNonAuthScenarios": { + "passed": 1, + "total": 1, + "percentage": 100.0, + }, + "officialAuthScenarios": { + "passed": 1, + "total": 1, + "percentage": 100.0, + }, + "harnessChecks": {"passed": 0, "total": 0, "percentage": 0.0}, + "supplementalChecks": { + "passed": 2, + "total": 4, + "percentage": 50.0, + }, + "transportCases": {"passed": 0, "total": 1, "percentage": 0.0}, + }, + ] + + +def test_matrix_reports_pass_fail_and_not_applicable() -> None: + legacy = CaseResult( + transport="stdio", + mode=LEGACY_VERSION, + checks=[ + CheckResult("inventory", True, "ok"), + CheckResult("echo_tool", True, "ok"), + ], + ) + modern = CaseResult( + transport="http", + mode=MODERN_VERSION, + checks=[ + CheckResult("inventory", False, "failed"), + CheckResult("http_header_mirroring", False, "failed"), + ], + ) + + matrix = _build_test_matrix([legacy, modern]) + + assert matrix["columns"] == [ + { + "key": f"stdio:{LEGACY_VERSION}", + "transport": "stdio", + "mode": LEGACY_VERSION, + }, + { + "key": f"http:{MODERN_VERSION}", + "transport": "http", + "mode": MODERN_VERSION, + }, + ] + assert matrix["rows"] == [ + { + "test": "inventory", + "results": { + f"stdio:{LEGACY_VERSION}": "PASS", + f"http:{MODERN_VERSION}": "FAIL", + }, + }, + { + "test": "echo_tool", + "results": { + f"stdio:{LEGACY_VERSION}": "PASS", + f"http:{MODERN_VERSION}": "N/A", + }, + }, + { + "test": "http_header_mirroring", + "results": { + f"stdio:{LEGACY_VERSION}": "N/A", + f"http:{MODERN_VERSION}": "FAIL", + }, + }, + ] + + +def test_scenario_ratio_includes_adapter_failures() -> None: + checks = [ + CheckResult( + "official/auth/example/assertion", + True, + "ok", + source="official", + scenario="auth/example", + category="auth", + ), + CheckResult( + "harness/auth/example/codex-adapter", + False, + "adapter failed", + source="harness", + scenario="auth/example", + category="auth", + ), + CheckResult( + "official/auth/other/assertion", + True, + "ok", + source="official", + scenario="auth/other", + category="auth", + ), + ] + + assert _scenario_ratio(checks, category="auth") == { + "passed": 1, + "total": 2, + "percentage": 50.0, + } + + +@pytest.mark.parametrize("mode", [LEGACY_VERSION, MODERN_VERSION]) +@pytest.mark.parametrize( + ("adapter_success", "expected_status"), + [(False, "FAIL"), (True, "PASS")], +) +def test_official_case_preserves_adapter_check_identity_on_failure_and_success( + mode: str, + adapter_success: bool, + expected_status: str, + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + scenario = "auth/pre-registration" + result = OfficialScenarioResult( + scenario=scenario, + success=adapter_success, + adapter_success=adapter_success, + adapter_detail="adapter succeeded" if adapter_success else "adapter failed", + ) + monkeypatch.setattr( + "run_codex_compliance.run_official_mode", + lambda **_kwargs: [result], + ) + + case = _run_official_case( + Path("/opt/codex"), + Path("/src/codex_conformance_adapter.py"), + conformance_command=["conformance"], + mode=mode, + scenarios=[scenario], + case_home=tmp_path / mode, + timeout_seconds=1.0, + enable_modern_feature=True, + ) + + assert case.checks == [ + CheckResult( + name=f"harness/{scenario}/codex-adapter", + success=adapter_success, + detail=result.adapter_detail, + status=expected_status, + source="harness", + scenario=scenario, + category="auth", + ) + ] + assert case.success is adapter_success + + +@pytest.mark.parametrize("mode", [LEGACY_VERSION, MODERN_VERSION]) +def test_official_case_preserves_independent_runner_failure_with_successful_adapter( + mode: str, + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + scenario = "auth/pre-registration" + result = OfficialScenarioResult( + scenario=scenario, + success=False, + adapter_success=True, + adapter_detail="adapter succeeded", + runner_detail="official runner failed", + ) + monkeypatch.setattr( + "run_codex_compliance.run_official_mode", + lambda **_kwargs: [result], + ) + + case = _run_official_case( + Path("/opt/codex"), + Path("/src/codex_conformance_adapter.py"), + conformance_command=["conformance"], + mode=mode, + scenarios=[scenario], + case_home=tmp_path / mode, + timeout_seconds=1.0, + enable_modern_feature=True, + ) + + assert case.checks == [ + CheckResult( + name=f"harness/{scenario}/codex-adapter", + success=True, + detail="adapter succeeded", + status="PASS", + source="harness", + scenario=scenario, + category="auth", + ), + CheckResult( + name=f"harness/{scenario}/official-runner", + success=False, + detail="official runner failed", + status="FAIL", + source="harness", + scenario=scenario, + category="auth", + ), + ] + assert case.success is False + assert case.diagnostics == f"{scenario}:\nofficial runner failed" + + +def _regression_report() -> dict[str, object]: + cases: list[dict[str, object]] = [] + known_modern = { + "auth/metadata-var3": "authorization-server-metadata", + "auth/scope-from-www-authenticate": "scope-from-www-authenticate", + "auth/scope-step-up": "scope-step-up-initial", + "auth/pre-registration": "pre-registration-auth", + } + scenarios = { + mode: list(scenarios_for_mode(mode, include_auth=True)) + for mode in REQUIRED_REGRESSION_MODES + } + + for mode in REQUIRED_REGRESSION_MODES: + cases.append( + { + "mode": mode, + "transport": "stdio", + "success": True, + "checks": [ + { + "name": "mcp_add", + "success": True, + "status": "PASS", + "source": "supplemental", + "scenario": None, + "check_id": None, + } + ], + } + ) + official_checks: list[dict[str, object]] = [] + for scenario in scenarios[mode]: + if mode == MODERN_VERSION and scenario == "request-metadata": + official_checks.append( + { + "name": "harness/request-metadata/official-runner", + "success": False, + "status": "FAIL", + "source": "harness", + "scenario": scenario, + "check_id": None, + } + ) + continue + + known = mode == MODERN_VERSION and scenario in known_modern + official_checks.append( + { + "name": f"official/{scenario}/assertion", + "success": not known, + "status": "FAIL" if known else "PASS", + "source": "official", + "scenario": scenario, + "check_id": ( + known_modern[scenario] if known else f"assertion-{scenario}" + ), + } + ) + if known: + official_checks.append( + { + "name": f"harness/{scenario}/codex-adapter", + "success": False, + "status": "FAIL", + "source": "harness", + "scenario": scenario, + "check_id": None, + } + ) + cases.append( + { + "mode": mode, + "transport": "official-http", + "success": mode == SHIPPING_LEGACY_VERSION, + "checks": official_checks, + } + ) + + return { + "schemaVersion": REPORT_SCHEMA_VERSION, + "success": False, + "versionCheck": {"success": True}, + "modernFeatureEnablement": True, + "automaticAuthRequired": False, + "officialConformance": { + "repository": OFFICIAL_CONFORMANCE_REPOSITORY, + "gitRef": OFFICIAL_CONFORMANCE_GIT_REF, + "authenticationIncluded": True, + "scenarios": scenarios, + }, + "cases": cases, + } + + +def _regression_case( + report: dict[str, object], + mode: str, + transport: str, +) -> dict[str, object]: + cases = report["cases"] + assert isinstance(cases, list) + return next( + case + for case in cases + if isinstance(case, dict) + and case["mode"] == mode + and case["transport"] == transport + ) + + +def _regression_scenario_check( + report: dict[str, object], + *, + mode: str = MODERN_VERSION, + scenario: str, + source: str = "official", +) -> dict[str, object]: + checks = _regression_case(report, mode, "official-http")["checks"] + assert isinstance(checks, list) + return next( + check + for check in checks + if isinstance(check, dict) + and check["scenario"] == scenario + and check["source"] == source + ) + + +def test_regression_gate_preserves_truthful_known_modern_failures() -> None: + baseline = _regression_report() + candidate = deepcopy(baseline) + + gate = _evaluate_regression_gate(candidate, baseline) + + assert candidate["success"] is False + assert gate["success"] is True + assert gate["requiredModes"] == [ + SHIPPING_LEGACY_VERSION, + LEGACY_VERSION, + MODERN_VERSION, + ] + assert len(gate["knownFailures"]) == 9 + assert gate["newFailures"] == [] + assert gate["missingChecks"] == [] + assert gate["configurationErrors"] == [] + + +def test_regression_gate_rejects_a_new_modern_failure() -> None: + baseline = _regression_report() + candidate = deepcopy(baseline) + check = _regression_scenario_check(candidate, scenario="tools_call") + check["success"] = False + check["status"] = "FAIL" + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert any(item["scenario"] == "tools_call" for item in gate["newFailures"]) + + +def test_regression_gate_rejects_a_new_intermediate_oauth_failure() -> None: + baseline = _regression_report() + candidate = deepcopy(baseline) + check = _regression_scenario_check( + candidate, + mode=LEGACY_VERSION, + scenario="auth/metadata-default", + ) + check["success"] = False + check["status"] = "FAIL" + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert any( + item["mode"] == LEGACY_VERSION and item["scenario"] == "auth/metadata-default" + for item in gate["newFailures"] + ) + + +def test_regression_gate_rejects_new_failures_inside_a_known_failing_scenario() -> None: + baseline = _regression_report() + candidate = deepcopy(baseline) + checks = _regression_case(candidate, MODERN_VERSION, "official-http")["checks"] + assert isinstance(checks, list) + checks.append( + { + "name": "official/auth/metadata-var3/new assertion", + "success": False, + "status": "FAIL", + "source": "official", + "scenario": "auth/metadata-var3", + "check_id": "previously-passing-metadata-assertion", + } + ) + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert any( + item["check_id"] == "previously-passing-metadata-assertion" + for item in gate["newFailures"] + ) + + +def test_regression_gate_accepts_fixed_known_failures() -> None: + baseline = _regression_report() + candidate = deepcopy(baseline) + check = _regression_scenario_check(candidate, scenario="auth/metadata-var3") + check["success"] = True + check["status"] = "PASS" + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is True + assert any( + item["check_id"] == "authorization-server-metadata" + for item in gate["fixedChecks"] + ) + + +def test_regression_gate_accepts_fixed_oauth_official_and_adapter_checks() -> None: + baseline = _regression_report() + candidate = deepcopy(baseline) + scenario = "auth/pre-registration" + for source in ("official", "harness"): + check = _regression_scenario_check( + candidate, + scenario=scenario, + source=source, + ) + check["success"] = True + check["status"] = "PASS" + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is True + assert gate["configurationErrors"] == [] + assert gate["newFailures"] == [] + assert gate["missingChecks"] == [] + assert gate["fixedChecks"] == [ + { + "mode": MODERN_VERSION, + "transport": "official-http", + "source": "harness", + "scenario": scenario, + "check_id": f"harness/{scenario}/codex-adapter", + }, + { + "mode": MODERN_VERSION, + "transport": "official-http", + "source": "official", + "scenario": scenario, + "check_id": "pre-registration-auth", + }, + ] + + +def test_regression_gate_rejects_a_missing_fixed_oauth_adapter_check() -> None: + baseline = _regression_report() + candidate = deepcopy(baseline) + scenario = "auth/pre-registration" + check = _regression_scenario_check(candidate, scenario=scenario) + check["success"] = True + check["status"] = "PASS" + checks = _regression_case(candidate, MODERN_VERSION, "official-http")["checks"] + assert isinstance(checks, list) + checks[:] = [ + item + for item in checks + if item.get("scenario") != scenario or item.get("source") != "harness" + ] + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert gate["configurationErrors"] == [] + assert gate["newFailures"] == [] + assert gate["missingChecks"] == [ + { + "mode": MODERN_VERSION, + "transport": "official-http", + "source": "harness", + "scenario": scenario, + "check_id": f"harness/{scenario}/codex-adapter", + } + ] + assert gate["fixedChecks"] == [ + { + "mode": MODERN_VERSION, + "transport": "official-http", + "source": "official", + "scenario": scenario, + "check_id": "pre-registration-auth", + } + ] + + +def test_regression_gate_rejects_a_skipped_previous_failure() -> None: + baseline = _regression_report() + candidate = deepcopy(baseline) + check = _regression_scenario_check(candidate, scenario="auth/metadata-var3") + check["success"] = True + check["status"] = "SKIP" + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert any( + item["check_id"] == "authorization-server-metadata" + for item in gate["missingChecks"] + ) + + +def test_regression_gate_rejects_a_missing_previously_passing_check() -> None: + baseline = _regression_report() + candidate = deepcopy(baseline) + checks = _regression_case(candidate, MODERN_VERSION, "official-http")["checks"] + assert isinstance(checks, list) + checks[:] = [check for check in checks if check.get("scenario") != "tools_call"] + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert any(item["scenario"] == "tools_call" for item in gate["missingChecks"]) + assert any("did not observe" in item for item in gate["configurationErrors"]) + + +@pytest.mark.parametrize("mode", REQUIRED_REGRESSION_MODES) +@pytest.mark.parametrize("transport", ["stdio", "official-http"]) +def test_regression_gate_requires_every_version_and_both_transports( + mode: str, + transport: str, +) -> None: + baseline = _regression_report() + candidate = deepcopy(baseline) + cases = candidate["cases"] + assert isinstance(cases, list) + cases[:] = [ + case + for case in cases + if case.get("mode") != mode or case.get("transport") != transport + ] + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert any( + f"missing the required {transport} case for {mode}" in item + for item in gate["configurationErrors"] + ) + + +def test_regression_gate_requires_the_complete_modern_authenticated_catalog() -> None: + baseline = _regression_report() + candidate = deepcopy(baseline) + official = candidate["officialConformance"] + assert isinstance(official, dict) + scenarios = official["scenarios"] + assert isinstance(scenarios, dict) + scenarios[MODERN_VERSION] = [ + scenario + for scenario in scenarios[MODERN_VERSION] + if scenario != "auth/pre-registration" + ] + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert any("complete authenticated" in item for item in gate["configurationErrors"]) + + +def test_regression_gate_requires_modern_feature_enablement() -> None: + baseline = _regression_report() + candidate = deepcopy(baseline) + candidate["modernFeatureEnablement"] = False + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert any("modern MCP feature" in item for item in gate["configurationErrors"]) + + +def test_regression_gate_requires_the_same_pinned_upstream_suite() -> None: + baseline = _regression_report() + candidate = deepcopy(baseline) + official = candidate["officialConformance"] + assert isinstance(official, dict) + official["gitRef"] = "0" * 40 + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert any("pinned upstream" in item for item in gate["configurationErrors"]) + + +def test_regression_gate_requires_oauth_scenarios() -> None: + baseline = _regression_report() + candidate = deepcopy(baseline) + official = candidate["officialConformance"] + assert isinstance(official, dict) + official["authenticationIncluded"] = False + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert any("required OAuth" in item for item in gate["configurationErrors"]) + + +def test_regression_gate_rejects_different_production_oauth_policies() -> None: + baseline = _regression_report() + candidate = deepcopy(baseline) + candidate["automaticAuthRequired"] = True + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert any( + "different production OAuth" in item for item in gate["configurationErrors"] + ) + + +def test_regression_gate_rejects_unsupported_report_schema() -> None: + baseline = _regression_report() + baseline["schemaVersion"] = REPORT_SCHEMA_VERSION + 1 + + gate = _evaluate_regression_gate(_regression_report(), baseline) + + assert gate["success"] is False + assert any("report schema" in item for item in gate["configurationErrors"]) + + +def test_regression_gate_rejects_any_shipping_failure() -> None: + baseline = _regression_report() + candidate = deepcopy(baseline) + check = _regression_scenario_check( + candidate, + mode=SHIPPING_LEGACY_VERSION, + scenario="initialize", + ) + check["success"] = False + check["status"] = "FAIL" + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert any( + "shipping MCP check failed" in item for item in gate["configurationErrors"] + ) + + +def test_regression_gate_ignores_retry_observation_renumbering() -> None: + baseline = _regression_report() + check = _regression_scenario_check(baseline, scenario="tools_call") + check["name"] = "official/tools_call/authorization metadata#4" + candidate = deepcopy(baseline) + candidate_check = _regression_scenario_check(candidate, scenario="tools_call") + candidate_check["name"] = "official/tools_call/authorization metadata#57" + checks = _regression_case(candidate, MODERN_VERSION, "official-http")["checks"] + assert isinstance(checks, list) + repeated = deepcopy(candidate_check) + repeated["name"] = "official/tools_call/authorization metadata#58" + checks.append(repeated) + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is True + assert gate["newFailures"] == [] + assert gate["missingChecks"] == [] + + +def test_regression_cli_accepts_a_baseline_report() -> None: + args = _parse_args( + ["/opt/codex", "--baseline-report", "/tmp/codex-mcp-baseline.json"] + ) + + assert args.baseline_report == Path("/tmp/codex-mcp-baseline.json") + assert args.mode == "all" + assert args.auth is True + assert args.enable_modern_feature is True + + +def test_regression_gate_accepts_a_compact_baseline() -> None: + report = _regression_report() + + compact = _compact_regression_baseline(report) + gate = _evaluate_regression_gate(deepcopy(report), compact) + + assert compact["baselineKind"] == COMPACT_REGRESSION_BASELINE_KIND + assert compact["requiredModes"] == [ + SHIPPING_LEGACY_VERSION, + LEGACY_VERSION, + MODERN_VERSION, + ] + assert compact["transports"] == ["stdio", "official-http"] + assert "cases" not in compact + assert gate["success"] is True + assert len(gate["knownFailures"]) == 9 + + +def test_regression_gate_accepts_the_real_in_memory_tuple_scenario_catalog() -> None: + baseline = _compact_regression_baseline(_regression_report()) + candidate = _regression_report() + official = candidate["officialConformance"] + assert isinstance(official, dict) + scenarios = official["scenarios"] + assert isinstance(scenarios, dict) + official["scenarios"] = { + mode: tuple(selected) for mode, selected in scenarios.items() + } + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is True + assert gate["configurationErrors"] == [] + + +def test_regression_gate_rejects_a_new_failure_against_a_compact_baseline() -> None: + baseline = _compact_regression_baseline(_regression_report()) + candidate = _regression_report() + check = _regression_scenario_check(candidate, scenario="tools_call") + check["success"] = False + check["status"] = "FAIL" + + gate = _evaluate_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert any(item["scenario"] == "tools_call" for item in gate["newFailures"]) + + +def test_regression_gate_rejects_malformed_compact_identities() -> None: + compact = _compact_regression_baseline(_regression_report()) + checks = compact["checks"] + assert isinstance(checks, dict) + passing = checks["passing"] + assert isinstance(passing, list) + passing[0]["check_id"] = "" + + gate = _evaluate_regression_gate(_regression_report(), compact) + + assert gate["success"] is False + assert any( + "malformed passing baseline identity" in e for e in gate["configurationErrors"] + ) + + +def test_regression_gate_rejects_an_unknown_compact_baseline_format() -> None: + compact = _compact_regression_baseline(_regression_report()) + compact["baselineKind"] = "unrecognized-baseline" + + gate = _evaluate_regression_gate(_regression_report(), compact) + + assert gate["success"] is False + assert any("unsupported compact baseline" in e for e in gate["configurationErrors"]) + + +def test_json_artifacts_are_sorted_human_readable_and_newline_terminated( + tmp_path: Path, +) -> None: + path = tmp_path / "nested" / "report.json" + payload = {"z": {"z": "café", "a": [2, 1]}, "a": True} + + _write_json_file(path, payload) + + assert path.read_text(encoding="utf-8") == ( + "{\n" + ' "a": true,\n' + ' "z": {\n' + ' "a": [\n' + " 2,\n" + " 1\n" + " ],\n" + ' "z": "café"\n' + " }\n" + "}\n" + ) + assert json.loads(path.read_text(encoding="utf-8")) == payload + + +def test_committed_regression_baseline_is_sorted_human_readable_json() -> None: + baseline = Path(__file__).with_name("regression-baseline-v1.json") + content = baseline.read_text(encoding="utf-8") + + assert content == ( + json.dumps(json.loads(content), ensure_ascii=False, indent=2, sort_keys=True) + + "\n" + ) + + +def test_compact_regression_baselines_are_deterministic(tmp_path: Path) -> None: + first = tmp_path / "first.json" + second = tmp_path / "second.json" + + _write_compact_regression_baseline(_regression_report(), first) + _write_compact_regression_baseline(_regression_report(), second) + + assert first.read_bytes() == second.read_bytes() + content = first.read_text(encoding="utf-8") + assert content == ( + json.dumps(json.loads(content), ensure_ascii=False, indent=2, sort_keys=True) + + "\n" + ) + assert json.loads(content)["baselineKind"] == (COMPACT_REGRESSION_BASELINE_KIND) + + +def test_regression_cli_extracts_compact_baseline_without_starting_conformance( + tmp_path: Path, +) -> None: + codex = tmp_path / "codex" + codex.write_text("#!/bin/sh\nexit 99\n", encoding="utf-8") + codex.chmod(0o755) + source = tmp_path / "full-report.json" + source.write_text(json.dumps(_regression_report()), encoding="utf-8") + extracted = tmp_path / "compact.json" + + assert ( + main( + [ + str(codex), + "--baseline-report", + str(source), + "--extract-baseline", + str(extracted), + ] + ) + == 0 + ) + assert json.loads(extracted.read_text(encoding="utf-8"))["baselineKind"] == ( + COMPACT_REGRESSION_BASELINE_KIND + ) + + +def test_regression_cli_rejects_extraction_without_a_source_report( + tmp_path: Path, + capsys: pytest.CaptureFixture[str], +) -> None: + codex = tmp_path / "codex" + codex.write_text("#!/bin/sh\nexit 0\n", encoding="utf-8") + codex.chmod(0o755) + + assert main([str(codex), "--extract-baseline", str(tmp_path / "compact.json")]) == 2 + assert "requires --baseline-report" in capsys.readouterr().err + + +@pytest.mark.parametrize("payload", ["{", "[]", "null"]) +def test_regression_cli_rejects_malformed_baselines( + payload: str, + tmp_path: Path, + capsys: pytest.CaptureFixture[str], +) -> None: + codex = tmp_path / "codex" + codex.write_text("#!/bin/sh\nexit 0\n", encoding="utf-8") + codex.chmod(0o755) + baseline = tmp_path / "baseline.json" + baseline.write_text(payload, encoding="utf-8") + + assert main([str(codex), "--baseline-report", str(baseline)]) == 2 + assert "baseline report" in capsys.readouterr().err + + +def test_regression_cli_rejects_missing_baseline_before_starting_conformance( + tmp_path: Path, + capsys: pytest.CaptureFixture[str], +) -> None: + codex = tmp_path / "codex" + codex.write_text("#!/bin/sh\nexit 0\n", encoding="utf-8") + codex.chmod(0o755) + + assert main([str(codex), "--baseline-report", str(tmp_path / "missing.json")]) == 2 + assert "cannot read baseline report" in capsys.readouterr().err diff --git a/scripts/mcp_conformance/test_official_conformance.py b/scripts/mcp_conformance/test_official_conformance.py new file mode 100644 index 0000000000..be8a9650ab --- /dev/null +++ b/scripts/mcp_conformance/test_official_conformance.py @@ -0,0 +1,685 @@ +import json +import subprocess +import sys +import urllib.parse +from pathlib import Path + +import codex_conformance_adapter +import official_conformance +import pytest +import run_codex_compliance +from codex_conformance_adapter import ( + CIMD_CLIENT_METADATA_URL, + AdapterFailure, + _exercise_auth_scenario, + _oauth_client_id, + _validate_oauth_secret_not_persisted, + _validated_callback_url, + _write_auth_registration, +) +from official_conformance import ( + LEGACY_VERSION, + MODERN_VERSION, + OFFICIAL_CONFORMANCE_GIT_REF, + SHIPPING_LEGACY_VERSION, + _make_adapter_launcher, + _run_scenario, + _scrub_retained_artifacts, + _terminate_process_group, + default_conformance_command, + redact_sensitive_text, + run_official_mode, + scenarios_for_mode, +) + + +def _authorization_url(redirect_uri: str, state: str = "test-state") -> str: + return "http://127.0.0.1:8765/authorize?" + urllib.parse.urlencode( + { + "client_id": "test-client", + "redirect_uri": redirect_uri, + "state": state, + } + ) + + +def test_auth_adapter_only_accepts_exact_loopback_callback() -> None: + redirect_uri = "http://127.0.0.1:32123/callback" + authorization_url = _authorization_url(redirect_uri) + callback_url = redirect_uri + "?code=test-code&state=test-state" + + assert _validated_callback_url(authorization_url, callback_url) == callback_url + + with pytest.raises(AdapterFailure, match="exact loopback"): + _validated_callback_url( + authorization_url, + "http://127.0.0.1:32124/callback?code=test-code&state=test-state", + ) + with pytest.raises(AdapterFailure, match="preserve OAuth state"): + _validated_callback_url( + authorization_url, + redirect_uri + "?code=test-code&state=wrong-state", + ) + with pytest.raises(AdapterFailure, match="safe loopback"): + _validated_callback_url( + _authorization_url("https://example.com/callback"), + "https://example.com/callback?code=test-code&state=test-state", + ) + + +@pytest.mark.parametrize( + ("query", "message"), + [ + ("code=first&code=second&state=test-state", "exactly one nonempty code"), + ("code=&state=test-state", "exactly one nonempty code"), + ("error=first&error=second&state=test-state", "exactly one nonempty error"), + ("error=&state=test-state", "exactly one nonempty error"), + ("code=test-code&error=access_denied&state=test-state", "exactly one of"), + ("code=test-code&state=test-state&state=test-state", "preserve OAuth state"), + ( + "code=test-code&state=test-state&iss=first&iss=second", + "exactly one nonempty iss", + ), + ("code=test-code&state=test-state&iss=", "exactly one nonempty iss"), + ("state=test-state", "exactly one of"), + ], +) +def test_auth_adapter_rejects_ambiguous_oauth_callback_parameters( + query: str, + message: str, +) -> None: + redirect_uri = "http://127.0.0.1:32123/callback" + + with pytest.raises(AdapterFailure, match=message): + _validated_callback_url( + _authorization_url(redirect_uri), + f"{redirect_uri}?{query}", + ) + + +def test_auth_adapter_allows_a_single_error_callback_for_client_validation() -> None: + redirect_uri = "http://127.0.0.1:32123/callback" + callback_url = redirect_uri + "?error=access_denied&state=test-state&iss=issuer" + + assert ( + _validated_callback_url(_authorization_url(redirect_uri), callback_url) + == callback_url + ) + + +def test_official_runner_is_pinned_instead_of_using_npm_latest() -> None: + command = default_conformance_command() + + assert OFFICIAL_CONFORMANCE_GIT_REF in command[-1] + assert "@latest" not in command[-1] + + +def test_official_adapter_launcher_preserves_strict_production_auth_mode( + tmp_path: Path, +) -> None: + adapter = tmp_path / "show_automatic_auth.py" + adapter.write_text( + 'import os\nprint(os.environ.get("CODEX_CONFORMANCE_REQUIRE_AUTOMATIC_AUTH", ""))\n', + encoding="utf-8", + ) + launcher = _make_adapter_launcher(adapter, require_automatic_auth=True) + + try: + result = subprocess.run( + [str(launcher)], + capture_output=True, + check=True, + text=True, + ) + finally: + launcher.unlink() + launcher.parent.rmdir() + + assert result.stdout.strip() == "1" + + +@pytest.mark.parametrize("require_automatic_auth", [False, True]) +def test_windows_official_adapter_launcher_quotes_checkout_paths( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + require_automatic_auth: bool, +) -> None: + adapter = tmp_path / "checkout with spaces" / "adapter script.py" + adapter.parent.mkdir() + adapter.write_text("raise AssertionError('launcher generation only')\n") + monkeypatch.setattr(official_conformance.sys, "platform", "win32") + + launcher = _make_adapter_launcher( + adapter, + require_automatic_auth=require_automatic_auth, + ) + try: + contents = launcher.read_text(encoding="utf-8") + finally: + launcher.unlink() + launcher.parent.rmdir() + + expected_auth = "1" if require_automatic_auth else "0" + assert launcher.name == "client.cmd" + assert contents.startswith("@echo off\n") + assert ( + f'set "CODEX_CONFORMANCE_REQUIRE_AUTOMATIC_AUTH={expected_auth}"\n' in contents + ) + assert subprocess.list2cmdline([sys.executable, str(adapter)]) + " %*\n" in contents + + +@pytest.mark.parametrize( + ("force", "expected_command"), + [ + (False, ["taskkill", "/T", "/PID", "417"]), + (True, ["taskkill", "/F", "/T", "/PID", "417"]), + ], +) +def test_windows_official_timeout_terminates_the_entire_process_tree( + monkeypatch: pytest.MonkeyPatch, + force: bool, + expected_command: list[str], +) -> None: + observed: list[list[str]] = [] + + class FakeProcess: + pid = 417 + + def terminate(self) -> None: + raise AssertionError("successful taskkill must own tree termination") + + def kill(self) -> None: + raise AssertionError("successful taskkill must own tree termination") + + def fake_run( + command: list[str], + *, + check: bool, + stdout: int, + stderr: int, + ) -> subprocess.CompletedProcess[str]: + observed.append(command) + assert check is False + assert stdout == subprocess.DEVNULL + assert stderr == subprocess.DEVNULL + return subprocess.CompletedProcess(command, 0) + + monkeypatch.setattr(official_conformance.sys, "platform", "win32") + monkeypatch.setattr(official_conformance.subprocess, "run", fake_run) + + _terminate_process_group( + FakeProcess(), # type: ignore[arg-type] - validate the Popen process contract. + force=force, + ) + + assert observed == [expected_command] + + +@pytest.mark.parametrize( + "scenario", + ["auth/scope-step-up", "auth/authorization-server-migration"], +) +def test_strict_auth_does_not_inject_a_second_login( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + scenario: str, +) -> None: + manual_reauthorizations: list[object] = [] + + monkeypatch.setattr( + codex_conformance_adapter, + "_oauth_login", + lambda _client, **_kwargs: (True, None), + ) + monkeypatch.setattr(codex_conformance_adapter, "_reload_mcp", lambda _client: None) + monkeypatch.setattr( + codex_conformance_adapter, "_auth_inventory", lambda _client: {} + ) + + def fail_tool_call(_client: object, _workspace: Path) -> None: + raise AdapterFailure("reauthorization is required") + + monkeypatch.setattr(codex_conformance_adapter, "_auth_tool_call", fail_tool_call) + monkeypatch.setattr( + codex_conformance_adapter, + "_login_reload_and_call", + lambda *args, **kwargs: manual_reauthorizations.append((args, kwargs)), + ) + + with pytest.raises(AdapterFailure, match="reauthorization is required"): + _exercise_auth_scenario( + object(), # type: ignore[arg-type] + scenario=scenario, + workspace=tmp_path, + timeout_seconds=1, + require_automatic_auth=True, + ) + + assert manual_reauthorizations == [] + + +@pytest.mark.parametrize( + "scenario", + ["auth/scope-step-up", "auth/authorization-server-migration"], +) +def test_strict_auth_accepts_product_owned_reauthentication( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + scenario: str, +) -> None: + tool_calls: list[Path] = [] + manual_reauthorizations: list[object] = [] + + monkeypatch.setattr( + codex_conformance_adapter, + "_oauth_login", + lambda _client, **_kwargs: (True, None), + ) + monkeypatch.setattr(codex_conformance_adapter, "_reload_mcp", lambda _client: None) + monkeypatch.setattr( + codex_conformance_adapter, "_auth_inventory", lambda _client: {} + ) + monkeypatch.setattr( + codex_conformance_adapter, + "_auth_tool_call", + lambda _client, workspace: tool_calls.append(workspace), + ) + monkeypatch.setattr( + codex_conformance_adapter, + "_login_reload_and_call", + lambda *args, **kwargs: manual_reauthorizations.append((args, kwargs)), + ) + + detail = _exercise_auth_scenario( + object(), # type: ignore[arg-type] + scenario=scenario, + workspace=tmp_path, + timeout_seconds=1, + require_automatic_auth=True, + ) + + assert "automatically" in detail + assert tool_calls == [tmp_path] + assert manual_reauthorizations == [] + + +def test_strict_auth_does_not_invent_a_production_client_metadata_url() -> None: + assert ( + _oauth_client_id( + "auth/basic-cimd", + {}, + require_production_client_identity=True, + ) + is None + ) + + +@pytest.mark.parametrize("filename", ["config.toml", ".credentials.json"]) +def test_oauth_client_secret_persistence_is_detected_without_disclosing_it( + tmp_path: Path, + filename: str, +) -> None: + secret = "reviewer-confidential-client-secret" + (tmp_path / filename).write_text( + json.dumps({"client_secret": secret}), + encoding="utf-8", + ) + + with pytest.raises(AdapterFailure, match="client secret was persisted") as error: + _validate_oauth_secret_not_persisted(tmp_path, secret) + + assert filename in str(error.value) + assert secret not in str(error.value) + + +def test_oauth_client_secret_environment_reference_is_not_a_persisted_secret( + tmp_path: Path, +) -> None: + (tmp_path / "config.toml").write_text( + 'client_secret_env_var = "MCP_CONFORMANCE_CLIENT_SECRET"\n', + encoding="utf-8", + ) + + _validate_oauth_secret_not_persisted( + tmp_path, "reviewer-confidential-client-secret" + ) + + +def test_auth_adapter_selects_only_scenario_provided_client_ids() -> None: + assert _oauth_client_id("auth/basic-cimd", {}) == CIMD_CLIENT_METADATA_URL + assert ( + _oauth_client_id( + "auth/basic-cimd", + {}, + require_production_client_identity=True, + ) + is None + ) + assert ( + _oauth_client_id( + "auth/pre-registration", + {"client_id": "pre-registered", "client_secret": "do-not-log"}, + ) + == "pre-registered" + ) + assert _oauth_client_id("auth/metadata-default", {}) is None + with pytest.raises(AdapterFailure, match="client_id"): + _oauth_client_id("auth/pre-registration", {}) + + +def test_auth_registration_does_not_persist_context_secret(tmp_path: Path) -> None: + config_path = tmp_path / "config.toml" + config_path.write_text('mcp_oauth_credentials_store = "file"\n', encoding="utf-8") + + _write_auth_registration( + config_path, + server_url="http://127.0.0.1:8765/mcp", + oauth_client_id="pre-registered-client", + oauth_client_secret_env_var="MCP_CONFORMANCE_CLIENT_SECRET", + ) + + config = config_path.read_text(encoding="utf-8") + assert 'url = "http://127.0.0.1:8765/mcp"' in config + assert 'client_id = "pre-registered-client"' in config + assert 'client_secret_env_var = "MCP_CONFORMANCE_CLIENT_SECRET"' in config + assert "do-not-log" not in config + assert "client_secret =" not in config + + +def test_official_diagnostics_redact_oauth_secrets() -> None: + diagnostic = ( + 'With context: {"client_id":"visible","client_secret":"secret-value",' + '"private_key_pem":"private-value"}\n' + "Authorize at http://localhost/authorize?state=state-value&code_challenge=pkce-value\n" + "Authorization: Bearer token-value" + ) + + redacted = redact_sensitive_text(diagnostic) + + assert '"client_id":"visible"' in redacted + assert '"client_secret":"[REDACTED]"' in redacted + assert '"private_key_pem":"[REDACTED]"' in redacted + assert "state=[REDACTED]" in redacted + assert "code_challenge=[REDACTED]" in redacted + assert "Bearer [REDACTED]" in redacted + serialized = redact_sensitive_text(json.dumps({"diagnostic": diagnostic})) + assert json.loads(serialized)["diagnostic"] + for secret in ( + "secret-value", + "private-value", + "state-value", + "pkce-value", + "token-value", + ): + assert secret not in redacted + assert secret not in serialized + + +def test_retained_artifacts_remove_oauth_credential_store(tmp_path: Path) -> None: + codex_home = tmp_path / "codex-home" + codex_home.mkdir() + credentials = codex_home / ".credentials.json" + credentials.write_text('{"client_secret":"do-not-retain"}', encoding="utf-8") + stdout = tmp_path / "stdout.txt" + stdout.write_text("Bearer token-value", encoding="utf-8") + + _scrub_retained_artifacts(tmp_path) + + assert not credentials.exists() + assert stdout.read_text(encoding="utf-8") == "Bearer [REDACTED]" + + +def test_pinned_full_scenarios_are_selected_by_protocol_version() -> None: + shipping_legacy = scenarios_for_mode(SHIPPING_LEGACY_VERSION) + legacy = scenarios_for_mode(LEGACY_VERSION) + modern = scenarios_for_mode(MODERN_VERSION) + + assert shipping_legacy == ( + "initialize", + "tools_call", + "auth/token-endpoint-auth-basic", + "auth/token-endpoint-auth-post", + "auth/token-endpoint-auth-none", + ) + assert scenarios_for_mode(SHIPPING_LEGACY_VERSION, include_auth=False) == ( + "initialize", + "tools_call", + ) + + assert legacy[:4] == ( + "initialize", + "tools_call", + "elicitation-sep1034-client-defaults", + "sse-retry", + ) + assert len(legacy) == 18 + assert len([scenario for scenario in legacy if scenario.startswith("auth/")]) == 14 + assert "auth/pre-registration" in legacy + + assert modern[:7] == ( + "tools_call", + "request-metadata", + "sep-2322-client-request-state", + "http-standard-headers", + "http-custom-headers", + "http-invalid-tool-headers", + "json-schema-ref-no-deref", + ) + assert len(modern) == 32 + assert len([scenario for scenario in modern if scenario.startswith("auth/")]) == 25 + assert "auth/resource-mismatch" in modern + assert "auth/metadata-issuer-mismatch" in modern + + assert scenarios_for_mode(LEGACY_VERSION, include_auth=False) == legacy[:4] + assert scenarios_for_mode(MODERN_VERSION, include_auth=False) == modern[:7] + assert scenarios_for_mode( + MODERN_VERSION, + ["tools_call", "http-custom-headers"], + ) == ("tools_call", "http-custom-headers") + assert len(OFFICIAL_CONFORMANCE_GIT_REF) == 40 + + +@pytest.mark.parametrize( + ("mode", "scenario"), + [ + (SHIPPING_LEGACY_VERSION, "request-metadata"), + (LEGACY_VERSION, "http-standard-headers"), + (MODERN_VERSION, "elicitation-sep1034-client-defaults"), + ], +) +def test_versioned_scenario_selection_rejects_an_incompatible_protocol( + mode: str, scenario: str +) -> None: + with pytest.raises( + ValueError, + match=rf"unavailable for MCP protocol version {mode}.*{scenario}", + ): + scenarios_for_mode(mode, [scenario], include_auth=False) + + +def test_cross_version_auth_subset_remains_available_to_every_mode() -> None: + requested = ( + "auth/token-endpoint-auth-basic", + "auth/resource-mismatch", + "auth/authorization-server-migration", + ) + + assert scenarios_for_mode(SHIPPING_LEGACY_VERSION, requested) == ( + "auth/token-endpoint-auth-basic", + ) + assert scenarios_for_mode(LEGACY_VERSION, requested) == ( + "auth/token-endpoint-auth-basic", + ) + assert scenarios_for_mode(MODERN_VERSION, requested) == requested + + +def test_official_cli_rejects_an_empty_versioned_scenario_run( + monkeypatch: pytest.MonkeyPatch, + capsys: pytest.CaptureFixture[str], +) -> None: + def unexpected_run(*_args: object, **_kwargs: object) -> None: + raise AssertionError("an incompatible selection must not execute conformance") + + monkeypatch.setattr(run_codex_compliance, "run_compliance", unexpected_run) + + exit_code = run_codex_compliance.main( + [ + sys.executable, + "--mode", + SHIPPING_LEGACY_VERSION, + "--transport", + "http", + "--official-scenario", + "request-metadata", + "--no-auth", + "--conformance-cli", + sys.executable, + ] + ) + + assert exit_code == 2 + assert ( + "requested scenarios are unavailable for MCP protocol version 2025-06-18: " + "request-metadata" + ) in capsys.readouterr().err + + +def test_official_driver_reads_checks_and_adapter_report(tmp_path: Path) -> None: + fake_cli = tmp_path / "fake_conformance.py" + fake_cli.write_text( + """ +import json +import os +import pathlib +import sys + +args = sys.argv[1:] +scenario = args[args.index("--scenario") + 1] +output = pathlib.Path(args[args.index("--output-dir") + 1]) +result_dir = output / f"{scenario}-timestamp" +result_dir.mkdir(parents=True) +(result_dir / "checks.json").write_text(json.dumps([ + { + "id": "request-trace", + "name": "RequestTrace", + "description": "not a conformance assertion", + "status": "INFO", + }, + { + "id": "official-check", + "name": "OfficialCheck", + "description": "observed", + "status": "SUCCESS", + }, +])) +pathlib.Path(os.environ["CODEX_CONFORMANCE_ADAPTER_REPORT"]).write_text( + json.dumps({"success": True, "steps": []}) +) +""".lstrip(), + encoding="utf-8", + ) + adapter = tmp_path / "adapter.py" + adapter.write_text("raise AssertionError('fake CLI should not run adapter')\n") + + results = run_official_mode( + conformance_command=[sys.executable, str(fake_cli)], + adapter_script=adapter, + codex_binary=Path("/opt/codex"), + mode=MODERN_VERSION, + scenarios=["tools_call"], + output_dir=tmp_path / "results", + timeout_seconds=1, + base_env={}, + ) + + assert len(results) == 1 + assert results[0].success + assert results[0].adapter_success + assert len(results[0].checks) == 1 + assert results[0].checks[0].check_id == "official-check" + + +def test_official_driver_terminates_timed_out_process_group(tmp_path: Path) -> None: + fake_cli = tmp_path / "hanging_conformance.py" + fake_cli.write_text( + """ +import subprocess +import sys +import time + +subprocess.Popen([sys.executable, "-c", "import time; time.sleep(60)"]) +time.sleep(60) +""".lstrip(), + encoding="utf-8", + ) + adapter = tmp_path / "adapter.py" + adapter.write_text("raise AssertionError('not reached')\n", encoding="utf-8") + launcher = _make_adapter_launcher(adapter) + try: + result = _run_scenario( + conformance_command=[sys.executable, str(fake_cli)], + adapter_launcher=launcher, + codex_binary=Path("/opt/codex"), + mode=MODERN_VERSION, + scenario="tools_call", + output_dir=tmp_path / "results", + timeout_seconds=0.05, + process_grace_seconds=0.05, + base_env={}, + ) + finally: + launcher.unlink() + launcher.parent.rmdir() + + assert not result.success + assert "timed out" in result.runner_detail + + +def test_official_driver_does_not_mislabel_runner_crash_as_adapter_failure( + tmp_path: Path, +) -> None: + fake_cli = tmp_path / "crashing_conformance.py" + fake_cli.write_text( + """ +import os +import subprocess +import sys + +script = ''' +import json +import os +import pathlib +import time + +time.sleep(0.1) +pathlib.Path(os.environ["CODEX_CONFORMANCE_ADAPTER_REPORT"]).write_text( + json.dumps({"success": True, "steps": []}) +) +''' +subprocess.Popen([sys.executable, "-c", script], env=os.environ) +sys.exit(1) +""".lstrip(), + encoding="utf-8", + ) + adapter = tmp_path / "adapter.py" + adapter.write_text("raise AssertionError('not reached')\n", encoding="utf-8") + launcher = _make_adapter_launcher(adapter) + try: + result = _run_scenario( + conformance_command=[sys.executable, str(fake_cli)], + adapter_launcher=launcher, + codex_binary=Path("/opt/codex"), + mode=MODERN_VERSION, + scenario="request-metadata", + output_dir=tmp_path / "results", + timeout_seconds=1, + process_grace_seconds=1, + base_env={}, + ) + finally: + launcher.unlink() + launcher.parent.rmdir() + + assert not result.success + assert result.adapter_success + assert "did not produce exactly one checks.json" in result.runner_detail diff --git a/scripts/mcp_conformance/test_review_regressions.py b/scripts/mcp_conformance/test_review_regressions.py new file mode 100644 index 0000000000..609769a3e5 --- /dev/null +++ b/scripts/mcp_conformance/test_review_regressions.py @@ -0,0 +1,1093 @@ +import json +import sys +import time +from copy import deepcopy +from pathlib import Path +from typing import Callable, Mapping, Sequence + +import pytest +from review_regressions import ( + CATALOG_BOUNDARY_PROFILES, + CATALOG_BOUNDARY_TRANSPORTS, + CATALOG_LIMIT_ERROR, + CATALOG_MAX_PROFILE, + CATALOG_OVER_LIMIT_PROFILE, + LEGACY_ENVIRONMENT_SENTINEL, + LEGACY_VERSION, + MAX_CATALOG_ITEMS, + MISMATCHED_DISCOVERY_ID_PROFILE, + MODERN_VERSION, + NULL_DISCOVERY_ID_PROFILE, + REPEATED_CURSOR_PROFILE, + REVIEW_EXACT_INTEGER, + REVIEW_BASELINE_KIND, + REVIEW_MODES, + REVIEW_PROFILE, + REVIEW_REPORT_SCHEMA_VERSION, + REVIEWER, + SHIPPING_LEGACY_VERSION, + SSE_COMMENT_FLOOD_PROFILE, + SSE_CR_COMMENTS_PROFILE, + TEST_SERVER_NAME, + AppServerError, + CaseResult, + _compact_review_regression_baseline, + _elicitation_schema, + _evaluate_review_regression_gate, + _exact_integer_property, + _mrtr_budget_is_bounded, + _parse_args, + _required_review_cases, + _required_review_checks, + _run_catalog_boundary_checks, + _review_elicitation_content, + _review_inventory_entry, + _review_registration_command, + _tool_input_schema, + _write_review_json, + main, + run_review_regressions, +) + + +def _integer_schema(value: int = REVIEW_EXACT_INTEGER) -> dict[str, object]: + return { + "type": "object", + "properties": { + "value": { + "type": "integer", + "minimum": value, + "maximum": 9_223_372_036_854_775_807, + "default": value, + } + }, + } + + +class _CatalogBoundaryClient: + def __init__( + self, + *, + status: str | None, + tool_count: int, + server_info: dict[str, object] | None, + error: str | None = None, + notification_name: str = TEST_SERVER_NAME, + preceding_statuses: Sequence[str] = (), + ) -> None: + self.events: list[dict[str, object]] = [] + self._status = status + self._error = error + self._notification_name = notification_name + self._preceding_statuses = preceding_statuses + self._entry: dict[str, object] = { + "name": TEST_SERVER_NAME, + "serverInfo": server_info, + "tools": { + f"catalog_boundary_{index:04d}": {} for index in range(tool_count) + }, + } + + def request( + self, method: str, params: Mapping[str, object] | None + ) -> dict[str, object]: + if method == "thread/start": + if self._status is not None: + for status in self._preceding_statuses: + self.events.append( + { + "method": "mcpServer/startupStatus/updated", + "params": { + "name": self._notification_name, + "status": status, + }, + } + ) + notification_params: dict[str, object] = { + "name": self._notification_name, + "status": self._status, + } + if self._error is not None: + notification_params["error"] = self._error + self.events.append( + { + "method": "mcpServer/startupStatus/updated", + "params": notification_params, + } + ) + return {"result": {"thread": {"id": "catalog-boundary-thread"}}} + if method == "mcpServerStatus/list": + return {"result": {"data": [self._entry]}} + raise AssertionError(f"unexpected app-server request: {method}") + + def wait_for_notification( + self, + method: str, + *, + predicate: Callable[[Mapping[str, object]], bool] | None = None, + after_event_index: int = 0, + ) -> dict[str, object]: + for event in self.events[after_event_index:]: + params = event.get("params") + if ( + event.get("method") == method + and isinstance(params, dict) + and (predicate is None or predicate(params)) + ): + return event + raise AppServerError(f"timed out waiting for app-server notification {method}") + + +def test_review_matrix_includes_the_actual_shipping_legacy_protocol() -> None: + assert REVIEW_MODES == ("2025-06-18", "2025-11-25", "2026-07-28") + assert SHIPPING_LEGACY_VERSION == "2025-06-18" + + +def test_exact_integer_schema_does_not_accept_float_rounded_values() -> None: + assert _exact_integer_property(_integer_schema()) + assert not _exact_integer_property(_integer_schema(REVIEW_EXACT_INTEGER - 1)) + + rounded = _integer_schema() + rounded["properties"]["value"]["minimum"] = float(REVIEW_EXACT_INTEGER) # type: ignore[index] + assert not _exact_integer_property(rounded) + + +def test_exact_integer_schema_does_not_accept_boolean_values() -> None: + schema = _integer_schema() + schema["properties"]["value"]["minimum"] = True # type: ignore[index] + + assert not _exact_integer_property(schema) + + +@pytest.mark.parametrize("key", ["requestedSchema", "requested_schema", "schema"]) +def test_extracts_the_actual_elicitation_requested_schema(key: str) -> None: + schema = _integer_schema() + + assert _elicitation_schema({key: schema}) == schema + assert _elicitation_schema({"params": {key: schema}}) == schema + + +def test_elicitation_response_preserves_the_exact_large_integer() -> None: + response = _review_elicitation_content({"requestedSchema": _integer_schema()}) + + assert response == {"value": 9_007_199_254_740_993} + + +@pytest.mark.parametrize( + ("request_count", "elapsed_seconds", "expected"), + [ + (0, 0.1, True), + (64, 1.0, True), + (65, 1.0, False), + (64, 10.1, False), + ], +) +def test_mrtr_budget_requires_both_request_and_time_limits( + request_count: int, + elapsed_seconds: float, + expected: bool, +) -> None: + assert ( + _mrtr_budget_is_bounded( + request_count, + elapsed_seconds=elapsed_seconds, + timeout_seconds=8, + ) + is expected + ) + + +def test_review_inventory_requires_the_registered_server() -> None: + entry = {"name": "mcp_spec_fixture", "tools": {}} + + assert _review_inventory_entry({"data": [entry]}) == entry + assert _review_inventory_entry({"data": [{"name": "another-server"}]}) is None + assert _review_inventory_entry({"data": "invalid"}) is None + + +@pytest.mark.parametrize("schema_key", ["inputSchema", "input_schema"]) +def test_review_inventory_recognizes_both_tool_schema_spellings( + schema_key: str, +) -> None: + schema = _integer_schema() + entry = {"tools": {"review_large_integer": {schema_key: schema}}} + + assert _tool_input_schema(entry, "review_large_integer") == schema + + +def test_shipping_legacy_registration_preserves_the_reserved_protocol_environment() -> ( + None +): + command = _review_registration_command( + Path("/opt/codex"), + Path("/src/server.py"), + transport="stdio", + mode=SHIPPING_LEGACY_VERSION, + profile=REVIEW_PROFILE, + http_url=None, + ) + + assert command == [ + "/opt/codex", + "mcp", + "add", + "mcp_spec_fixture", + "--env", + f"CODEX_MCP_PROTOCOL_VERSION={LEGACY_ENVIRONMENT_SENTINEL}", + "--", + sys.executable, + "/src/server.py", + "--mode", + SHIPPING_LEGACY_VERSION, + "--transport", + "stdio", + "--profile", + REVIEW_PROFILE, + ] + + +def test_modern_registration_explicitly_opts_into_the_modern_protocol() -> None: + command = _review_registration_command( + Path("/opt/codex"), + Path("/src/server.py"), + transport="stdio", + mode=MODERN_VERSION, + profile=REVIEW_PROFILE, + http_url=None, + ) + + assert f"CODEX_MCP_PROTOCOL_VERSION={MODERN_VERSION}" in command + + +def test_http_review_registration_requires_a_real_fixture_url() -> None: + with pytest.raises(ValueError, match="fixture URL"): + _review_registration_command( + Path("/opt/codex"), + Path("/src/server.py"), + transport="http", + mode=MODERN_VERSION, + profile=REVIEW_PROFILE, + http_url=None, + ) + + +def test_review_cli_supports_the_real_shipping_protocol_and_json_report() -> None: + args = _parse_args( + [ + "/opt/codex", + "--mode", + SHIPPING_LEGACY_VERSION, + "--report", + "/tmp/mcp-review-regressions.json", + ] + ) + + assert args.mode == "2025-06-18" + assert args.report == Path("/tmp/mcp-review-regressions.json") + + +def test_review_cli_reports_a_missing_codex_binary( + capsys: pytest.CaptureFixture[str], +) -> None: + assert main(["/path/to/nonexistent-review-codex"]) == 2 + assert "not executable" in capsys.readouterr().err + + +@pytest.mark.parametrize("mode", REVIEW_MODES) +def test_catalog_at_limit_requires_a_ready_server_and_all_tools( + mode: str, tmp_path: Path +) -> None: + case = CaseResult(transport="review-stdio:catalog-max", mode=mode) + client = _CatalogBoundaryClient( + status="ready", + tool_count=MAX_CATALOG_ITEMS, + server_info={"name": "catalog-boundary-server"}, + ) + + _run_catalog_boundary_checks( + case, + client, # type: ignore[arg-type] - exercise the real app-server client contract. + profile=CATALOG_MAX_PROFILE, + workspace=tmp_path, + ) + case.finish(time.monotonic()) + + assert case.success + assert [check.name for check in case.checks] == [ + "review/ephemeral-thread", + "review/catalog-at-limit/startup-ready", + "review/catalog-at-limit/configured-server", + "review/catalog-at-limit/server-info", + "review/catalog-at-limit/tool-count", + ] + + +@pytest.mark.parametrize("mode", REVIEW_MODES) +def test_clean_catalog_over_limit_rejection_is_an_expected_pass( + mode: str, tmp_path: Path +) -> None: + case = CaseResult(transport="review-http:catalog-over-limit", mode=mode) + client = _CatalogBoundaryClient( + status="failed", + tool_count=0, + server_info=None, + error=f"MCP startup failed: {CATALOG_LIMIT_ERROR}", + ) + + _run_catalog_boundary_checks( + case, + client, # type: ignore[arg-type] - exercise the real app-server client contract. + profile=CATALOG_OVER_LIMIT_PROFILE, + workspace=tmp_path, + ) + case.finish(time.monotonic()) + + assert case.success + assert [check.name for check in case.checks] == [ + "review/ephemeral-thread", + "review/catalog-over-limit/startup-failed", + "review/catalog-over-limit/exact-limit-error", + "review/catalog-over-limit/configured-server", + "review/catalog-over-limit/server-info", + "review/catalog-over-limit/tool-count", + ] + + +@pytest.mark.parametrize( + ("profile", "terminal_status", "tool_count"), + [ + (CATALOG_MAX_PROFILE, "ready", MAX_CATALOG_ITEMS), + (CATALOG_OVER_LIMIT_PROFILE, "failed", 0), + ], +) +def test_catalog_boundary_ignores_starting_until_the_real_terminal_notification( + profile: str, + terminal_status: str, + tool_count: int, + tmp_path: Path, +) -> None: + at_limit = profile == CATALOG_MAX_PROFILE + case = CaseResult(transport=f"review-http:{profile}", mode=MODERN_VERSION) + client = _CatalogBoundaryClient( + status=terminal_status, + tool_count=tool_count, + server_info={"name": "catalog-boundary-server"} if at_limit else None, + error=f"MCP startup failed: {CATALOG_LIMIT_ERROR}" if not at_limit else None, + preceding_statuses=("starting",), + ) + + _run_catalog_boundary_checks( + case, + client, # type: ignore[arg-type] - exercise the real app-server client contract. + profile=profile, + workspace=tmp_path, + ) + case.finish(time.monotonic()) + + assert case.success + assert client.events[0]["params"]["status"] == "starting" # type: ignore[index] + assert client.events[1]["params"]["status"] == terminal_status # type: ignore[index] + + +def test_catalog_over_limit_does_not_accept_an_unrelated_startup_failure( + tmp_path: Path, +) -> None: + case = CaseResult(transport="review-stdio:catalog-over-limit", mode=MODERN_VERSION) + client = _CatalogBoundaryClient( + status="failed", + tool_count=0, + server_info=None, + error="MCP startup failed: connection unexpectedly closed", + ) + + _run_catalog_boundary_checks( + case, + client, # type: ignore[arg-type] - exercise the real app-server client contract. + profile=CATALOG_OVER_LIMIT_PROFILE, + workspace=tmp_path, + ) + case.finish(time.monotonic()) + + assert not case.success + assert [check.name for check in case.checks if not check.success] == [ + "review/catalog-over-limit/exact-limit-error" + ] + + +@pytest.mark.parametrize("mode", REVIEW_MODES) +@pytest.mark.parametrize("transport", CATALOG_BOUNDARY_TRANSPORTS) +def test_catalog_over_limit_records_all_checks_when_startup_incorrectly_succeeds( + mode: str, transport: str, tmp_path: Path +) -> None: + case = CaseResult(transport=f"review-{transport}:catalog-over-limit", mode=mode) + client = _CatalogBoundaryClient( + status="ready", + tool_count=MAX_CATALOG_ITEMS + 1, + server_info={"name": "catalog-boundary-server"}, + ) + + _run_catalog_boundary_checks( + case, + client, # type: ignore[arg-type] - exercise the real app-server client contract. + profile=CATALOG_OVER_LIMIT_PROFILE, + workspace=tmp_path, + ) + case.finish(time.monotonic()) + + assert not case.success + assert [check.name for check in case.checks] == [ + "review/ephemeral-thread", + "review/catalog-over-limit/startup-failed", + "review/catalog-over-limit/exact-limit-error", + "review/catalog-over-limit/configured-server", + "review/catalog-over-limit/server-info", + "review/catalog-over-limit/tool-count", + ] + assert [check.name for check in case.checks if not check.success] == [ + "review/catalog-over-limit/startup-failed", + "review/catalog-over-limit/exact-limit-error", + "review/catalog-over-limit/server-info", + "review/catalog-over-limit/tool-count", + ] + + +def test_catalog_over_limit_does_not_treat_a_missing_notification_as_rejection( + tmp_path: Path, +) -> None: + case = CaseResult(transport="review-stdio:catalog-over-limit", mode=LEGACY_VERSION) + client = _CatalogBoundaryClient( + status=None, + tool_count=0, + server_info=None, + ) + + _run_catalog_boundary_checks( + case, + client, # type: ignore[arg-type] - exercise the real app-server client contract. + profile=CATALOG_OVER_LIMIT_PROFILE, + workspace=tmp_path, + ) + case.finish(time.monotonic()) + + assert not case.success + assert [check.name for check in case.checks if not check.success] == [ + "review/catalog-over-limit/startup-failed" + ] + + +@pytest.mark.parametrize( + ("profile", "status", "failure_name"), + [ + ( + CATALOG_MAX_PROFILE, + "ready", + "review/catalog-at-limit/startup-ready", + ), + ( + CATALOG_OVER_LIMIT_PROFILE, + "failed", + "review/catalog-over-limit/startup-failed", + ), + ], +) +def test_catalog_boundary_does_not_accept_another_servers_startup_notification( + profile: str, status: str, failure_name: str, tmp_path: Path +) -> None: + at_limit = profile == CATALOG_MAX_PROFILE + case = CaseResult(transport=f"review-stdio:{profile}", mode=MODERN_VERSION) + client = _CatalogBoundaryClient( + status=status, + tool_count=MAX_CATALOG_ITEMS if at_limit else 0, + server_info={"name": "catalog-boundary-server"} if at_limit else None, + error=CATALOG_LIMIT_ERROR if not at_limit else None, + notification_name="another-mcp-server", + ) + + _run_catalog_boundary_checks( + case, + client, # type: ignore[arg-type] - exercise the real app-server client contract. + profile=profile, + workspace=tmp_path, + ) + case.finish(time.monotonic()) + + assert not case.success + assert [check.name for check in case.checks if not check.success] == [failure_name] + + +@pytest.mark.parametrize("tool_count", [MAX_CATALOG_ITEMS - 1, MAX_CATALOG_ITEMS + 1]) +def test_catalog_at_limit_rejects_incomplete_or_excessive_discovery( + tool_count: int, tmp_path: Path +) -> None: + case = CaseResult(transport="review-stdio:catalog-max", mode=MODERN_VERSION) + client = _CatalogBoundaryClient( + status="ready", + tool_count=tool_count, + server_info={"name": "catalog-boundary-server"}, + ) + + _run_catalog_boundary_checks( + case, + client, # type: ignore[arg-type] - exercise the real app-server client contract. + profile=CATALOG_MAX_PROFILE, + workspace=tmp_path, + ) + case.finish(time.monotonic()) + + assert not case.success + assert [check.name for check in case.checks if not check.success] == [ + "review/catalog-at-limit/tool-count" + ] + + +@pytest.mark.parametrize("mode", REVIEW_MODES) +@pytest.mark.parametrize("profile", CATALOG_BOUNDARY_PROFILES) +def test_catalog_boundary_stdio_registration_uses_the_exact_fixture_profile( + mode: str, profile: str +) -> None: + command = _review_registration_command( + Path("/opt/codex"), + Path("/src/server.py"), + transport="stdio", + mode=mode, + profile=profile, + http_url=None, + ) + + assert command[-6:] == [ + "--mode", + mode, + "--transport", + "stdio", + "--profile", + profile, + ] + + +def test_review_matrix_appends_all_transport_catalog_boundaries( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + observed: list[tuple[str, str, str]] = [] + + def fake_run_review_case( + codex_binary: Path, + server_script: Path, + *, + mode: str, + profile: str, + transport: str, + case_home: Path, + timeout_seconds: float, + ) -> CaseResult: + observed.append((mode, profile, transport)) + case = CaseResult(transport=f"review-{transport}:{profile}", mode=mode) + case.check("review/fake-preserved-case", True, "ok") + case.finish(time.monotonic()) + return case + + monkeypatch.setattr("review_regressions._run_review_case", fake_run_review_case) + report = run_review_regressions( + Path("/opt/codex"), + modes=REVIEW_MODES, + server_script=Path("/src/server.py"), + artifact_parent=tmp_path, + ) + + assert observed[:9] == [ + (SHIPPING_LEGACY_VERSION, REVIEW_PROFILE, "stdio"), + (LEGACY_VERSION, REVIEW_PROFILE, "stdio"), + (MODERN_VERSION, REVIEW_PROFILE, "stdio"), + (MODERN_VERSION, REVIEW_PROFILE, "http"), + (MODERN_VERSION, REPEATED_CURSOR_PROFILE, "http"), + (MODERN_VERSION, MISMATCHED_DISCOVERY_ID_PROFILE, "http"), + (MODERN_VERSION, NULL_DISCOVERY_ID_PROFILE, "http"), + (MODERN_VERSION, SSE_CR_COMMENTS_PROFILE, "http"), + (MODERN_VERSION, SSE_COMMENT_FLOOD_PROFILE, "http"), + ] + assert observed[9:] == [ + (mode, profile, transport) + for mode in REVIEW_MODES + for transport in CATALOG_BOUNDARY_TRANSPORTS + for profile in CATALOG_BOUNDARY_PROFILES + ] + assert report["success"] is True + assert report["summary"] == { + "passed": 21, + "failed": 0, + "total": 21, + "casesPassed": 21, + "casesTotal": 21, + } + + +_KNOWN_REVIEW_FAILURES = { + (MODERN_VERSION, f"review-{transport}:{REVIEW_PROFILE}", check_id) + for transport in ("stdio", "http") + for check_id in ( + "review/exact-large-integer-elicitation-schema", + "review/exact-large-integer-elicitation-round-trip", + "review/mrtr-input-requests-bounded-to-64", + ) +} + +_KNOWN_MAIN_CATALOG_FAILURES = { + (mode, f"review-{transport}:{CATALOG_OVER_LIMIT_PROFILE}", check_id) + for mode in REVIEW_MODES + for transport in CATALOG_BOUNDARY_TRANSPORTS + for check_id in ( + "review/catalog-over-limit/startup-failed", + "review/catalog-over-limit/exact-limit-error", + "review/catalog-over-limit/server-info", + "review/catalog-over-limit/tool-count", + ) +} + + +def _refresh_review_gate_report(report: dict[str, object]) -> None: + cases = report["cases"] + assert isinstance(cases, list) + passed = 0 + failed = 0 + cases_passed = 0 + for case in cases: + assert isinstance(case, dict) + checks = case["checks"] + assert isinstance(checks, list) + successes = [] + for check in checks: + assert isinstance(check, dict) + success = check["success"] + assert isinstance(success, bool) + successes.append(success) + case_success = bool(checks) and all(successes) + case["success"] = case_success + passed += sum(successes) + failed += len(successes) - sum(successes) + cases_passed += case_success + report["success"] = bool(cases) and failed == 0 + report["summary"] = { + "passed": passed, + "failed": failed, + "total": passed + failed, + "casesPassed": cases_passed, + "casesTotal": len(cases), + } + + +def _review_gate_report( + *, + failures: set[tuple[str, str, str]] | None = None, +) -> dict[str, object]: + failures = _KNOWN_REVIEW_FAILURES if failures is None else failures + cases: list[dict[str, object]] = [] + for mode, transport in sorted(_required_review_cases()): + check_ids = ["review/mcp-registration"] + if transport.endswith(f":{CATALOG_OVER_LIMIT_PROFILE}"): + check_ids.extend( + ( + "review/catalog-over-limit/startup-failed", + "review/catalog-over-limit/exact-limit-error", + "review/catalog-over-limit/configured-server", + "review/catalog-over-limit/server-info", + "review/catalog-over-limit/tool-count", + ) + ) + if mode == MODERN_VERSION and transport in { + f"review-stdio:{REVIEW_PROFILE}", + f"review-http:{REVIEW_PROFILE}", + }: + check_ids.extend( + ( + "review/exact-large-integer-elicitation-schema", + "review/exact-large-integer-elicitation-round-trip", + "review/mrtr-input-requests-bounded-to-64", + ) + ) + cases.append( + { + "mode": mode, + "transport": transport, + "checks": [ + { + "name": check_id, + "success": (mode, transport, check_id) not in failures, + "detail": "synthetic real-probe identity", + } + for check_id in check_ids + ], + } + ) + report: dict[str, object] = { + "schemaVersion": REVIEW_REPORT_SCHEMA_VERSION, + "reviewer": REVIEWER, + "modes": list(REVIEW_MODES), + "cases": cases, + } + _refresh_review_gate_report(report) + return report + + +def test_reviewer_baseline_preserves_all_required_cases_and_known_failures() -> None: + baseline = _compact_review_regression_baseline(_review_gate_report()) + + assert baseline["baselineKind"] == REVIEW_BASELINE_KIND + assert baseline["requiredModes"] == list(REVIEW_MODES) + assert len(baseline["requiredCases"]) == 21 # type: ignore[arg-type] + checks = baseline["checks"] + assert isinstance(checks, dict) + failures = checks["failing"] + assert isinstance(failures, list) + assert { + (failure["mode"], failure["transport"], failure["check_id"]) + for failure in failures + } == _KNOWN_REVIEW_FAILURES + + +def test_committed_reviewer_baseline_records_all_real_production_checks() -> None: + path = Path(__file__).with_name("review-regression-baseline-v1.json") + baseline = json.loads(path.read_text(encoding="utf-8")) + errors: list[str] = [] + checks = _required_review_checks(baseline, label="baseline", errors=errors) + + assert errors == [] + assert len(checks) == 186 + assert sum(checks.values()) == 156 + assert { + (identity.mode, identity.transport, identity.check_id) + for identity, success in checks.items() + if not success + } == _KNOWN_REVIEW_FAILURES | _KNOWN_MAIN_CATALOG_FAILURES + assert baseline["summary"] == { + "passed": 156, + "failed": 30, + "total": 186, + "casesPassed": 13, + "casesTotal": 21, + } + + +def test_reviewer_gate_reports_known_failures_without_calling_them_passes() -> None: + report = _review_gate_report() + baseline = _compact_review_regression_baseline(report) + + gate = _evaluate_review_regression_gate(report, baseline) + + assert report["success"] is False + assert gate["success"] is True + assert gate["configurationErrors"] == [] + assert gate["newFailures"] == [] + assert gate["missingChecks"] == [] + assert gate["fixedChecks"] == [] + known = gate["knownFailures"] + assert isinstance(known, list) + assert { + (failure["mode"], failure["transport"], failure["check_id"]) + for failure in known + } == _KNOWN_REVIEW_FAILURES + + +def test_reviewer_gate_preserves_catalog_failures_observed_on_current_main() -> None: + main_failures = _KNOWN_REVIEW_FAILURES | _KNOWN_MAIN_CATALOG_FAILURES + report = _review_gate_report(failures=main_failures) + baseline = _compact_review_regression_baseline(report) + + gate = _evaluate_review_regression_gate(report, baseline) + + assert report["success"] is False + assert gate["success"] is True + assert gate["configurationErrors"] == [] + assert gate["newFailures"] == [] + assert gate["missingChecks"] == [] + known = gate["knownFailures"] + assert isinstance(known, list) + assert { + (failure["mode"], failure["transport"], failure["check_id"]) + for failure in known + } == main_failures + + +def test_reviewer_gate_reports_all_future_catalog_boundary_fixes() -> None: + main_failures = _KNOWN_REVIEW_FAILURES | _KNOWN_MAIN_CATALOG_FAILURES + baseline = _compact_review_regression_baseline( + _review_gate_report(failures=main_failures) + ) + candidate = _review_gate_report() + + gate = _evaluate_review_regression_gate(candidate, baseline) + + assert gate["success"] is True + assert gate["configurationErrors"] == [] + assert gate["newFailures"] == [] + assert gate["missingChecks"] == [] + fixed = gate["fixedChecks"] + assert isinstance(fixed, list) + assert { + (identity["mode"], identity["transport"], identity["check_id"]) + for identity in fixed + } == _KNOWN_MAIN_CATALOG_FAILURES + known = gate["knownFailures"] + assert isinstance(known, list) + assert { + (identity["mode"], identity["transport"], identity["check_id"]) + for identity in known + } == _KNOWN_REVIEW_FAILURES + + +def test_reviewer_gate_rejects_a_newly_failing_previously_passing_check() -> None: + baseline = _compact_review_regression_baseline(_review_gate_report()) + new_failure = ( + SHIPPING_LEGACY_VERSION, + f"review-stdio:{REVIEW_PROFILE}", + "review/mcp-registration", + ) + candidate = _review_gate_report(failures=_KNOWN_REVIEW_FAILURES | {new_failure}) + + gate = _evaluate_review_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert gate["newFailures"] == [ + { + "mode": new_failure[0], + "transport": new_failure[1], + "check_id": new_failure[2], + } + ] + + +def test_reviewer_gate_reports_when_a_known_failure_is_fixed() -> None: + baseline = _compact_review_regression_baseline(_review_gate_report()) + fixed = sorted(_KNOWN_REVIEW_FAILURES)[0] + candidate = _review_gate_report(failures=_KNOWN_REVIEW_FAILURES - {fixed}) + + gate = _evaluate_review_regression_gate(candidate, baseline) + + assert gate["success"] is True + assert gate["fixedChecks"] == [ + {"mode": fixed[0], "transport": fixed[1], "check_id": fixed[2]} + ] + assert len(gate["knownFailures"]) == 5 # type: ignore[arg-type] + + +def test_reviewer_gate_rejects_an_omitted_required_case() -> None: + baseline = _compact_review_regression_baseline(_review_gate_report()) + candidate = _review_gate_report() + cases = candidate["cases"] + assert isinstance(cases, list) + omitted = cases.pop(0) + assert isinstance(omitted, dict) + _refresh_review_gate_report(candidate) + + gate = _evaluate_review_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert any( + "missing required reviewer case" in error + for error in gate["configurationErrors"] # type: ignore[union-attr] + ) + assert any( + missing["mode"] == omitted["mode"] + and missing["transport"] == omitted["transport"] + for missing in gate["missingChecks"] # type: ignore[union-attr] + ) + + +def test_reviewer_gate_rejects_an_omitted_required_check() -> None: + baseline = _compact_review_regression_baseline(_review_gate_report()) + candidate = _review_gate_report() + cases = candidate["cases"] + assert isinstance(cases, list) + case = next( + case + for case in cases + if isinstance(case, dict) + and case["mode"] == MODERN_VERSION + and case["transport"] == f"review-stdio:{REVIEW_PROFILE}" + ) + checks = case["checks"] + assert isinstance(checks, list) + omitted = checks.pop(0) + assert isinstance(omitted, dict) + _refresh_review_gate_report(candidate) + + gate = _evaluate_review_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert gate["missingChecks"] == [ + { + "mode": MODERN_VERSION, + "transport": f"review-stdio:{REVIEW_PROFILE}", + "check_id": omitted["name"], + } + ] + + +@pytest.mark.parametrize("field", ["schemaVersion", "reviewer", "requiredModes"]) +def test_reviewer_gate_rejects_an_incompatible_baseline(field: str) -> None: + baseline = _compact_review_regression_baseline(_review_gate_report()) + baseline[field] = [] if field == "requiredModes" else "wrong" + + gate = _evaluate_review_regression_gate(_review_gate_report(), baseline) + + assert gate["success"] is False + assert gate["configurationErrors"] + + +def test_reviewer_gate_rejects_duplicate_baseline_check_identities() -> None: + baseline = _compact_review_regression_baseline(_review_gate_report()) + buckets = baseline["checks"] + assert isinstance(buckets, dict) + passing = buckets["passing"] + assert isinstance(passing, list) + passing.append(deepcopy(passing[0])) + + gate = _evaluate_review_regression_gate(_review_gate_report(), baseline) + + assert gate["success"] is False + assert any( + "duplicate reviewer check" in error + for error in gate["configurationErrors"] # type: ignore[union-attr] + ) + + +def test_reviewer_gate_rejects_inconsistent_candidate_totals() -> None: + baseline = _compact_review_regression_baseline(_review_gate_report()) + candidate = _review_gate_report() + candidate["summary"] = {"passed": 999, "failed": 0, "total": 999} + + gate = _evaluate_review_regression_gate(candidate, baseline) + + assert gate["success"] is False + assert any( + "inconsistent reviewer check totals" in error + for error in gate["configurationErrors"] # type: ignore[union-attr] + ) + + +def test_review_json_is_pretty_sorted_and_newline_terminated(tmp_path: Path) -> None: + report_path = tmp_path / "nested" / "review.json" + + _write_review_json(report_path, {"z": 1, "a": {"second": 2, "first": 1}}) + + assert report_path.read_text(encoding="utf-8") == ( + '{\n "a": {\n "first": 1,\n "second": 2\n },\n "z": 1\n}\n' + ) + + +def test_review_cli_supports_a_reviewed_production_baseline() -> None: + args = _parse_args( + [ + "/opt/codex", + "--baseline-report", + "/tmp/review-regression-baseline-v1.json", + "--report", + "/tmp/mcp-review-regressions.json", + ] + ) + + assert args.baseline_report == Path("/tmp/review-regression-baseline-v1.json") + + +def test_review_cli_requires_a_source_report_when_extracting_a_baseline( + capsys: pytest.CaptureFixture[str], +) -> None: + result = main( + [sys.executable, "--extract-baseline", "/tmp/review-baseline-test.json"] + ) + + assert result == 2 + assert "--extract-baseline requires --baseline-report" in capsys.readouterr().err + + +def test_review_cli_exits_successfully_only_for_a_passing_regression_gate( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + candidate = _review_gate_report() + baseline_path = tmp_path / "review-baseline.json" + report_path = tmp_path / "review-report.json" + _write_review_json( + baseline_path, _compact_review_regression_baseline(deepcopy(candidate)) + ) + monkeypatch.setattr( + "review_regressions.run_review_regressions", + lambda *args, **kwargs: deepcopy(candidate), + ) + + result = main( + [ + sys.executable, + "--baseline-report", + str(baseline_path), + "--report", + str(report_path), + ] + ) + + report = json.loads(report_path.read_text(encoding="utf-8")) + assert result == 0 + assert report["success"] is False + assert report["regressionGate"]["success"] is True + assert len(report["regressionGate"]["knownFailures"]) == 6 + + +def test_review_cli_does_not_accept_new_failures_as_known_regressions( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + baseline_path = tmp_path / "review-baseline.json" + report_path = tmp_path / "review-report.json" + _write_review_json( + baseline_path, _compact_review_regression_baseline(_review_gate_report()) + ) + new_failure = ( + SHIPPING_LEGACY_VERSION, + f"review-stdio:{REVIEW_PROFILE}", + "review/mcp-registration", + ) + monkeypatch.setattr( + "review_regressions.run_review_regressions", + lambda *args, **kwargs: _review_gate_report( + failures=_KNOWN_REVIEW_FAILURES | {new_failure} + ), + ) + + result = main( + [ + sys.executable, + "--baseline-report", + str(baseline_path), + "--report", + str(report_path), + ] + ) + + report = json.loads(report_path.read_text(encoding="utf-8")) + assert result == 1 + assert report["success"] is False + assert report["regressionGate"]["success"] is False + assert len(report["regressionGate"]["newFailures"]) == 1 + + +def test_review_cli_extracts_a_deterministic_complete_baseline( + tmp_path: Path, +) -> None: + source_path = tmp_path / "full-review-report.json" + baseline_path = tmp_path / "review-regression-baseline-v1.json" + report = _review_gate_report() + _write_review_json(source_path, report) + + result = main( + [ + sys.executable, + "--baseline-report", + str(source_path), + "--extract-baseline", + str(baseline_path), + ] + ) + + expected = _compact_review_regression_baseline(report) + assert result == 0 + assert baseline_path.read_text(encoding="utf-8") == ( + json.dumps(expected, ensure_ascii=False, indent=2, sort_keys=True) + "\n" + ) diff --git a/scripts/mcp_conformance/test_server.py b/scripts/mcp_conformance/test_server.py new file mode 100644 index 0000000000..0828ff8db7 --- /dev/null +++ b/scripts/mcp_conformance/test_server.py @@ -0,0 +1,1590 @@ +import base64 +import http.client +import io +import json +import threading +from collections.abc import Iterator +from contextlib import contextmanager +from unittest.mock import patch + +import pytest + +from server import ( + CATALOG_MAX_PROFILE, + CATALOG_OVER_LIMIT_PROFILE, + DEFAULT_PROFILE, + HEADER_MISMATCH, + INVALID_PARAMS, + LEGACY_VERSION, + MAX_CATALOG_ITEMS, + MISMATCHED_DISCOVERY_ID_PROFILE, + MISSING_REQUIRED_CLIENT_CAPABILITY, + MODERN_VERSION, + NULL_DISCOVERY_ID_PROFILE, + REPEATED_CURSOR_PROFILE, + RESOURCE_URIS, + REVIEW_EXACT_INTEGER, + REVIEW_MAX_INTEGER, + REVIEW_MRTR_INPUT_REQUEST_COUNT, + REVIEW_PAGE_CURSOR, + REVIEW_PROFILE, + REVIEW_SSE_COMMENT_LINE_BYTES, + REVIEW_SSE_COMMENT_LINE_COUNT, + REVIEW_SSE_EVENT_LIMIT_BYTES, + SHIPPING_LEGACY_VERSION, + SSE_COMMENT_FLOOD_PROFILE, + SSE_CR_COMMENTS_PROFILE, + UNSUPPORTED_PROTOCOL_VERSION, + ConnectionState, + ProtocolServer, + ResponsePlan, + _encode_sse_messages, + _parse_args, + make_http_server, + run_stdio, +) + + +def modern_meta( + *, + version: str = MODERN_VERSION, + capabilities: dict[str, object] | None = None, +) -> dict[str, object]: + return { + "io.modelcontextprotocol/protocolVersion": version, + "io.modelcontextprotocol/clientInfo": { + "name": "fixture-test-client", + "version": "1.0.0", + }, + "io.modelcontextprotocol/clientCapabilities": capabilities or {}, + } + + +def request( + method: str, + *, + request_id: int = 1, + params: dict[str, object] | None = None, +) -> dict[str, object]: + return { + "jsonrpc": "2.0", + "id": request_id, + "method": method, + "params": {"_meta": modern_meta()} if params is None else params, + } + + +def result(plan: ResponsePlan) -> dict[str, object]: + response = plan.response + assert isinstance(response, dict) + value = response.get("result") + assert isinstance(value, dict) + return value + + +def error(plan: ResponsePlan) -> dict[str, object]: + response = plan.response + assert isinstance(response, dict) + value = response.get("error") + assert isinstance(value, dict) + return value + + +def initialize_legacy( + server: ProtocolServer, + state: ConnectionState, + *, + version: str = LEGACY_VERSION, +) -> None: + server.handle( + request( + "initialize", + params={ + "protocolVersion": version, + "capabilities": {}, + "clientInfo": {"name": "legacy-client", "version": "1.0.0"}, + }, + ), + state, + ) + server.handle( + { + "jsonrpc": "2.0", + "method": "notifications/initialized", + "params": {}, + }, + state, + ) + + +def test_modern_discovery_is_stateless_and_cacheable() -> None: + server = ProtocolServer(MODERN_VERSION) + + plan = server.handle(request("server/discover"), ConnectionState()) + value = result(plan) + + assert value["resultType"] == "complete" + assert value["supportedVersions"] == [MODERN_VERSION] + assert value["ttlMs"] == 60_000 + assert value["cacheScope"] == "public" + assert value["_meta"] == { + "io.modelcontextprotocol/serverInfo": { + "name": "openai-mcp-spec-test-server", + "version": "0.1.0", + } + } + + +def test_modern_request_requires_metadata() -> None: + server = ProtocolServer(MODERN_VERSION) + + plan = server.handle( + request("tools/list", params={}), + ConnectionState(), + ) + + assert error(plan)["code"] == INVALID_PARAMS + assert plan.http_status == 400 + + +def test_modern_request_reports_supported_version() -> None: + server = ProtocolServer(MODERN_VERSION) + message = request( + "tools/list", + params={"_meta": modern_meta(version=LEGACY_VERSION)}, + ) + + plan = server.handle(message, ConnectionState()) + + assert error(plan) == { + "code": UNSUPPORTED_PROTOCOL_VERSION, + "message": "Unsupported protocol version", + "data": { + "supported": [MODERN_VERSION], + "requested": LEGACY_VERSION, + }, + } + + +def test_legacy_probe_falls_back_then_uses_initialize() -> None: + server = ProtocolServer(LEGACY_VERSION) + state = ConnectionState() + + probe = server.handle(request("server/discover"), state) + assert error(probe)["code"] == -32601 + + initialize_legacy(server, state) + tools = result(server.handle(request("tools/list", params={}), state)) + + assert "resultType" not in tools + assert "ttlMs" not in tools + assert [tool["name"] for tool in tools["tools"]] == sorted( + tool["name"] for tool in tools["tools"] + ) + + +def test_shipping_legacy_mode_negotiates_the_real_product_protocol() -> None: + server = ProtocolServer(SHIPPING_LEGACY_VERSION) + state = ConnectionState() + + plan = server.handle( + request( + "initialize", + params={ + "protocolVersion": SHIPPING_LEGACY_VERSION, + "capabilities": {}, + "clientInfo": {"name": "shipping-client", "version": "1.0.0"}, + }, + ), + state, + ) + + assert result(plan)["protocolVersion"] == "2025-06-18" + assert result(plan)["instructions"] == "Legacy 2025-06-18 compatibility fixture." + assert "resultType" not in result(plan) + assert state.initialize_seen + + +def test_shipping_legacy_discovery_falls_back_to_initialized_legacy_tools() -> None: + server = ProtocolServer(SHIPPING_LEGACY_VERSION) + state = ConnectionState() + + probe = server.handle(request("server/discover"), state) + + assert error(probe)["code"] == -32601 + + initialize_legacy(server, state, version=SHIPPING_LEGACY_VERSION) + tools = result(server.handle(request("tools/list", params={}), state)) + + assert "resultType" not in tools + assert "ttlMs" not in tools + assert [tool["name"] for tool in tools["tools"]] == [ + "client_metadata", + "echo", + "fail", + "header_echo", + "progress", + ] + + +def test_cli_accepts_the_shipping_legacy_protocol_mode() -> None: + args = _parse_args(["--mode", SHIPPING_LEGACY_VERSION, "--transport", "stdio"]) + + assert args.mode == "2025-06-18" + assert args.transport == "stdio" + assert args.profile == DEFAULT_PROFILE + + +@pytest.mark.parametrize( + "mode", (SHIPPING_LEGACY_VERSION, LEGACY_VERSION, MODERN_VERSION) +) +@pytest.mark.parametrize("transport", ("stdio", "http")) +@pytest.mark.parametrize("profile", (CATALOG_MAX_PROFILE, CATALOG_OVER_LIMIT_PROFILE)) +def test_catalog_boundary_profiles_are_available_from_the_cli( + mode: str, transport: str, profile: str +) -> None: + args = _parse_args(["--mode", mode, "--transport", transport, "--profile", profile]) + + assert (args.mode, args.transport, args.profile) == (mode, transport, profile) + + +def test_mrtr_requires_capability_and_echoes_state() -> None: + server = ProtocolServer(MODERN_VERSION) + tools = result(server.handle(request("tools/list"), ConnectionState()))["tools"] + tools_by_name = {tool["name"]: tool for tool in tools} + assert "outputSchema" not in tools_by_name["request_input"] + assert "outputSchema" in tools_by_name["echo"] + + first = request( + "tools/call", + params={ + "_meta": modern_meta(), + "name": "request_input", + "arguments": {}, + }, + ) + + missing_capability = server.handle(first, ConnectionState()) + assert error(missing_capability)["code"] == MISSING_REQUIRED_CLIENT_CAPABILITY + + first["params"]["_meta"] = modern_meta(capabilities={"elicitation": {"form": {}}}) + interim = result(server.handle(first, ConnectionState())) + assert interim["resultType"] == "input_required" + assert interim["_meta"] == { + "io.modelcontextprotocol/serverInfo": { + "name": "openai-mcp-spec-test-server", + "version": "0.1.0", + } + } + assert "structuredContent" not in interim + assert interim["requestState"] == "opaque:request_input:v1" + assert interim["inputRequests"]["confirmation"]["method"] == "elicitation/create" + + retry = request( + "tools/call", + request_id=2, + params={ + "_meta": modern_meta(capabilities={"elicitation": {"form": {}}}), + "name": "request_input", + "arguments": {}, + "requestState": interim["requestState"], + "inputResponses": { + "confirmation": { + "action": "accept", + "content": {"confirmation": "confirmed"}, + } + }, + }, + ) + completed = result(server.handle(retry, ConnectionState())) + + assert completed["resultType"] == "complete" + assert completed["structuredContent"] == {"confirmation": "confirmed"} + + +def test_subscription_is_acknowledged_tagged_and_closed() -> None: + server = ProtocolServer(MODERN_VERSION) + plan = server.handle( + request( + "subscriptions/listen", + request_id=42, + params={ + "_meta": modern_meta(), + "notifications": { + "toolsListChanged": True, + "resourceSubscriptions": [RESOURCE_URIS[0], "test://unknown"], + }, + }, + ), + ConnectionState(), + ) + + assert plan.force_sse + assert plan.notifications[0]["method"] == ( + "notifications/subscriptions/acknowledged" + ) + for notification in plan.notifications: + assert ( + notification["params"]["_meta"]["io.modelcontextprotocol/subscriptionId"] + == 42 + ) + assert result(plan)["_meta"]["io.modelcontextprotocol/subscriptionId"] == 42 + + +def test_progress_and_logging_are_per_request_opt_ins() -> None: + server = ProtocolServer(MODERN_VERSION) + message = request( + "tools/call", + params={ + "_meta": { + **modern_meta(), + "progressToken": "progress-1", + "io.modelcontextprotocol/logLevel": "info", + }, + "name": "progress", + "arguments": {}, + }, + ) + + plan = server.handle(message, ConnectionState()) + + assert [notification["method"] for notification in plan.notifications] == [ + "notifications/progress", + "notifications/message", + ] + assert result(plan)["resultType"] == "complete" + + +def test_stdio_supports_both_eras() -> None: + modern_input = io.StringIO( + json.dumps(request("tools/list"), separators=(",", ":")) + "\n" + ) + modern_output = io.StringIO() + run_stdio(ProtocolServer(MODERN_VERSION), modern_input, modern_output) + modern_response = json.loads(modern_output.getvalue()) + assert modern_response["result"]["resultType"] == "complete" + + legacy_messages = [ + request( + "initialize", + params={ + "protocolVersion": LEGACY_VERSION, + "capabilities": {}, + "clientInfo": {"name": "legacy", "version": "1.0.0"}, + }, + ), + { + "jsonrpc": "2.0", + "method": "notifications/initialized", + "params": {}, + }, + request("tools/list", request_id=2, params={}), + ] + legacy_input = io.StringIO( + "".join( + json.dumps(item, separators=(",", ":")) + "\n" for item in legacy_messages + ) + ) + legacy_output = io.StringIO() + run_stdio(ProtocolServer(LEGACY_VERSION), legacy_input, legacy_output) + responses = [json.loads(line) for line in legacy_output.getvalue().splitlines()] + assert len(responses) == 2 + assert "resultType" not in responses[1]["result"] + + +def test_stdio_negotiates_and_lists_tools_for_the_real_shipping_legacy_protocol() -> ( + None +): + messages = [ + request( + "initialize", + params={ + "protocolVersion": SHIPPING_LEGACY_VERSION, + "capabilities": {}, + "clientInfo": {"name": "shipping-legacy", "version": "1.0.0"}, + }, + ), + { + "jsonrpc": "2.0", + "method": "notifications/initialized", + "params": {}, + }, + request("tools/list", request_id=2, params={}), + ] + stdin = io.StringIO( + "".join( + json.dumps(message, separators=(",", ":")) + "\n" for message in messages + ) + ) + stdout = io.StringIO() + + run_stdio(ProtocolServer(SHIPPING_LEGACY_VERSION), stdin, stdout) + + responses = [json.loads(line) for line in stdout.getvalue().splitlines()] + assert len(responses) == 2 + assert responses[0]["result"]["protocolVersion"] == "2025-06-18" + assert [tool["name"] for tool in responses[1]["result"]["tools"]] == [ + "client_metadata", + "echo", + "fail", + "header_echo", + "progress", + ] + assert "resultType" not in responses[1]["result"] + + +def test_default_profile_preserves_existing_legacy_and_modern_tool_lists() -> None: + modern = ProtocolServer(MODERN_VERSION) + modern_tools = result(modern.handle(request("tools/list"), ConnectionState())) + + assert modern.profile == DEFAULT_PROFILE + assert [tool["name"] for tool in modern_tools["tools"]] == [ + "client_metadata", + "echo", + "fail", + "header_echo", + "progress", + "request_input", + ] + assert "nextCursor" not in modern_tools + + legacy = ProtocolServer(LEGACY_VERSION) + legacy_state = ConnectionState() + initialize_legacy(legacy, legacy_state) + legacy_tools = result(legacy.handle(request("tools/list", params={}), legacy_state)) + + assert [tool["name"] for tool in legacy_tools["tools"]] == [ + "client_metadata", + "echo", + "fail", + "header_echo", + "progress", + ] + assert "nextCursor" not in legacy_tools + + +@pytest.mark.parametrize( + "mode", (SHIPPING_LEGACY_VERSION, LEGACY_VERSION, MODERN_VERSION) +) +@pytest.mark.parametrize( + ("profile", "expected_size"), + ( + (CATALOG_MAX_PROFILE, MAX_CATALOG_ITEMS), + (CATALOG_OVER_LIMIT_PROFILE, MAX_CATALOG_ITEMS + 1), + ), +) +def test_catalog_boundary_profiles_return_exact_valid_unique_tool_counts( + mode: str, profile: str, expected_size: int +) -> None: + server = ProtocolServer(mode, profile=profile) + state = ConnectionState() + if mode != MODERN_VERSION: + initialize_legacy(server, state, version=mode) + + params = {"_meta": modern_meta()} if mode == MODERN_VERSION else {} + catalog = result(server.handle(request("tools/list", params=params), state)) + tools = catalog["tools"] + names = [tool["name"] for tool in tools] + + assert len(tools) == expected_size + assert len(set(names)) == expected_size + assert names == sorted(names) + assert {"client_metadata", "echo", "fail", "header_echo", "progress"} <= set(names) + assert all(tool["inputSchema"]["type"] == "object" for tool in tools) + assert "nextCursor" not in catalog + if mode == MODERN_VERSION: + assert "request_input" in names + assert catalog["resultType"] == "complete" + else: + assert "resultType" not in catalog + + +@pytest.mark.parametrize( + "mode", (SHIPPING_LEGACY_VERSION, LEGACY_VERSION, MODERN_VERSION) +) +@pytest.mark.parametrize( + ("profile", "expected_size"), + ( + (CATALOG_MAX_PROFILE, MAX_CATALOG_ITEMS), + (CATALOG_OVER_LIMIT_PROFILE, MAX_CATALOG_ITEMS + 1), + ), +) +def test_catalog_boundary_profiles_round_trip_over_real_stdio( + mode: str, profile: str, expected_size: int +) -> None: + messages: list[dict[str, object]] = [] + if mode != MODERN_VERSION: + messages.extend( + [ + request( + "initialize", + params={ + "protocolVersion": mode, + "capabilities": {}, + "clientInfo": {"name": "catalog-boundary", "version": "1.0.0"}, + }, + ), + { + "jsonrpc": "2.0", + "method": "notifications/initialized", + "params": {}, + }, + ] + ) + messages.append( + request( + "tools/list", + request_id=2, + params={"_meta": modern_meta()} if mode == MODERN_VERSION else {}, + ) + ) + stdin = io.StringIO( + "".join( + json.dumps(message, separators=(",", ":")) + "\n" for message in messages + ) + ) + stdout = io.StringIO() + + run_stdio(ProtocolServer(mode, profile=profile), stdin, stdout) + + catalog = json.loads(stdout.getvalue().splitlines()[-1])["result"] + names = [tool["name"] for tool in catalog["tools"]] + assert len(names) == expected_size + assert len(set(names)) == expected_size + assert "nextCursor" not in catalog + assert ("resultType" in catalog) is (mode == MODERN_VERSION) + + +def test_review_profile_hides_regression_tools_on_the_second_page() -> None: + server = ProtocolServer(MODERN_VERSION, profile=REVIEW_PROFILE) + + first = result(server.handle(request("tools/list"), ConnectionState())) + assert [tool["name"] for tool in first["tools"]] == ["client_metadata", "echo"] + assert first["nextCursor"] == REVIEW_PAGE_CURSOR + assert first["resultType"] == "complete" + + second = result( + server.handle( + request( + "tools/list", + request_id=2, + params={"_meta": modern_meta(), "cursor": first["nextCursor"]}, + ), + ConnectionState(), + ) + ) + + assert [tool["name"] for tool in second["tools"]] == [ + "fail", + "header_echo", + "progress", + "request_input", + "review_integer_elicitation", + "review_large_integer", + "review_mrtr_cap", + "review_protocol_env", + ] + assert "nextCursor" not in second + + +def test_review_profile_rejects_an_invalid_tool_cursor() -> None: + server = ProtocolServer(MODERN_VERSION, profile=REVIEW_PROFILE) + + plan = server.handle( + request( + "tools/list", + params={"_meta": modern_meta(), "cursor": "not-a-review-cursor"}, + ), + ConnectionState(), + ) + + assert error(plan) == { + "code": INVALID_PARAMS, + "message": "Invalid tool cursor", + } + + +def test_repeated_cursor_profile_deterministically_repeats_the_cursor() -> None: + server = ProtocolServer(MODERN_VERSION, profile=REPEATED_CURSOR_PROFILE) + + first = result(server.handle(request("tools/list"), ConnectionState())) + second = result( + server.handle( + request( + "tools/list", + request_id=2, + params={"_meta": modern_meta(), "cursor": first["nextCursor"]}, + ), + ConnectionState(), + ) + ) + + assert first["nextCursor"] == REVIEW_PAGE_CURSOR + assert second["nextCursor"] == REVIEW_PAGE_CURSOR + assert any(tool["name"] == "review_large_integer" for tool in second["tools"]) + + +def test_review_large_integer_schema_preserves_exact_json_integer_boundaries() -> None: + server = ProtocolServer(MODERN_VERSION, profile=REVIEW_PROFILE) + page = result( + server.handle( + request( + "tools/list", + params={"_meta": modern_meta(), "cursor": REVIEW_PAGE_CURSOR}, + ), + ConnectionState(), + ) + ) + definition = next( + tool for tool in page["tools"] if tool["name"] == "review_large_integer" + ) + schema = definition["inputSchema"]["properties"]["value"] + + assert schema == { + "type": "integer", + "minimum": 9_007_199_254_740_993, + "maximum": 9_223_372_036_854_775_807, + "default": 9_007_199_254_740_993, + } + assert json.loads(json.dumps(schema)) == schema + assert REVIEW_EXACT_INTEGER > 2**53 + assert REVIEW_MAX_INTEGER == 2**63 - 1 + + +def test_review_large_integer_tool_echoes_without_float_rounding() -> None: + server = ProtocolServer(MODERN_VERSION, profile=REVIEW_PROFILE) + + plan = server.handle( + request( + "tools/call", + params={ + "_meta": modern_meta(), + "name": "review_large_integer", + "arguments": {"value": REVIEW_EXACT_INTEGER}, + }, + ), + ConnectionState(), + ) + + assert result(plan)["structuredContent"] == {"value": 9_007_199_254_740_993} + assert json.loads(result(plan)["content"][0]["text"]) == { + "value": 9_007_199_254_740_993 + } + + +def test_review_large_integer_rejects_floats_booleans_and_out_of_range_values() -> None: + server = ProtocolServer(MODERN_VERSION, profile=REVIEW_PROFILE) + + for invalid_value in ( + float(REVIEW_EXACT_INTEGER), + True, + REVIEW_EXACT_INTEGER - 1, + REVIEW_MAX_INTEGER + 1, + str(REVIEW_EXACT_INTEGER), + ): + plan = server.handle( + request( + "tools/call", + params={ + "_meta": modern_meta(), + "name": "review_large_integer", + "arguments": {"value": invalid_value}, + }, + ), + ConnectionState(), + ) + + assert error(plan)["code"] == INVALID_PARAMS + + +def test_review_large_integer_survives_stdio_json_serialization() -> None: + message = request( + "tools/call", + params={ + "_meta": modern_meta(), + "name": "review_large_integer", + "arguments": {"value": REVIEW_EXACT_INTEGER}, + }, + ) + stdin = io.StringIO(json.dumps(message, separators=(",", ":")) + "\n") + stdout = io.StringIO() + + run_stdio(ProtocolServer(MODERN_VERSION, profile=REVIEW_PROFILE), stdin, stdout) + + raw_response = stdout.getvalue() + assert "9007199254740993" in raw_response + assert json.loads(raw_response)["result"]["structuredContent"] == { + "value": 9_007_199_254_740_993 + } + + +def test_review_integer_elicitation_preserves_the_exact_requested_form_schema() -> None: + server = ProtocolServer(MODERN_VERSION, profile=REVIEW_PROFILE) + + plan = server.handle( + request( + "tools/call", + params={ + "_meta": modern_meta(capabilities={"elicitation": {"form": {}}}), + "name": "review_integer_elicitation", + "arguments": {}, + }, + ), + ConnectionState(), + ) + value = result(plan) + pending = value["inputRequests"]["large_integer"] + + assert value["resultType"] == "input_required" + assert value["requestState"] == "opaque:review_integer_elicitation:v1" + assert pending["method"] == "elicitation/create" + assert pending["params"]["requestedSchema"]["properties"]["value"] == { + "type": "integer", + "minimum": 9_007_199_254_740_993, + "maximum": 9_223_372_036_854_775_807, + "default": 9_007_199_254_740_993, + } + assert "9007199254740993" in json.dumps(value) + + +def test_review_integer_elicitation_round_trip_keeps_the_exact_integer() -> None: + server = ProtocolServer(MODERN_VERSION, profile=REVIEW_PROFILE) + meta = modern_meta(capabilities={"elicitation": {"form": {}}}) + + first = result( + server.handle( + request( + "tools/call", + params={ + "_meta": meta, + "name": "review_integer_elicitation", + "arguments": {}, + }, + ), + ConnectionState(), + ) + ) + completed = result( + server.handle( + request( + "tools/call", + request_id=2, + params={ + "_meta": meta, + "name": "review_integer_elicitation", + "arguments": {}, + "requestState": first["requestState"], + "inputResponses": { + "large_integer": { + "action": "accept", + "content": {"value": REVIEW_EXACT_INTEGER}, + } + }, + }, + ), + ConnectionState(), + ) + ) + + assert completed["resultType"] == "complete" + assert completed["structuredContent"] == {"value": 9_007_199_254_740_993} + assert completed["content"] == [{"type": "text", "text": "9007199254740993"}] + + +def test_review_integer_elicitation_rejects_float_rounded_form_response() -> None: + server = ProtocolServer(MODERN_VERSION, profile=REVIEW_PROFILE) + + plan = server.handle( + request( + "tools/call", + params={ + "_meta": modern_meta(capabilities={"elicitation": {"form": {}}}), + "name": "review_integer_elicitation", + "arguments": {}, + "requestState": "opaque:review_integer_elicitation:v1", + "inputResponses": { + "large_integer": { + "action": "accept", + "content": {"value": float(REVIEW_EXACT_INTEGER)}, + } + }, + }, + ), + ConnectionState(), + ) + + assert error(plan)["code"] == INVALID_PARAMS + + +def test_review_integer_elicitation_requires_advertised_form_capability() -> None: + server = ProtocolServer(MODERN_VERSION, profile=REVIEW_PROFILE) + + plan = server.handle( + request( + "tools/call", + params={ + "_meta": modern_meta(), + "name": "review_integer_elicitation", + "arguments": {}, + }, + ), + ConnectionState(), + ) + + assert error(plan)["code"] == MISSING_REQUIRED_CLIENT_CAPABILITY + assert plan.http_status == 400 + + +def test_review_mrtr_cap_returns_65_deterministic_pending_requests() -> None: + server = ProtocolServer(MODERN_VERSION, profile=REVIEW_PROFILE) + + plan = server.handle( + request( + "tools/call", + params={ + "_meta": modern_meta(capabilities={"elicitation": {"form": {}}}), + "name": "review_mrtr_cap", + "arguments": {}, + }, + ), + ConnectionState(), + ) + value = result(plan) + + assert value["resultType"] == "input_required" + assert value["requestState"] == "opaque:review_mrtr_cap:v1" + assert len(value["inputRequests"]) == REVIEW_MRTR_INPUT_REQUEST_COUNT == 65 + assert list(value["inputRequests"]) == [ + f"review-input-{index:02d}" for index in range(65) + ] + assert all( + pending["method"] == "elicitation/create" + for pending in value["inputRequests"].values() + ) + + +def test_review_mrtr_cap_requires_advertised_elicitation_capability() -> None: + server = ProtocolServer(MODERN_VERSION, profile=REVIEW_PROFILE) + + plan = server.handle( + request( + "tools/call", + params={ + "_meta": modern_meta(), + "name": "review_mrtr_cap", + "arguments": {}, + }, + ), + ConnectionState(), + ) + + assert error(plan)["code"] == MISSING_REQUIRED_CLIENT_CAPABILITY + assert plan.http_status == 400 + + +def test_review_protocol_env_is_available_to_a_legacy_client() -> None: + server = ProtocolServer(LEGACY_VERSION, profile=REVIEW_PROFILE) + state = ConnectionState() + initialize_legacy(server, state) + tools = result(server.handle(request("tools/list", params={}), state)) + + assert any(tool["name"] == "review_protocol_env" for tool in tools["tools"]) + + with patch.dict( + "os.environ", {"CODEX_MCP_PROTOCOL_VERSION": "legacy-review-sentinel"} + ): + plan = server.handle( + request( + "tools/call", + params={"name": "review_protocol_env", "arguments": {}}, + ), + state, + ) + + assert result(plan)["structuredContent"] == {"value": "legacy-review-sentinel"} + + +def test_review_protocol_env_is_available_to_the_shipping_legacy_client() -> None: + server = ProtocolServer(SHIPPING_LEGACY_VERSION, profile=REVIEW_PROFILE) + state = ConnectionState() + initialize_legacy(server, state, version=SHIPPING_LEGACY_VERSION) + tools = result(server.handle(request("tools/list", params={}), state)) + + assert any(tool["name"] == "review_protocol_env" for tool in tools["tools"]) + + with patch.dict( + "os.environ", {"CODEX_MCP_PROTOCOL_VERSION": "shipping-legacy-sentinel"} + ): + plan = server.handle( + request( + "tools/call", + params={"name": "review_protocol_env", "arguments": {}}, + ), + state, + ) + + assert result(plan)["structuredContent"] == {"value": "shipping-legacy-sentinel"} + + +def test_review_protocol_env_is_available_to_a_modern_client() -> None: + server = ProtocolServer(MODERN_VERSION, profile=REVIEW_PROFILE) + + with patch.dict("os.environ", {"CODEX_MCP_PROTOCOL_VERSION": MODERN_VERSION}): + plan = server.handle( + request( + "tools/call", + params={ + "_meta": modern_meta(), + "name": "review_protocol_env", + "arguments": {}, + }, + ), + ConnectionState(), + ) + + assert result(plan)["structuredContent"] == {"value": MODERN_VERSION} + + +def test_mismatched_discovery_profile_returns_the_wrong_jsonrpc_id() -> None: + server = ProtocolServer(MODERN_VERSION, profile=MISMATCHED_DISCOVERY_ID_PROFILE) + + plan = server.handle(request("server/discover", request_id=42), ConnectionState()) + + assert plan.response is not None + assert plan.response["id"] == "review-mismatched-42" + assert result(plan)["supportedVersions"] == [MODERN_VERSION] + + +def test_null_discovery_profile_returns_a_null_jsonrpc_id() -> None: + server = ProtocolServer(MODERN_VERSION, profile=NULL_DISCOVERY_ID_PROFILE) + + plan = server.handle(request("server/discover", request_id=42), ConnectionState()) + + assert plan.response is not None + assert plan.response["id"] is None + assert result(plan)["supportedVersions"] == [MODERN_VERSION] + + +def test_sse_comment_flood_profile_is_available_from_the_cli() -> None: + args = _parse_args( + [ + "--mode", + MODERN_VERSION, + "--transport", + "http", + "--profile", + SSE_COMMENT_FLOOD_PROFILE, + ] + ) + + assert args.profile == "sse-comment-flood" + + +def test_sse_comment_flood_is_valid_bounded_and_exceeds_the_data_event_limit() -> None: + message: dict[str, object] = {"jsonrpc": "2.0", "id": 42, "result": {}} + + body = _encode_sse_messages([message], profile=SSE_COMMENT_FLOOD_PROFILE) + + assert REVIEW_SSE_COMMENT_LINE_COUNT == 4_097 + assert len(body) > REVIEW_SSE_EVENT_LIMIT_BYTES + assert len(body) < ( + REVIEW_SSE_EVENT_LIMIT_BYTES + REVIEW_SSE_COMMENT_LINE_BYTES + 4_096 + ) + assert body.count(b": reviewer keepalive ") == REVIEW_SSE_COMMENT_LINE_COUNT + assert b"\n" not in body + assert body.endswith(b"\r\r") + assert json.loads(body.rsplit(b"data: ", 1)[1].removesuffix(b"\r\r")) == message + + +@contextmanager +def running_http_server( + mode: str, + *, + profile: str = DEFAULT_PROFILE, +) -> Iterator[tuple[str, int]]: + server = make_http_server(ProtocolServer(mode, profile=profile), "127.0.0.1", 0) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + try: + host, port = server.server_address + yield str(host), int(port) + finally: + server.shutdown() + server.server_close() + thread.join(timeout=2) + + +def post_json( + address: tuple[str, int], + message: dict[str, object], + headers: dict[str, str], +) -> tuple[int, dict[str, str], dict[str, object] | None]: + connection = http.client.HTTPConnection(*address, timeout=2) + body = json.dumps(message, ensure_ascii=False, separators=(",", ":")).encode() + connection.request( + "POST", + "/mcp", + body=body, + headers={"Content-Type": "application/json", **headers}, + ) + response = connection.getresponse() + response_body = response.read() + response_headers = {name.lower(): value for name, value in response.getheaders()} + connection.close() + return ( + response.status, + response_headers, + json.loads(response_body) if response_body else None, + ) + + +def modern_headers( + method: str, + *, + name: str | None = None, + extra: dict[str, str] | None = None, +) -> dict[str, str]: + headers = { + "Accept": "application/json, text/event-stream", + "MCP-Protocol-Version": MODERN_VERSION, + "Mcp-Method": method, + } + if name is not None: + headers["Mcp-Name"] = name + if extra is not None: + headers.update(extra) + return headers + + +@pytest.mark.parametrize( + "mode", (SHIPPING_LEGACY_VERSION, LEGACY_VERSION, MODERN_VERSION) +) +@pytest.mark.parametrize( + ("profile", "expected_size"), + ( + (CATALOG_MAX_PROFILE, MAX_CATALOG_ITEMS), + (CATALOG_OVER_LIMIT_PROFILE, MAX_CATALOG_ITEMS + 1), + ), +) +def test_catalog_boundary_profiles_round_trip_over_localhost_http( + mode: str, profile: str, expected_size: int +) -> None: + with running_http_server(mode, profile=profile) as address: + if mode == MODERN_VERSION: + status, _, body = post_json( + address, + request("tools/list"), + modern_headers("tools/list"), + ) + else: + status, headers, body = post_json( + address, + request( + "initialize", + params={ + "protocolVersion": mode, + "capabilities": {}, + "clientInfo": {"name": "catalog-boundary", "version": "1.0.0"}, + }, + ), + {}, + ) + assert status == 200 + assert body is not None + assert body["result"]["protocolVersion"] == mode + session_headers = {"Mcp-Session-Id": headers["mcp-session-id"]} + + initialized_status, _, initialized_body = post_json( + address, + { + "jsonrpc": "2.0", + "method": "notifications/initialized", + "params": {}, + }, + session_headers, + ) + assert initialized_status == 202 + assert initialized_body is None + + status, _, body = post_json( + address, + request("tools/list", request_id=2, params={}), + session_headers, + ) + + assert status == 200 + assert body is not None + catalog = body["result"] + names = [tool["name"] for tool in catalog["tools"]] + assert len(names) == expected_size + assert len(set(names)) == expected_size + assert "nextCursor" not in catalog + assert ("resultType" in catalog) is (mode == MODERN_VERSION) + + +def test_modern_http_validates_standard_and_custom_headers() -> None: + with running_http_server(MODERN_VERSION) as address: + message = request( + "tools/call", + params={ + "_meta": modern_meta(), + "name": "header_echo", + "arguments": { + "region": "us-west1", + "attempt": 3, + "enabled": True, + "greeting": "Hello, 世界", + }, + }, + ) + encoded_greeting = base64.b64encode("Hello, 世界".encode()).decode() + headers = modern_headers( + "tools/call", + name="header_echo", + extra={ + "Mcp-Param-Region": "us-west1", + "Mcp-Param-Attempt": "3", + "Mcp-Param-Enabled": "true", + "Mcp-Param-Greeting": f"=?base64?{encoded_greeting}?=", + }, + ) + + status, _, body = post_json(address, message, headers) + assert status == 200 + assert body["result"]["resultType"] == "complete" + + headers["Mcp-Name"] = "different" + status, _, body = post_json(address, message, headers) + assert status == 400 + assert body["error"]["code"] == HEADER_MISMATCH + + +def test_modern_http_accepts_base64_mcp_name() -> None: + with running_http_server(MODERN_VERSION) as address: + uri = RESOURCE_URIS[1] + encoded_uri = base64.b64encode(uri.encode()).decode() + message = request( + "resources/read", + params={"_meta": modern_meta(), "uri": uri}, + ) + + status, _, body = post_json( + address, + message, + modern_headers( + "resources/read", + name=f"=?base64?{encoded_uri}?=", + ), + ) + + assert status == 200 + assert body["result"]["contents"][0]["uri"] == uri + + +def test_modern_http_streams_subscription_then_closes_gracefully() -> None: + with running_http_server(MODERN_VERSION) as address: + message = request( + "subscriptions/listen", + request_id=77, + params={ + "_meta": modern_meta(), + "notifications": {"toolsListChanged": True}, + }, + ) + connection = http.client.HTTPConnection(*address, timeout=2) + connection.request( + "POST", + "/mcp", + body=json.dumps(message, separators=(",", ":")), + headers={ + "Content-Type": "application/json", + **modern_headers("subscriptions/listen"), + }, + ) + + response = connection.getresponse() + body = response.read().decode() + connection.close() + events = [ + json.loads(block.removeprefix("data: ")) + for block in body.strip().split("\n\n") + ] + + assert response.status == 200 + assert response.getheader("Content-Type") == "text/event-stream" + assert events[0]["method"] == "notifications/subscriptions/acknowledged" + assert events[1]["method"] == "notifications/tools/list_changed" + assert events[-1]["id"] == 77 + assert events[-1]["result"]["resultType"] == "complete" + + +def test_modern_http_rejects_removed_methods_and_unsupported_versions() -> None: + with running_http_server(MODERN_VERSION) as address: + message = request( + "tools/list", + params={"_meta": modern_meta(version=LEGACY_VERSION)}, + ) + headers = modern_headers("tools/list") + headers["MCP-Protocol-Version"] = LEGACY_VERSION + + status, _, body = post_json(address, message, headers) + assert status == 400 + assert body["error"]["code"] == UNSUPPORTED_PROTOCOL_VERSION + + connection = http.client.HTTPConnection(*address, timeout=2) + connection.request("GET", "/mcp") + get_response = connection.getresponse() + get_response.read() + assert get_response.status == 405 + + connection.request("DELETE", "/mcp") + delete_response = connection.getresponse() + delete_response.read() + connection.close() + assert delete_response.status == 405 + + +def test_modern_http_accepts_notification_without_request_headers() -> None: + with running_http_server(MODERN_VERSION) as address: + status, _, body = post_json( + address, + { + "jsonrpc": "2.0", + "method": "notifications/cancelled", + "params": {"requestId": 1, "reason": "fixture test"}, + }, + {}, + ) + + assert status == 202 + assert body is None + + +def test_modern_http_rejects_untrusted_origin() -> None: + with running_http_server(MODERN_VERSION) as address: + status, _, body = post_json( + address, + request("server/discover"), + { + **modern_headers("server/discover"), + "Origin": "https://attacker.example", + }, + ) + + assert status == 403 + assert body["error"]["message"] == "Origin is not allowed" + + +def test_legacy_http_mints_and_terminates_session() -> None: + with running_http_server(LEGACY_VERSION) as address: + initialize = request( + "initialize", + params={ + "protocolVersion": LEGACY_VERSION, + "capabilities": {}, + "clientInfo": {"name": "legacy", "version": "1.0.0"}, + }, + ) + status, headers, body = post_json(address, initialize, {}) + assert status == 200 + assert body["result"]["protocolVersion"] == LEGACY_VERSION + session_id = headers["mcp-session-id"] + + initialized = { + "jsonrpc": "2.0", + "method": "notifications/initialized", + "params": {}, + } + status, _, body = post_json( + address, initialized, {"Mcp-Session-Id": session_id} + ) + assert status == 202 + assert body is None + + status, _, body = post_json( + address, + request("tools/list", request_id=2, params={}), + {"Mcp-Session-Id": session_id}, + ) + assert status == 200 + assert "resultType" not in body["result"] + + connection = http.client.HTTPConnection(*address, timeout=2) + connection.request( + "DELETE", + "/mcp", + headers={"Mcp-Session-Id": session_id}, + ) + response = connection.getresponse() + response.read() + connection.close() + assert response.status == 200 + + +def test_shipping_legacy_http_negotiates_and_retains_a_real_legacy_session() -> None: + with running_http_server(SHIPPING_LEGACY_VERSION) as address: + status, headers, body = post_json( + address, + request( + "initialize", + params={ + "protocolVersion": SHIPPING_LEGACY_VERSION, + "capabilities": {}, + "clientInfo": {"name": "shipping-legacy", "version": "1.0.0"}, + }, + ), + {}, + ) + + assert status == 200 + assert body is not None + assert body["result"]["protocolVersion"] == "2025-06-18" + session_id = headers["mcp-session-id"] + + status, _, body = post_json( + address, + { + "jsonrpc": "2.0", + "method": "notifications/initialized", + "params": {}, + }, + {"Mcp-Session-Id": session_id}, + ) + + assert status == 202 + assert body is None + + status, _, body = post_json( + address, + request("tools/list", request_id=2, params={}), + {"Mcp-Session-Id": session_id}, + ) + + assert status == 200 + assert body is not None + assert [tool["name"] for tool in body["result"]["tools"]] == [ + "client_metadata", + "echo", + "fail", + "header_echo", + "progress", + ] + assert "resultType" not in body["result"] + + +def test_review_http_lists_tools_across_two_pages() -> None: + with running_http_server(MODERN_VERSION, profile=REVIEW_PROFILE) as address: + status, _, first_body = post_json( + address, + request("tools/list"), + modern_headers("tools/list"), + ) + + assert status == 200 + assert first_body is not None + first = first_body["result"] + assert [tool["name"] for tool in first["tools"]] == ["client_metadata", "echo"] + assert first["nextCursor"] == REVIEW_PAGE_CURSOR + + status, _, second_body = post_json( + address, + request( + "tools/list", + request_id=2, + params={"_meta": modern_meta(), "cursor": REVIEW_PAGE_CURSOR}, + ), + modern_headers("tools/list"), + ) + + assert status == 200 + assert second_body is not None + assert any( + tool["name"] == "review_large_integer" + for tool in second_body["result"]["tools"] + ) + assert "nextCursor" not in second_body["result"] + + +def test_review_http_preserves_integer_precision_in_schema_and_tool_results() -> None: + with running_http_server(MODERN_VERSION, profile=REVIEW_PROFILE) as address: + status, _, tools_body = post_json( + address, + request( + "tools/list", + params={"_meta": modern_meta(), "cursor": REVIEW_PAGE_CURSOR}, + ), + modern_headers("tools/list"), + ) + + assert status == 200 + assert tools_body is not None + definition = next( + tool + for tool in tools_body["result"]["tools"] + if tool["name"] == "review_large_integer" + ) + assert definition["inputSchema"]["properties"]["value"]["minimum"] == ( + 9_007_199_254_740_993 + ) + + status, _, call_body = post_json( + address, + request( + "tools/call", + params={ + "_meta": modern_meta(), + "name": "review_large_integer", + "arguments": {"value": REVIEW_EXACT_INTEGER}, + }, + ), + modern_headers("tools/call", name="review_large_integer"), + ) + + assert status == 200 + assert call_body is not None + assert call_body["result"]["structuredContent"] == { + "value": 9_007_199_254_740_993 + } + + +def test_review_http_preserves_exact_integer_in_elicitation_form_schema() -> None: + with running_http_server(MODERN_VERSION, profile=REVIEW_PROFILE) as address: + status, _, body = post_json( + address, + request( + "tools/call", + params={ + "_meta": modern_meta(capabilities={"elicitation": {"form": {}}}), + "name": "review_integer_elicitation", + "arguments": {}, + }, + ), + modern_headers("tools/call", name="review_integer_elicitation"), + ) + + assert status == 200 + assert body is not None + pending = body["result"]["inputRequests"]["large_integer"] + assert pending["params"]["requestedSchema"]["properties"]["value"] == { + "type": "integer", + "minimum": 9_007_199_254_740_993, + "maximum": 9_223_372_036_854_775_807, + "default": 9_007_199_254_740_993, + } + + +def test_review_http_returns_mismatched_discovery_response_id() -> None: + with running_http_server( + MODERN_VERSION, + profile=MISMATCHED_DISCOVERY_ID_PROFILE, + ) as address: + status, _, body = post_json( + address, + request("server/discover", request_id=42), + modern_headers("server/discover"), + ) + + assert status == 200 + assert body is not None + assert body["id"] == "review-mismatched-42" + + +def test_review_http_returns_null_discovery_response_id() -> None: + with running_http_server( + MODERN_VERSION, profile=NULL_DISCOVERY_ID_PROFILE + ) as address: + status, _, body = post_json( + address, + request("server/discover", request_id=42), + modern_headers("server/discover"), + ) + + assert status == 200 + assert body is not None + assert body["id"] is None + + +def test_review_http_streams_sse_with_carriage_returns_and_comments() -> None: + with running_http_server( + MODERN_VERSION, profile=SSE_CR_COMMENTS_PROFILE + ) as address: + message = request( + "subscriptions/listen", + request_id=77, + params={ + "_meta": modern_meta(), + "notifications": {"toolsListChanged": True}, + }, + ) + connection = http.client.HTTPConnection(*address, timeout=2) + connection.request( + "POST", + "/mcp", + body=json.dumps(message, separators=(",", ":")), + headers={ + "Content-Type": "application/json", + **modern_headers("subscriptions/listen"), + }, + ) + + response = connection.getresponse() + body = response.read() + connection.close() + + assert response.status == 200 + assert response.getheader("Content-Type") == "text/event-stream" + assert body.startswith(b": reviewer keepalive\rdata: ") + assert b"\n" not in body + events = [ + json.loads(line.removeprefix(b"data: ")) + for line in body.split(b"\r") + if line.startswith(b"data: ") + ] + assert [event.get("method") for event in events[:-1]] == [ + "notifications/subscriptions/acknowledged", + "notifications/tools/list_changed", + ] + assert events[-1]["id"] == 77 + assert events[-1]["result"]["resultType"] == "complete" + + +def test_review_http_streams_bounded_keepalive_flood_before_valid_sse_events() -> None: + with running_http_server( + MODERN_VERSION, profile=SSE_COMMENT_FLOOD_PROFILE + ) as address: + message = request( + "subscriptions/listen", + request_id=77, + params={ + "_meta": modern_meta(), + "notifications": {"toolsListChanged": True}, + }, + ) + connection = http.client.HTTPConnection(*address, timeout=10) + connection.request( + "POST", + "/mcp", + body=json.dumps(message, separators=(",", ":")), + headers={ + "Content-Type": "application/json", + **modern_headers("subscriptions/listen"), + }, + ) + + response = connection.getresponse() + body = response.read() + connection.close() + + assert response.status == 200 + assert response.getheader("Content-Type") == "text/event-stream" + assert response.getheader("Content-Length") == str(len(body)) + assert ( + REVIEW_SSE_EVENT_LIMIT_BYTES + < len(body) + < (REVIEW_SSE_EVENT_LIMIT_BYTES + REVIEW_SSE_COMMENT_LINE_BYTES + 4_096) + ) + assert body.count(b": reviewer keepalive ") == REVIEW_SSE_COMMENT_LINE_COUNT + assert b"\n" not in body + completed = json.loads(body.rsplit(b"data: ", 1)[1].removesuffix(b"\r\r")) + assert completed["id"] == 77 + assert completed["result"]["resultType"] == "complete" diff --git a/sdk/python/tests/test_mcp_conformance_fixtures.py b/sdk/python/tests/test_mcp_conformance_fixtures.py new file mode 100644 index 0000000000..57f6bc1592 --- /dev/null +++ b/sdk/python/tests/test_mcp_conformance_fixtures.py @@ -0,0 +1,37 @@ +import os +import subprocess +import sys +from pathlib import Path + + +def test_mcp_conformance_fixture_self_tests() -> None: + repository_root = Path(__file__).resolve().parents[3] + candidates = ( + repository_root / "public" / "scripts" / "mcp_conformance", + repository_root / "scripts" / "mcp_conformance", + ) + fixture_directory = next((candidate for candidate in candidates if candidate.is_dir()), None) + assert fixture_directory is not None, ( + "MCP conformance fixtures are missing; checked " + + " and ".join(str(candidate) for candidate in candidates) + ) + + env = os.environ.copy() + env["PYTEST_DISABLE_PLUGIN_AUTOLOAD"] = "1" + result = subprocess.run( + [sys.executable, "-m", "pytest", "-q", str(fixture_directory)], + cwd=repository_root, + env=env, + capture_output=True, + text=True, + timeout=120, + check=False, + ) + + assert result.returncode == 0, ( + f"MCP conformance fixture self-tests failed with exit " + f"{result.returncode}.\n" + f"Fixture directory: {fixture_directory}\n" + f"STDOUT:\n{result.stdout}\n" + f"STDERR:\n{result.stderr}" + ) diff --git a/sdk/typescript/package.json b/sdk/typescript/package.json index 13a7eca028..d75c8f0317 100644 --- a/sdk/typescript/package.json +++ b/sdk/typescript/package.json @@ -45,6 +45,7 @@ "prepare": "pnpm run build" }, "devDependencies": { + "@modelcontextprotocol/conformance": "github:modelcontextprotocol/conformance#49103de6ed70804e940637bf3e9e29e4a3f54e64", "@modelcontextprotocol/sdk": "^1.24.0", "@types/jest": "^29.5.14", "@types/node": "^20.19.18", diff --git a/sdk/typescript/tests/mcpConformance.test.ts b/sdk/typescript/tests/mcpConformance.test.ts new file mode 100644 index 0000000000..28ad908f07 --- /dev/null +++ b/sdk/typescript/tests/mcpConformance.test.ts @@ -0,0 +1,158 @@ +import { spawnSync } from "node:child_process"; +import { existsSync } from "node:fs"; +import { createRequire } from "node:module"; +import os from "node:os"; +import path from "node:path"; + +import { describe, it } from "@jest/globals"; + +import { codexExecPath } from "./testCodex"; + +const conformanceTimeoutMs = 300_000; +const reviewerRegressionTimeoutMs = 180_000; + +function requireFile(filePath: string, description: string): string { + if (!existsSync(filePath)) { + throw new Error(`MCP conformance ${description} does not exist: ${filePath}`); + } + + return filePath; +} + +function conformanceDirectory(): string { + const repositoryRoot = path.resolve(process.cwd(), "../.."); + const candidates = [ + path.join(repositoryRoot, "public", "scripts", "mcp_conformance"), + path.join(repositoryRoot, "scripts", "mcp_conformance"), + ]; + + for (const candidate of candidates) { + if (existsSync(path.join(candidate, "run_codex_compliance.py"))) { + return candidate; + } + } + + throw new Error(`MCP conformance harness is missing; checked ${candidates.join(" and ")}`); +} + +describe("MCP client conformance", () => { + it( + "does not introduce shipping, intermediate, or modern official conformance regressions", + () => { + const directory = conformanceDirectory(); + const require = createRequire(import.meta.url); + const conformancePackage = require.resolve("@modelcontextprotocol/conformance/package.json"); + const officialConformanceCli = requireFile( + path.join(path.dirname(conformancePackage), "dist", "index.js"), + "pinned official CLI", + ); + const codexBinary = requireFile(codexExecPath, "built Codex binary"); + const baseline = requireFile( + path.join(directory, "regression-baseline-v1.json"), + "committed regression baseline", + ); + const reportDirectory = process.env.RUNNER_TEMP ?? os.tmpdir(); + const reportPath = path.join(reportDirectory, `codex-mcp-conformance-${process.pid}.json`); + const result = spawnSync( + "python3", + [ + path.join(directory, "run_codex_compliance.py"), + codexBinary, + "--mode", + "all", + "--transport", + "all", + "--auth", + "--conformance-cli", + officialConformanceCli, + "--baseline-report", + baseline, + "--report", + reportPath, + ], + { + encoding: "utf8", + maxBuffer: 16 * 1024 * 1024, + timeout: conformanceTimeoutMs - 10_000, + }, + ); + + if (result.error) { + throw new Error(`MCP conformance runner could not complete: ${result.error.message}`, { + cause: result.error, + }); + } + + if (result.status !== 0) { + throw new Error( + [ + `MCP conformance regression gate exited with ${result.status ?? result.signal}.`, + `Report: ${reportPath}`, + result.stdout, + result.stderr, + ] + .filter(Boolean) + .join("\n"), + ); + } + }, + conformanceTimeoutMs, + ); + + it( + "does not introduce production reviewer or catalog-boundary regressions", + () => { + const directory = conformanceDirectory(); + const codexBinary = requireFile(codexExecPath, "built Codex binary"); + const reviewer = requireFile( + path.join(directory, "review_regressions.py"), + "production reviewer regression runner", + ); + const baseline = requireFile( + path.join(directory, "review-regression-baseline-v1.json"), + "committed production reviewer regression baseline", + ); + const reportDirectory = process.env.RUNNER_TEMP ?? os.tmpdir(); + const reportPath = path.join(reportDirectory, `codex-mcp-review-${process.pid}.json`); + const result = spawnSync( + "python3", + [ + reviewer, + codexBinary, + "--mode", + "all", + "--baseline-report", + baseline, + "--report", + reportPath, + ], + { + encoding: "utf8", + maxBuffer: 16 * 1024 * 1024, + timeout: reviewerRegressionTimeoutMs - 10_000, + }, + ); + + if (result.error) { + throw new Error( + `MCP production reviewer regression runner could not complete: ${result.error.message}`, + { cause: result.error }, + ); + } + + if (result.status !== 0) { + throw new Error( + [ + `MCP production reviewer regression gate exited with ${result.status ?? result.signal}.`, + `Report: ${reportPath}`, + result.stdout, + result.stderr, + ] + .filter(Boolean) + .join("\n"), + ); + } + }, + reviewerRegressionTimeoutMs, + ); +});