From 0bfaa099d02805907d3aea1d35458a5f5e2f42a1 Mon Sep 17 00:00:00 2001 From: speakeasybot Date: Tue, 11 Aug 2026 14:05:48 +0000 Subject: [PATCH 1/2] =?UTF-8?q?##=20Python=20SDK=20Changes:=20*=20`open=5F?= =?UTF-8?q?router.benchmarks.get=5Fbenchmarks()`:=20=20=20*=20=20`request`?= =?UTF-8?q?=20**Changed**=20=20=20*=20=20`response.data[]`=20**Changed**?= =?UTF-8?q?=20(Breaking=20=E2=9A=A0=EF=B8=8F)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .speakeasy/gen.lock | 116 ++++++++---- .speakeasy/gen.yaml | 2 +- .speakeasy/out.openapi.yaml | 174 +++++++++++++++++- .speakeasy/workflow.lock | 10 +- RELEASES.md | 12 +- docs/components/primarymetric.mdx | 22 +++ docs/components/searchsurface.mdx | 22 +++ docs/components/unifiedbenchmarksoritem.mdx | 22 +-- ... unifiedbenchmarksoritembenchmarktype.mdx} | 6 +- .../unifiedbenchmarksresponsedata.mdx | 6 + .../unifiedbenchmarkssearchitem.mdx | 21 +++ ...ifiedbenchmarkssearchitembenchmarktype.mdx | 24 +++ .../unifiedbenchmarkssearchitemsource.mdx | 17 ++ .../unifiedbenchmarkssearchrunconfig.mdx | 14 ++ docs/operations/benchmarktype.mdx | 26 +++ docs/operations/getbenchmarksrequest.mdx | 24 ++- docs/operations/searchsurface.mdx | 22 +++ docs/operations/tasktype.mdx | 3 +- docs/sdks/benchmarks/README.mdx | 30 +-- pyproject.toml | 2 +- src/openrouter/_version.py | 4 +- src/openrouter/benchmarks.py | 32 +++- src/openrouter/components/__init__.py | 37 +++- .../components/unifiedbenchmarksoritem.py | 6 +- .../components/unifiedbenchmarksresponse.py | 48 ++--- .../components/unifiedbenchmarkssearchitem.py | 142 ++++++++++++++ .../unifiedbenchmarkssearchrunconfig.py | 44 +++++ src/openrouter/operations/__init__.py | 6 + src/openrouter/operations/getbenchmarks.py | 67 ++++++- uv.lock | 2 +- 30 files changed, 817 insertions(+), 146 deletions(-) create mode 100644 docs/components/primarymetric.mdx create mode 100644 docs/components/searchsurface.mdx rename docs/components/{benchmarktype.mdx => unifiedbenchmarksoritembenchmarktype.mdx} (61%) create mode 100644 docs/components/unifiedbenchmarkssearchitem.mdx create mode 100644 docs/components/unifiedbenchmarkssearchitembenchmarktype.mdx create mode 100644 docs/components/unifiedbenchmarkssearchitemsource.mdx create mode 100644 docs/components/unifiedbenchmarkssearchrunconfig.mdx create mode 100644 docs/operations/benchmarktype.mdx create mode 100644 docs/operations/searchsurface.mdx create mode 100644 src/openrouter/components/unifiedbenchmarkssearchitem.py create mode 100644 src/openrouter/components/unifiedbenchmarkssearchrunconfig.py diff --git a/.speakeasy/gen.lock b/.speakeasy/gen.lock index a243234c..9a815164 100644 --- a/.speakeasy/gen.lock +++ b/.speakeasy/gen.lock @@ -1,19 +1,19 @@ lockVersion: 2.0.0 id: c48cf606-fb42-4a45-9c23-8f0555307828 management: - docChecksum: 75d736c349cb105e920d684fd2506fca + docChecksum: e408687b6809c0eb28ce2608b3c9d155 docVersion: 1.0.0 speakeasyVersion: 1.787.0 generationVersion: 2.914.0 - releaseVersion: 1.1.43 - configChecksum: cccf7b2204d0f401df2776c4b1e3a363 + releaseVersion: 1.1.44 + configChecksum: 58e13146341fef6b0ea6ae8882827d37 repoURL: https://github.com/OpenRouterTeam/python-sdk.git installationURL: https://github.com/OpenRouterTeam/python-sdk.git published: true persistentEdits: - generation_id: 229674de-e3e7-4794-8be0-367148eb3a30 - pristine_commit_hash: af2265f9d619fb2150d3848f261d729d66277467 - pristine_tree_hash: 663823946e41b46acc3200e982caa55cb1d0eafa + generation_id: f961c45e-5096-4334-bf45-ef26d2b749e8 + pristine_commit_hash: ba9a7f326ed5f0c2a84c6b81ced43935892e8629 + pristine_tree_hash: e87ac6459017910906d4f2566ca55e1cebef6340 features: python: acceptHeaders: 3.0.0 @@ -760,10 +760,6 @@ trackedFiles: id: 73f42e4bc945 last_write_checksum: sha1:599705c7ef520285d8da9e5510109846aa7fd332 pristine_git_object: 539a7c32200bef803260226e81f3225e9529688e - docs/components/benchmarktype.mdx: - id: c191e897ed61 - last_write_checksum: sha1:819054a8e4d3096134355828f732c73e90fa2d67 - pristine_git_object: 2a473fdfeeccc38689147c863d1f857560a97ab4 docs/components/billable.mdx: id: b983fa757056 last_write_checksum: sha1:80d142a13e98a8f13da2850753282adf2ff59521 @@ -4260,6 +4256,10 @@ trackedFiles: id: 74675f38c1c6 last_write_checksum: sha1:ca4bf606df26dd9e2b636d7d9e9f0fd183b21145 pristine_git_object: 5fb1b765b2b0e46d40042f341bbcc91aa0d92762 + docs/components/primarymetric.mdx: + id: c6ec097bacaf + last_write_checksum: sha1:fc7d89ac02ff585424b5bc8cc9b0ac82338c7b6f + pristine_git_object: 6a99a2737d18a3c227bf7fcabf58b481d3c7d94b docs/components/promptcachebreakpoint.mdx: id: c135956dc623 last_write_checksum: sha1:6df345ea99da1fdcae22895faa2ee36339ba7332 @@ -4688,6 +4688,10 @@ trackedFiles: id: 0cbced049b6f last_write_checksum: sha1:106b7467c7ab738c7ce04ab44d83a91388dc0808 pristine_git_object: 09c661196e0bbf039ff1e117c55aa1981ca1ef4a + docs/components/searchsurface.mdx: + id: 4bcafd219190 + last_write_checksum: sha1:f0c602a940edebf01020c33cb012245082040338 + pristine_git_object: 9ef3f90e268fa74f1f6081afbdae17cc178fb850 docs/components/security.mdx: id: 126e702b7c97 last_write_checksum: sha1:3fbb5c3145b96472437de6e6384ae222b0ae5210 @@ -5358,8 +5362,12 @@ trackedFiles: pristine_git_object: 23ef4c9a45aea5d918e554f7868f772429ef364c docs/components/unifiedbenchmarksoritem.mdx: id: 642fc71c5d26 - last_write_checksum: sha1:5c136345f1f69e6f8db7269a31050aac5ada0894 - pristine_git_object: 99228c0beefaf20bb7b0134946a75404c9d1d0b6 + last_write_checksum: sha1:2543b293158a67f93f19ee23331483cb56893c5b + pristine_git_object: ec2f59459e9fb089bb24c122a6a205da6c37de28 + docs/components/unifiedbenchmarksoritembenchmarktype.mdx: + id: dad93e1758c9 + last_write_checksum: sha1:7d0430ad93a25a95c8de517749ec1f89d0ca4e1b + pristine_git_object: 18316442f3346d326cf4ccbc453cf26f36a5fd74 docs/components/unifiedbenchmarksoritemsource.mdx: id: b89557fa192a last_write_checksum: sha1:37dbe5e1fc82f2dd6983383ad74f82048c367062 @@ -5370,8 +5378,24 @@ trackedFiles: pristine_git_object: ee48e645716b05d6473f5ba0199fb84264a44d12 docs/components/unifiedbenchmarksresponsedata.mdx: id: fd372e1c57e8 - last_write_checksum: sha1:303a04665c6586ab8cd3662ff5640d16db8e6c53 - pristine_git_object: ff4c06716df56052ed05ba3a5ac7e4c01d5ea27f + last_write_checksum: sha1:4606f3c0a2f1be4ddb0558960b8da50657255e23 + pristine_git_object: 50ec3eb6b4bc4c414a1fc79603a00939ae71f976 + docs/components/unifiedbenchmarkssearchitem.mdx: + id: 426474b17840 + last_write_checksum: sha1:5e2b83ba24cf9c6073771e0a99fc98100dc613df + pristine_git_object: 40b4f0bea4b0b07b3ce2e5b7801f106eacc18d2a + docs/components/unifiedbenchmarkssearchitembenchmarktype.mdx: + id: 12cfb7df44cb + last_write_checksum: sha1:0555a969885d6080df39854a57950edcbfc3661d + pristine_git_object: dfd7cc2da5eb13fb9153fd9cbef2c8ec2642a4a9 + docs/components/unifiedbenchmarkssearchitemsource.mdx: + id: 2be77d920124 + last_write_checksum: sha1:e00a6c233f2b3bcc3787c3a8943611d340ddef1d + pristine_git_object: cf189d077669409efa26fcc6d1a6570298153978 + docs/components/unifiedbenchmarkssearchrunconfig.mdx: + id: 0646f31a49fb + last_write_checksum: sha1:bedb0fdd46bfdda0deacab2042b304e06cd49296 + pristine_git_object: 21f463fd8c660165b0533c8a883f43dd2fd3e961 docs/components/uniqueinsight.mdx: id: 02fd71bcb47a last_write_checksum: sha1:793e90c78ab450e36533710d3fad6f4a75432e86 @@ -5728,6 +5752,10 @@ trackedFiles: id: fd0c961ea26d last_write_checksum: sha1:48d0112803ed464cb9f9fe06d4b58f982f936522 pristine_git_object: 455ab607b237f29dd06d6548926f8ee08be410aa + docs/operations/benchmarktype.mdx: + id: 7de894bb7280 + last_write_checksum: sha1:12f34fb51ed9e31b78aa69307fcf42e6f99b4f31 + pristine_git_object: 390b7aca29e54383b436ea5a6e0201c8765001f2 docs/operations/bulkaddworkspacemembersglobals.mdx: id: f198abbe4de0 last_write_checksum: sha1:ca25e8bae0fa1b09844c8ed66e174d7a8c06542d @@ -6218,8 +6246,8 @@ trackedFiles: pristine_git_object: 679923bfbff14ca3bc63ce5d451089b1996e5c6e docs/operations/getbenchmarksrequest.mdx: id: 704fb5d64a7c - last_write_checksum: sha1:506d8cf7a2eed421ae52c60703e3400cdfa3a652 - pristine_git_object: c5acba5bb9a6a2ae287d03f0aef8da1bc49df8e6 + last_write_checksum: sha1:a657060a69596b6542cf6fb972ac336d756d9afc + pristine_git_object: e45dab262f662cd66690cfe35511957422db2e7c docs/operations/getbyokkeyglobals.mdx: id: ac592fd40012 last_write_checksum: sha1:1eac912896162f772846e2fe5e6f3b50ae8d20f2 @@ -6872,6 +6900,10 @@ trackedFiles: id: ee0d3c3547d7 last_write_checksum: sha1:11ebdfcfbf9a1197a42516d3daa0eaa3e2955f81 pristine_git_object: 9e1219b8c4da58c9ee1d0707aa809d3d3b54f372 + docs/operations/searchsurface.mdx: + id: 2d463024ea29 + last_write_checksum: sha1:f612f65bd78b46bd424a0bbbbfb6260331738acc + pristine_git_object: cf8fe4f0a66b46ea069704ab9f791e3c2a4adb22 docs/operations/sendchatcompletionrequestglobals.mdx: id: 8ad074be0768 last_write_checksum: sha1:93462ef7b25e039e770050019663f6b41dbf4187 @@ -6902,8 +6934,8 @@ trackedFiles: pristine_git_object: 18f5cc3e2fe30014037f8f68d5513a4981d513b9 docs/operations/tasktype.mdx: id: d3da8106d2d1 - last_write_checksum: sha1:06aede3448490e1fb77b5c1b758238a11ce1dceb - pristine_git_object: 7cdaa5db9c5d84013c3aeeaf1f7493c1768c1fef + last_write_checksum: sha1:0955731aba6ddddeeaee3b34ca7e353dc2dd9700 + pristine_git_object: ca10bf29d177ce5724fe59c84d929b20e40781c6 docs/operations/timerange.mdx: id: 03a218211f06 last_write_checksum: sha1:3cdcfbfcdb71dcc17a0c3b4cd5704eac3718fa0f @@ -7046,8 +7078,8 @@ trackedFiles: pristine_git_object: c8541c5c189bae01e80db73fd88767391d7dbb9b docs/sdks/benchmarks/README.mdx: id: 5b483a6770ba - last_write_checksum: sha1:2d935127bdbf18d1237b3e3af23f9145216806b2 - pristine_git_object: da27ed830fab524c5b9bf2c72031749bb37ed365 + last_write_checksum: sha1:885b70fcbd2be60f51a2f58e133fd6452eb21600 + pristine_git_object: 5026ac2a283240480026d80d2f77d5528ae8bdc8 docs/sdks/betaanalytics/README.mdx: id: 239279ebf01b last_write_checksum: sha1:5a76d28d77f68efe34ebbcb72850e6a87c74d3b1 @@ -7158,8 +7190,8 @@ trackedFiles: pristine_git_object: 3e38f1a929f7d6b1d6de74604aa87e3d8f010544 pyproject.toml: id: 5d07e7d72637 - last_write_checksum: sha1:5fc67580b2031b06ee7e749fe9f41554b646b425 - pristine_git_object: dd8146be33b81f978739b06319b61584fc37cabe + last_write_checksum: sha1:18256d8b9cc609a66b10b73bc29e1f89e71a5ce1 + pristine_git_object: 2633c255582f9f232999f764a31f1a5d170c712f scripts/prepare_readme.py: id: e0c5957a6035 last_write_checksum: sha1:77f44b60b98bc126557ec27391f91dfba764bb54 @@ -7186,8 +7218,8 @@ trackedFiles: pristine_git_object: 86713cfea633e09d33b3d4e65281071fe20e6137 src/openrouter/_version.py: id: d8d15ad6c586 - last_write_checksum: sha1:37abc296599992615a7d4aa8c880825df579140b - pristine_git_object: 5f253b27c28788047892ef19fc7f069c3799b795 + last_write_checksum: sha1:bf92de0c3824a4280f6bb520e285a00d20188ed7 + pristine_git_object: 055275608059735910d646e0ada351d41f4a6017 src/openrouter/analytics.py: id: cb406b5aaabb last_write_checksum: sha1:3ea0f1c73fb9c101b7bc5719da4dd7bf62ff469a @@ -7202,8 +7234,8 @@ trackedFiles: pristine_git_object: 7a562e21c7e66c9db666ead83748b73adefe4278 src/openrouter/benchmarks.py: id: 178d2ad8d706 - last_write_checksum: sha1:c5a8d0fa780efff4f9f81a12e102cb82d825eb57 - pristine_git_object: 5dcb0e0427dd7e820662fb44bed1099420be25dd + last_write_checksum: sha1:15b8a7ea05a7550032afb15981958aa73b756881 + pristine_git_object: f163fcbb9c21fc1327d6c9a1eefcc60cbf9091ea src/openrouter/beta.py: id: fffdf54fd8f5 last_write_checksum: sha1:4e34fb96ffe38673ca72e0f8576a8eddd4134d4b @@ -7230,8 +7262,8 @@ trackedFiles: pristine_git_object: ad3d247954547814054c01989a2dff3d12b3e4e1 src/openrouter/components/__init__.py: id: 81754e97b3f4 - last_write_checksum: sha1:f60cd4f5bc011539964e984662bf4de255c22493 - pristine_git_object: 670eeb2dbe8c686babcc1bf5dd23cd0d572fedff + last_write_checksum: sha1:7fc8366c0da24c31337bded97df0f1c8f9cdb173 + pristine_git_object: f714cc3337bfe431ca2007e9511a81d970dd4e62 src/openrouter/components/aabenchmarkentry.py: id: e2e0f0b48c82 last_write_checksum: sha1:fab4d9a24d2cea937bb749d46c5f83941e99d65c @@ -9470,12 +9502,20 @@ trackedFiles: pristine_git_object: 68d76b7b6dbd1817f825e38902556719b179015d src/openrouter/components/unifiedbenchmarksoritem.py: id: 8da361fd40c8 - last_write_checksum: sha1:2f375462570405680ccf18a7e265c3e25fc67dd8 - pristine_git_object: c58ba941246a96890970e6eb1b53588600b2addd + last_write_checksum: sha1:f946862e244121ae6f27017a5f8652fc3f964d67 + pristine_git_object: c1671f3ef45eaaf719dc7d29b2f6a3824fb98934 src/openrouter/components/unifiedbenchmarksresponse.py: id: 4f7bbccaba03 - last_write_checksum: sha1:2eedc8ad728768805d1bc504caa30989c751ce1f - pristine_git_object: 311cb86d0cc7f42bd9f9842e3783517c182ffe4f + last_write_checksum: sha1:a7f33321910d7e05e31f9533da54ee2706250aec + pristine_git_object: e2ed5fb98fd78b6e9a6184f467bd80617ca9371f + src/openrouter/components/unifiedbenchmarkssearchitem.py: + id: cc77d4dc2470 + last_write_checksum: sha1:7dadc353af8a9c3b803e69da3d4c7fd840403eae + pristine_git_object: f0b74435764621178fa70a138da1419aa1eaf245 + src/openrouter/components/unifiedbenchmarkssearchrunconfig.py: + id: 189239ce6b4f + last_write_checksum: sha1:9d85897b7857707fb6a759aaa938f4b5bf5662c2 + pristine_git_object: e1b94da161012480960c726a459b3fbd7c1a9bf4 src/openrouter/components/unprocessableentityresponseerrordata.py: id: e8ca4a51f994 last_write_checksum: sha1:e97c277b49f4bb6bcfad8eef8a46b382730729a0 @@ -9790,8 +9830,8 @@ trackedFiles: pristine_git_object: 5c553b76ab24d9eaef51c3bf6d85aa74391e335e src/openrouter/operations/__init__.py: id: 9afcea1e7161 - last_write_checksum: sha1:7462feea63bad287e4382464c6a5bf24b401eaa2 - pristine_git_object: 37c3c8ce47455812f19b303f0b4d34738ee1237a + last_write_checksum: sha1:1a29771015fee7983b677eb476aeb281fad9c2dc + pristine_git_object: 50ebc6f58e44ee099607fa6140232132a7990767 src/openrouter/operations/bulkaddworkspacemembers.py: id: e0ed56117619 last_write_checksum: sha1:5c44eb0d40fdece3ac084615f6c6082be4cf1d5a @@ -9938,8 +9978,8 @@ trackedFiles: pristine_git_object: a8e00f731e950d64ec8ca7ab10f6fca96884c8e7 src/openrouter/operations/getbenchmarks.py: id: 5fb88644491e - last_write_checksum: sha1:2df9c948624b4126685e5347e5c9442d2a05945d - pristine_git_object: 62573095f6d816050964d68fb35f61b570acd5d5 + last_write_checksum: sha1:c059365dfa135f5d6d929464c06af3421cf89c62 + pristine_git_object: 51f1f39f8f61ebb2edc9770ad79e09bd7886fd2d src/openrouter/operations/getbyokkey.py: id: d141452bd88a last_write_checksum: sha1:d6d391b230d90f51945144e63e9c53772650bd7b @@ -11796,6 +11836,9 @@ examples: application/json: {"error": {"code": 500, "message": "Internal Server Error"}} getBenchmarks: speakeasy-default-get-benchmarks: + parameters: + query: + include_run_config: true responses: "200": application/json: {"data": [{"agentic_index": 58.3, "coding_index": 65.8, "display_name": "GPT-4o", "intelligence_index": 71.2, "model_permaslug": "openai/gpt-4o", "pricing": {"completion": "0.00001", "prompt": "0.0000025"}, "source": "artificial-analysis"}, {"accuracy": 0.72, "accuracy_stddev": 0.03, "avg_cost_per_task": 0.002, "benchmark_type": "gpqa_diamond", "display_name": "GPT-4o", "last_run_timestamp": "2026-06-03T12:00:00Z", "model_permaslug": "openai/gpt-4o", "source": "openrouter", "total_tasks": 300}], "meta": {"as_of": "2026-06-03T12:00:00Z", "citation": null, "model_count": 1, "source": null, "source_url": null, "task_type": null, "version": "v1"}} @@ -12023,3 +12066,4 @@ examples: "500": application/json: {"error": {"code": 500, "message": "Internal Server Error"}} examplesVersion: 1.0.2 +releaseNotes: "## Python SDK Changes:\n* `open_router.benchmarks.get_benchmarks()`: \n * `request` **Changed**\n * `response.data[]` **Changed** (Breaking ⚠️)\n" diff --git a/.speakeasy/gen.yaml b/.speakeasy/gen.yaml index b0b0cd37..7066bde3 100644 --- a/.speakeasy/gen.yaml +++ b/.speakeasy/gen.yaml @@ -36,7 +36,7 @@ generation: documentation: mintlify preApplyUnionDiscriminators: true python: - version: 1.1.43 + version: 1.1.44 additionalDependencies: dev: {} main: {} diff --git a/.speakeasy/out.openapi.yaml b/.speakeasy/out.openapi.yaml index 23377f0b..6244780a 100644 --- a/.speakeasy/out.openapi.yaml +++ b/.speakeasy/out.openapi.yaml @@ -24472,16 +24472,11 @@ components: properties: data: items: - discriminator: - mapping: - artificial-analysis: '#/components/schemas/UnifiedBenchmarksAAItem' - design-arena: '#/components/schemas/UnifiedBenchmarksDAItem' - openrouter: '#/components/schemas/UnifiedBenchmarksORItem' - propertyName: 'source' oneOf: - $ref: '#/components/schemas/UnifiedBenchmarksAAItem' - $ref: '#/components/schemas/UnifiedBenchmarksDAItem' - $ref: '#/components/schemas/UnifiedBenchmarksORItem' + - $ref: '#/components/schemas/UnifiedBenchmarksSearchItem' type: 'array' meta: $ref: '#/components/schemas/UnifiedBenchmarksMeta' @@ -24489,6 +24484,121 @@ components: - 'data' - 'meta' type: 'object' + UnifiedBenchmarksSearchItem: + properties: + avg_cost_per_task: + description: 'Average cost per task in USD, or null if unavailable.' + example: 0.031 + format: 'double' + type: + - 'number' + - 'null' + avg_latency_per_task_ms: + description: 'Average wall-clock latency per task in milliseconds, or null if unavailable.' + example: 45210 + format: 'double' + type: + - 'number' + - 'null' + benchmark_type: + description: 'OpenRouter search benchmark.' + enum: + - 'search_browsecomp' + - 'search_hle' + - 'search_dsqa' + - 'search_widesearch' + example: 'search_browsecomp' + type: 'string' + x-speakeasy-unknown-values: allow + display_name: + description: 'Human-readable model name.' + example: 'GPT-4o' + type: 'string' + last_run_timestamp: + description: 'Timestamp of the newest qualifying run in the published configuration''s lane.' + example: '2026-07-28T13:38:18Z' + type: 'string' + model_permaslug: + description: 'Stable OpenRouter model identifier.' + example: 'openai/gpt-4o' + type: 'string' + primary_metric: + description: 'Identifies the meaning of `primary_score`: `f1_by_item` for WideSearch, `accuracy` for all other search benchmarks.' + enum: + - 'accuracy' + - 'f1_by_item' + example: 'accuracy' + type: 'string' + x-speakeasy-unknown-values: allow + primary_score: + description: 'The benchmark''s headline score from 0 to 1. Its meaning is identified by `primary_metric`: item-weighted F1 for WideSearch or strict accuracy for the other search benchmarks. Higher is better.' + example: 0.72 + format: 'double' + type: 'number' + run_config: + $ref: '#/components/schemas/UnifiedBenchmarksSearchRunConfig' + search_engine: + description: 'Search engine the published configuration used.' + example: 'exa' + type: 'string' + search_surface: + description: 'Request surface the published configuration went through.' + enum: + - 'server-tool' + - 'plugin' + example: 'server-tool' + type: 'string' + x-speakeasy-unknown-values: allow + source: + description: 'Benchmark source discriminator.' + enum: + - 'openrouter' + type: 'string' + total_tasks: + description: 'Tasks evaluated across the published configuration''s runs.' + example: 100 + type: 'integer' + required: + - 'source' + - 'model_permaslug' + - 'display_name' + - 'benchmark_type' + - 'primary_metric' + - 'primary_score' + - 'total_tasks' + - 'avg_cost_per_task' + - 'avg_latency_per_task_ms' + - 'search_engine' + - 'search_surface' + - 'last_run_timestamp' + type: 'object' + UnifiedBenchmarksSearchRunConfig: + description: 'Published lane configuration, included only when include_run_config=true. Only the agent turn count, reasoning effort, and temperature are exposed; other harness settings are intentionally not part of the public contract.' + properties: + max_agent_turns: + description: 'Agent-turn count for the published lane, or null for plugin lanes.' + example: 25 + type: + - 'integer' + - 'null' + reasoning_effort: + description: 'Reasoning effort configured for the published lane, or null when omitted.' + example: 'high' + type: + - 'string' + - 'null' + temperature: + description: 'Sampling temperature configured for the published lane, or null when omitted.' + example: 0.2 + format: 'double' + type: + - 'number' + - 'null' + required: + - 'max_agent_turns' + - 'reasoning_effort' + - 'temperature' + type: 'object' UnprocessableEntityResponse: description: 'Unprocessable Entity - Semantic validation failure' example: @@ -27379,7 +27489,7 @@ paths: - $ref: "#/components/parameters/AppCategories" /benchmarks: get: - description: 'Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter''s own tau-bench and GPQA evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account.' + description: 'Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter''s own tau-bench, GPQA, and web-search evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Use task_type=search (or a search_* benchmark_type) for OpenRouter''s search benchmarks, which publish each model''s highest-scoring eligible evaluation configuration with same-configuration runs combined by task-weighted mean. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account.' operationId: 'getBenchmarks' parameters: - description: 'Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources.' @@ -27395,19 +27505,65 @@ paths: example: 'artificial-analysis' type: 'string' x-speakeasy-unknown-values: allow - - description: 'Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category.' + - description: 'Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only.' in: 'query' name: 'task_type' required: false schema: - description: 'Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category.' + description: 'Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only.' enum: - 'coding' - 'intelligence' - 'agentic' + - 'search' example: 'coding' type: 'string' x-speakeasy-unknown-values: allow + - description: 'Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources'' items as they are.' + in: 'query' + name: 'benchmark_type' + required: false + schema: + description: 'Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources'' items as they are.' + enum: + - 'gpqa_diamond' + - 'tau_bench_verified_airline' + - 'search_browsecomp' + - 'search_hle' + - 'search_dsqa' + - 'search_widesearch' + example: 'search_widesearch' + type: 'string' + x-speakeasy-unknown-values: allow + - description: 'Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract.' + in: 'query' + name: 'include_run_config' + required: false + schema: + default: false + description: 'Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract.' + example: true + type: 'boolean' + - description: 'OpenRouter search benchmarks only: filter by the search engine used.' + in: 'query' + name: 'search_engine' + required: false + schema: + description: 'OpenRouter search benchmarks only: filter by the search engine used.' + example: 'exa' + type: 'string' + - description: 'OpenRouter search benchmarks only: filter by the request surface the lane ran on.' + in: 'query' + name: 'search_surface' + required: false + schema: + description: 'OpenRouter search benchmarks only: filter by the request surface the lane ran on.' + enum: + - 'server-tool' + - 'plugin' + example: 'server-tool' + type: 'string' + x-speakeasy-unknown-values: allow - description: 'Design Arena only: arena to query. Defaults to `models` when source is `design-arena`.' in: 'query' name: 'arena' diff --git a/.speakeasy/workflow.lock b/.speakeasy/workflow.lock index a3ebe171..757af635 100644 --- a/.speakeasy/workflow.lock +++ b/.speakeasy/workflow.lock @@ -2,8 +2,8 @@ speakeasyVersion: 1.787.0 sources: OpenRouter API: sourceNamespace: open-router-chat-completions-api - sourceRevisionDigest: sha256:c4947c72f13c5189bdfaf2524d748e70ddad983f99c52a513e72a0cf92981788 - sourceBlobDigest: sha256:d38579d85cc5d56410b05042748d7d0e0a1dd730ca2e0af8ef3fe13b67c00372 + sourceRevisionDigest: sha256:0f0480623513ea8eb6966a24c5e7ea07750f656ec7a181a95c05b0688a23e203 + sourceBlobDigest: sha256:86789908c3b9752a073bf734375a2597651a71927911c6b9470089ada3d6d255 tags: - latest - 1.0.0 @@ -11,10 +11,10 @@ targets: open-router: source: OpenRouter API sourceNamespace: open-router-chat-completions-api - sourceRevisionDigest: sha256:c4947c72f13c5189bdfaf2524d748e70ddad983f99c52a513e72a0cf92981788 - sourceBlobDigest: sha256:d38579d85cc5d56410b05042748d7d0e0a1dd730ca2e0af8ef3fe13b67c00372 + sourceRevisionDigest: sha256:0f0480623513ea8eb6966a24c5e7ea07750f656ec7a181a95c05b0688a23e203 + sourceBlobDigest: sha256:86789908c3b9752a073bf734375a2597651a71927911c6b9470089ada3d6d255 codeSamplesNamespace: open-router-python-code-samples - codeSamplesRevisionDigest: sha256:431abf36979e4683c5acf6341ee055a3632f3ec81a916d560a120295beab9107 + codeSamplesRevisionDigest: sha256:96ade224222d016e3a3e25dcf397a3b3a191917caa45946af874d0f9e9cb4d91 workflow: workflowVersion: 1.0.0 speakeasyVersion: 1.787.0 diff --git a/RELEASES.md b/RELEASES.md index 1009d3ec..72c3e00f 100644 --- a/RELEASES.md +++ b/RELEASES.md @@ -1219,4 +1219,14 @@ Based on: ### Generated - [python v1.1.43] . ### Releases -- [PyPI v1.1.43] https://pypi.org/project/openrouter/1.1.43 - . \ No newline at end of file +- [PyPI v1.1.43] https://pypi.org/project/openrouter/1.1.43 - . + +## 2026-08-11 14:03:44 +### Changes +Based on: +- OpenAPI Doc +- Speakeasy CLI 1.787.0 (2.914.0) https://github.com/speakeasy-api/speakeasy +### Generated +- [python v1.1.44] . +### Releases +- [PyPI v1.1.44] https://pypi.org/project/openrouter/1.1.44 - . \ No newline at end of file diff --git a/docs/components/primarymetric.mdx b/docs/components/primarymetric.mdx new file mode 100644 index 00000000..6a99a273 --- /dev/null +++ b/docs/components/primarymetric.mdx @@ -0,0 +1,22 @@ +--- +title: "PrimaryMetric" +--- + +Identifies the meaning of `primary_score`: `f1_by_item` for WideSearch, `accuracy` for all other search benchmarks. + +## Example Usage + +```python +from openrouter.components import PrimaryMetric + +# Open enum: unrecognized values are captured as UnrecognizedStr +value: PrimaryMetric = "accuracy" +``` + + +## Values + +This is an open enum. Unrecognized values will not fail type checks. + +- `"accuracy"` +- `"f1_by_item"` diff --git a/docs/components/searchsurface.mdx b/docs/components/searchsurface.mdx new file mode 100644 index 00000000..9ef3f90e --- /dev/null +++ b/docs/components/searchsurface.mdx @@ -0,0 +1,22 @@ +--- +title: "SearchSurface" +--- + +Request surface the published configuration went through. + +## Example Usage + +```python +from openrouter.components import SearchSurface + +# Open enum: unrecognized values are captured as UnrecognizedStr +value: SearchSurface = "server-tool" +``` + + +## Values + +This is an open enum. Unrecognized values will not fail type checks. + +- `"server-tool"` +- `"plugin"` diff --git a/docs/components/unifiedbenchmarksoritem.mdx b/docs/components/unifiedbenchmarksoritem.mdx index 99228c0b..ec2f5945 100644 --- a/docs/components/unifiedbenchmarksoritem.mdx +++ b/docs/components/unifiedbenchmarksoritem.mdx @@ -4,14 +4,14 @@ title: "UnifiedBenchmarksORItem" ## Fields -| Field | Type | Required | Description | Example | -| ------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------ | -| `accuracy` | *float* | :heavy_check_mark: | Aggregate accuracy score from 0 to 1. Higher is better. | 0.72 | -| `accuracy_stddev` | *Nullable[float]* | :heavy_check_mark: | Standard deviation of run accuracy, or null for a single run. | 0.03 | -| `avg_cost_per_task` | *Nullable[float]* | :heavy_check_mark: | Average cost per task in USD, or null if unavailable. | 0.002 | -| `benchmark_type` | [components.BenchmarkType](../components/benchmarktype.mdx) | :heavy_check_mark: | OpenRouter benchmark evaluation type. | gpqa_diamond | -| `display_name` | *str* | :heavy_check_mark: | Human-readable model name. | GPT-4o | -| `last_run_timestamp` | *str* | :heavy_check_mark: | Timestamp of the most recent public benchmark run. | 2026-06-03T12:00:00Z | -| `model_permaslug` | *str* | :heavy_check_mark: | Stable OpenRouter model identifier. | openai/gpt-4o | -| `source` | [components.UnifiedBenchmarksORItemSource](../components/unifiedbenchmarksoritemsource.mdx) | :heavy_check_mark: | Benchmark source discriminator. | | -| `total_tasks` | *int* | :heavy_check_mark: | Total benchmark tasks across runs. | 300 | \ No newline at end of file +| Field | Type | Required | Description | Example | +| -------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------- | +| `accuracy` | *float* | :heavy_check_mark: | Aggregate accuracy score from 0 to 1. Higher is better. | 0.72 | +| `accuracy_stddev` | *Nullable[float]* | :heavy_check_mark: | Standard deviation of run accuracy, or null for a single run. | 0.03 | +| `avg_cost_per_task` | *Nullable[float]* | :heavy_check_mark: | Average cost per task in USD, or null if unavailable. | 0.002 | +| `benchmark_type` | [components.UnifiedBenchmarksORItemBenchmarkType](../components/unifiedbenchmarksoritembenchmarktype.mdx) | :heavy_check_mark: | OpenRouter benchmark evaluation type. | gpqa_diamond | +| `display_name` | *str* | :heavy_check_mark: | Human-readable model name. | GPT-4o | +| `last_run_timestamp` | *str* | :heavy_check_mark: | Timestamp of the most recent public benchmark run. | 2026-06-03T12:00:00Z | +| `model_permaslug` | *str* | :heavy_check_mark: | Stable OpenRouter model identifier. | openai/gpt-4o | +| `source` | [components.UnifiedBenchmarksORItemSource](../components/unifiedbenchmarksoritemsource.mdx) | :heavy_check_mark: | Benchmark source discriminator. | | +| `total_tasks` | *int* | :heavy_check_mark: | Total benchmark tasks across runs. | 300 | \ No newline at end of file diff --git a/docs/components/benchmarktype.mdx b/docs/components/unifiedbenchmarksoritembenchmarktype.mdx similarity index 61% rename from docs/components/benchmarktype.mdx rename to docs/components/unifiedbenchmarksoritembenchmarktype.mdx index 2a473fdf..18316442 100644 --- a/docs/components/benchmarktype.mdx +++ b/docs/components/unifiedbenchmarksoritembenchmarktype.mdx @@ -1,5 +1,5 @@ --- -title: "BenchmarkType" +title: "UnifiedBenchmarksORItemBenchmarkType" --- OpenRouter benchmark evaluation type. @@ -7,10 +7,10 @@ OpenRouter benchmark evaluation type. ## Example Usage ```python -from openrouter.components import BenchmarkType +from openrouter.components import UnifiedBenchmarksORItemBenchmarkType # Open enum: unrecognized values are captured as UnrecognizedStr -value: BenchmarkType = "gpqa_diamond" +value: UnifiedBenchmarksORItemBenchmarkType = "gpqa_diamond" ``` diff --git a/docs/components/unifiedbenchmarksresponsedata.mdx b/docs/components/unifiedbenchmarksresponsedata.mdx index ff4c0671..50ec3eb6 100644 --- a/docs/components/unifiedbenchmarksresponsedata.mdx +++ b/docs/components/unifiedbenchmarksresponsedata.mdx @@ -22,3 +22,9 @@ value: components.UnifiedBenchmarksDAItem = /* values here */ value: components.UnifiedBenchmarksORItem = /* values here */ ``` +### `components.UnifiedBenchmarksSearchItem` + +```python +value: components.UnifiedBenchmarksSearchItem = /* values here */ +``` + diff --git a/docs/components/unifiedbenchmarkssearchitem.mdx b/docs/components/unifiedbenchmarkssearchitem.mdx new file mode 100644 index 00000000..40b4f0be --- /dev/null +++ b/docs/components/unifiedbenchmarkssearchitem.mdx @@ -0,0 +1,21 @@ +--- +title: "UnifiedBenchmarksSearchItem" +--- + +## Fields + +| Field | Type | Required | Description | Example | +| ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `avg_cost_per_task` | *Nullable[float]* | :heavy_check_mark: | Average cost per task in USD, or null if unavailable. | 0.031 | +| `avg_latency_per_task_ms` | *Nullable[float]* | :heavy_check_mark: | Average wall-clock latency per task in milliseconds, or null if unavailable. | 45210 | +| `benchmark_type` | [components.UnifiedBenchmarksSearchItemBenchmarkType](../components/unifiedbenchmarkssearchitembenchmarktype.mdx) | :heavy_check_mark: | OpenRouter search benchmark. | search_browsecomp | +| `display_name` | *str* | :heavy_check_mark: | Human-readable model name. | GPT-4o | +| `last_run_timestamp` | *str* | :heavy_check_mark: | Timestamp of the newest qualifying run in the published configuration's lane. | 2026-07-28T13:38:18Z | +| `model_permaslug` | *str* | :heavy_check_mark: | Stable OpenRouter model identifier. | openai/gpt-4o | +| `primary_metric` | [components.PrimaryMetric](../components/primarymetric.mdx) | :heavy_check_mark: | Identifies the meaning of `primary_score`: `f1_by_item` for WideSearch, `accuracy` for all other search benchmarks. | accuracy | +| `primary_score` | *float* | :heavy_check_mark: | The benchmark's headline score from 0 to 1. Its meaning is identified by `primary_metric`: item-weighted F1 for WideSearch or strict accuracy for the other search benchmarks. Higher is better. | 0.72 | +| `run_config` | [Optional[components.UnifiedBenchmarksSearchRunConfig]](../components/unifiedbenchmarkssearchrunconfig.mdx) | :heavy_minus_sign: | Published lane configuration, included only when include_run_config=true. Only the agent turn count, reasoning effort, and temperature are exposed; other harness settings are intentionally not part of the public contract. | | +| `search_engine` | *str* | :heavy_check_mark: | Search engine the published configuration used. | exa | +| `search_surface` | [components.SearchSurface](../components/searchsurface.mdx) | :heavy_check_mark: | Request surface the published configuration went through. | server-tool | +| `source` | [components.UnifiedBenchmarksSearchItemSource](../components/unifiedbenchmarkssearchitemsource.mdx) | :heavy_check_mark: | Benchmark source discriminator. | | +| `total_tasks` | *int* | :heavy_check_mark: | Tasks evaluated across the published configuration's runs. | 100 | \ No newline at end of file diff --git a/docs/components/unifiedbenchmarkssearchitembenchmarktype.mdx b/docs/components/unifiedbenchmarkssearchitembenchmarktype.mdx new file mode 100644 index 00000000..dfd7cc2d --- /dev/null +++ b/docs/components/unifiedbenchmarkssearchitembenchmarktype.mdx @@ -0,0 +1,24 @@ +--- +title: "UnifiedBenchmarksSearchItemBenchmarkType" +--- + +OpenRouter search benchmark. + +## Example Usage + +```python +from openrouter.components import UnifiedBenchmarksSearchItemBenchmarkType + +# Open enum: unrecognized values are captured as UnrecognizedStr +value: UnifiedBenchmarksSearchItemBenchmarkType = "search_browsecomp" +``` + + +## Values + +This is an open enum. Unrecognized values will not fail type checks. + +- `"search_browsecomp"` +- `"search_hle"` +- `"search_dsqa"` +- `"search_widesearch"` diff --git a/docs/components/unifiedbenchmarkssearchitemsource.mdx b/docs/components/unifiedbenchmarkssearchitemsource.mdx new file mode 100644 index 00000000..cf189d07 --- /dev/null +++ b/docs/components/unifiedbenchmarkssearchitemsource.mdx @@ -0,0 +1,17 @@ +--- +title: "UnifiedBenchmarksSearchItemSource" +--- + +Benchmark source discriminator. + +## Example Usage + +```python +from openrouter.components import UnifiedBenchmarksSearchItemSource +value: UnifiedBenchmarksSearchItemSource = "openrouter" +``` + + +## Values + +- `"openrouter"` diff --git a/docs/components/unifiedbenchmarkssearchrunconfig.mdx b/docs/components/unifiedbenchmarkssearchrunconfig.mdx new file mode 100644 index 00000000..21f463fd --- /dev/null +++ b/docs/components/unifiedbenchmarkssearchrunconfig.mdx @@ -0,0 +1,14 @@ +--- +title: "UnifiedBenchmarksSearchRunConfig" +--- + +Published lane configuration, included only when include_run_config=true. Only the agent turn count, reasoning effort, and temperature are exposed; other harness settings are intentionally not part of the public contract. + + +## Fields + +| Field | Type | Required | Description | Example | +| ----------------------------------------------------------------------------- | ----------------------------------------------------------------------------- | ----------------------------------------------------------------------------- | ----------------------------------------------------------------------------- | ----------------------------------------------------------------------------- | +| `max_agent_turns` | *Nullable[int]* | :heavy_check_mark: | Agent-turn count for the published lane, or null for plugin lanes. | 25 | +| `reasoning_effort` | *Nullable[str]* | :heavy_check_mark: | Reasoning effort configured for the published lane, or null when omitted. | high | +| `temperature` | *Nullable[float]* | :heavy_check_mark: | Sampling temperature configured for the published lane, or null when omitted. | 0.2 | \ No newline at end of file diff --git a/docs/operations/benchmarktype.mdx b/docs/operations/benchmarktype.mdx new file mode 100644 index 00000000..390b7aca --- /dev/null +++ b/docs/operations/benchmarktype.mdx @@ -0,0 +1,26 @@ +--- +title: "BenchmarkType" +--- + +Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources' items as they are. + +## Example Usage + +```python +from openrouter.operations import BenchmarkType + +# Open enum: unrecognized values are captured as UnrecognizedStr +value: BenchmarkType = "gpqa_diamond" +``` + + +## Values + +This is an open enum. Unrecognized values will not fail type checks. + +- `"gpqa_diamond"` +- `"tau_bench_verified_airline"` +- `"search_browsecomp"` +- `"search_hle"` +- `"search_dsqa"` +- `"search_widesearch"` diff --git a/docs/operations/getbenchmarksrequest.mdx b/docs/operations/getbenchmarksrequest.mdx index c5acba5b..e45dab26 100644 --- a/docs/operations/getbenchmarksrequest.mdx +++ b/docs/operations/getbenchmarksrequest.mdx @@ -4,13 +4,17 @@ title: "GetBenchmarksRequest" ## Fields -| Field | Type | Required | Description | Example | -| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `http_referer` | *Optional[str]* | :heavy_minus_sign: | The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
| | -| `x_open_router_title` | *Optional[str]* | :heavy_minus_sign: | The app display name allows you to customize how your app appears in OpenRouter's dashboard.
| | -| `x_open_router_categories` | *Optional[str]* | :heavy_minus_sign: | Comma-separated list of app categories (e.g. "cli-agent,cloud-agent"). Used for marketplace rankings.
| | -| `source` | [Optional[operations.Source]](../operations/source.mdx) | :heavy_minus_sign: | Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources. | artificial-analysis | -| `task_type` | [Optional[operations.TaskType]](../operations/tasktype.mdx) | :heavy_minus_sign: | Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. | coding | -| `arena` | [Optional[operations.Arena]](../operations/arena.mdx) | :heavy_minus_sign: | Design Arena only: arena to query. Defaults to `models` when source is `design-arena`. | models | -| `category` | *Optional[str]* | :heavy_minus_sign: | Design Arena only: category within the arena (e.g. `codecategories`, `uicomponent`, `gamedev`, `3d`, `dataviz`, `image`, `video`, `svg`). When omitted, returns all categories. | codecategories | -| `max_results` | *Optional[int]* | :heavy_minus_sign: | Maximum number of items to return. When omitted, all matching results are returned. | 50 | \ No newline at end of file +| Field | Type | Required | Description | Example | +| ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `http_referer` | *Optional[str]* | :heavy_minus_sign: | The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
| | +| `x_open_router_title` | *Optional[str]* | :heavy_minus_sign: | The app display name allows you to customize how your app appears in OpenRouter's dashboard.
| | +| `x_open_router_categories` | *Optional[str]* | :heavy_minus_sign: | Comma-separated list of app categories (e.g. "cli-agent,cloud-agent"). Used for marketplace rankings.
| | +| `source` | [Optional[operations.Source]](../operations/source.mdx) | :heavy_minus_sign: | Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources. | artificial-analysis | +| `task_type` | [Optional[operations.TaskType]](../operations/tasktype.mdx) | :heavy_minus_sign: | Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only. | coding | +| `benchmark_type` | [Optional[operations.BenchmarkType]](../operations/benchmarktype.mdx) | :heavy_minus_sign: | Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources' items as they are. | search_widesearch | +| `include_run_config` | *Optional[bool]* | :heavy_minus_sign: | Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract. | true | +| `search_engine` | *Optional[str]* | :heavy_minus_sign: | OpenRouter search benchmarks only: filter by the search engine used. | exa | +| `search_surface` | [Optional[operations.SearchSurface]](../operations/searchsurface.mdx) | :heavy_minus_sign: | OpenRouter search benchmarks only: filter by the request surface the lane ran on. | server-tool | +| `arena` | [Optional[operations.Arena]](../operations/arena.mdx) | :heavy_minus_sign: | Design Arena only: arena to query. Defaults to `models` when source is `design-arena`. | models | +| `category` | *Optional[str]* | :heavy_minus_sign: | Design Arena only: category within the arena (e.g. `codecategories`, `uicomponent`, `gamedev`, `3d`, `dataviz`, `image`, `video`, `svg`). When omitted, returns all categories. | codecategories | +| `max_results` | *Optional[int]* | :heavy_minus_sign: | Maximum number of items to return. When omitted, all matching results are returned. | 50 | \ No newline at end of file diff --git a/docs/operations/searchsurface.mdx b/docs/operations/searchsurface.mdx new file mode 100644 index 00000000..cf8fe4f0 --- /dev/null +++ b/docs/operations/searchsurface.mdx @@ -0,0 +1,22 @@ +--- +title: "SearchSurface" +--- + +OpenRouter search benchmarks only: filter by the request surface the lane ran on. + +## Example Usage + +```python +from openrouter.operations import SearchSurface + +# Open enum: unrecognized values are captured as UnrecognizedStr +value: SearchSurface = "server-tool" +``` + + +## Values + +This is an open enum. Unrecognized values will not fail type checks. + +- `"server-tool"` +- `"plugin"` diff --git a/docs/operations/tasktype.mdx b/docs/operations/tasktype.mdx index 7cdaa5db..ca10bf29 100644 --- a/docs/operations/tasktype.mdx +++ b/docs/operations/tasktype.mdx @@ -2,7 +2,7 @@ title: "TaskType" --- -Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. +Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only. ## Example Usage @@ -21,3 +21,4 @@ This is an open enum. Unrecognized values will not fail type checks. - `"coding"` - `"intelligence"` - `"agentic"` +- `"search"` diff --git a/docs/sdks/benchmarks/README.mdx b/docs/sdks/benchmarks/README.mdx index da27ed83..5026ac2a 100644 --- a/docs/sdks/benchmarks/README.mdx +++ b/docs/sdks/benchmarks/README.mdx @@ -13,7 +13,7 @@ Benchmarks endpoints ## get_benchmarks -Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter's own tau-bench and GPQA evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account. +Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter's own tau-bench, GPQA, and web-search evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Use task_type=search (or a search_* benchmark_type) for OpenRouter's search benchmarks, which publish each model's highest-scoring eligible evaluation configuration with same-configuration runs combined by task-weighted mean. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account. ### Example Usage @@ -29,7 +29,7 @@ with OpenRouter( api_key=os.getenv("OPENROUTER_API_KEY", ""), ) as open_router: - res = open_router.benchmarks.get_benchmarks() + res = open_router.benchmarks.get_benchmarks(include_run_config=True) # Handle response print(res) @@ -38,17 +38,21 @@ with OpenRouter( ### Parameters -| Parameter | Type | Required | Description | Example | -| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `http_referer` | *Optional[str]* | :heavy_minus_sign: | The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
| | -| `x_open_router_title` | *Optional[str]* | :heavy_minus_sign: | The app display name allows you to customize how your app appears in OpenRouter's dashboard.
| | -| `x_open_router_categories` | *Optional[str]* | :heavy_minus_sign: | Comma-separated list of app categories (e.g. "cli-agent,cloud-agent"). Used for marketplace rankings.
| | -| `source` | [Optional[operations.Source]](../../operations/source.mdx) | :heavy_minus_sign: | Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources. | artificial-analysis | -| `task_type` | [Optional[operations.TaskType]](../../operations/tasktype.mdx) | :heavy_minus_sign: | Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. | coding | -| `arena` | [Optional[operations.Arena]](../../operations/arena.mdx) | :heavy_minus_sign: | Design Arena only: arena to query. Defaults to `models` when source is `design-arena`. | models | -| `category` | *Optional[str]* | :heavy_minus_sign: | Design Arena only: category within the arena (e.g. `codecategories`, `uicomponent`, `gamedev`, `3d`, `dataviz`, `image`, `video`, `svg`). When omitted, returns all categories. | codecategories | -| `max_results` | *Optional[int]* | :heavy_minus_sign: | Maximum number of items to return. When omitted, all matching results are returned. | 50 | -| `retries` | [Optional[utils.RetryConfig]](../../models/utils/retryconfig.mdx) | :heavy_minus_sign: | Configuration to override the default retry behavior of the client. | | +| Parameter | Type | Required | Description | Example | +| ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `http_referer` | *Optional[str]* | :heavy_minus_sign: | The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
| | +| `x_open_router_title` | *Optional[str]* | :heavy_minus_sign: | The app display name allows you to customize how your app appears in OpenRouter's dashboard.
| | +| `x_open_router_categories` | *Optional[str]* | :heavy_minus_sign: | Comma-separated list of app categories (e.g. "cli-agent,cloud-agent"). Used for marketplace rankings.
| | +| `source` | [Optional[operations.Source]](../../operations/source.mdx) | :heavy_minus_sign: | Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources. | artificial-analysis | +| `task_type` | [Optional[operations.TaskType]](../../operations/tasktype.mdx) | :heavy_minus_sign: | Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only. | coding | +| `benchmark_type` | [Optional[operations.BenchmarkType]](../../operations/benchmarktype.mdx) | :heavy_minus_sign: | Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources' items as they are. | search_widesearch | +| `include_run_config` | *Optional[bool]* | :heavy_minus_sign: | Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract. | true | +| `search_engine` | *Optional[str]* | :heavy_minus_sign: | OpenRouter search benchmarks only: filter by the search engine used. | exa | +| `search_surface` | [Optional[operations.SearchSurface]](../../operations/searchsurface.mdx) | :heavy_minus_sign: | OpenRouter search benchmarks only: filter by the request surface the lane ran on. | server-tool | +| `arena` | [Optional[operations.Arena]](../../operations/arena.mdx) | :heavy_minus_sign: | Design Arena only: arena to query. Defaults to `models` when source is `design-arena`. | models | +| `category` | *Optional[str]* | :heavy_minus_sign: | Design Arena only: category within the arena (e.g. `codecategories`, `uicomponent`, `gamedev`, `3d`, `dataviz`, `image`, `video`, `svg`). When omitted, returns all categories. | codecategories | +| `max_results` | *Optional[int]* | :heavy_minus_sign: | Maximum number of items to return. When omitted, all matching results are returned. | 50 | +| `retries` | [Optional[utils.RetryConfig]](../../models/utils/retryconfig.mdx) | :heavy_minus_sign: | Configuration to override the default retry behavior of the client. | | ### Response diff --git a/pyproject.toml b/pyproject.toml index dd8146be..2633c255 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "openrouter" -version = "1.1.43" +version = "1.1.44" description = "Official Python Client SDK for OpenRouter." authors = [{ name = "OpenRouter" },] readme = "README-PYPI.md" diff --git a/src/openrouter/_version.py b/src/openrouter/_version.py index 5f253b27..05527560 100644 --- a/src/openrouter/_version.py +++ b/src/openrouter/_version.py @@ -3,10 +3,10 @@ import importlib.metadata __title__: str = "openrouter" -__version__: str = "1.1.43" +__version__: str = "1.1.44" __openapi_doc_version__: str = "1.0.0" __gen_version__: str = "2.914.0" -__user_agent__: str = "speakeasy-sdk/python 1.1.43 2.914.0 1.0.0 openrouter" +__user_agent__: str = "speakeasy-sdk/python 1.1.44 2.914.0 1.0.0 openrouter" try: if __package__ is not None: diff --git a/src/openrouter/benchmarks.py b/src/openrouter/benchmarks.py index 5dcb0e04..f163fcbb 100644 --- a/src/openrouter/benchmarks.py +++ b/src/openrouter/benchmarks.py @@ -20,6 +20,10 @@ def get_benchmarks( x_open_router_categories: Optional[str] = None, source: Optional[operations.Source] = None, task_type: Optional[operations.TaskType] = None, + benchmark_type: Optional[operations.BenchmarkType] = None, + include_run_config: Optional[bool] = False, + search_engine: Optional[str] = None, + search_surface: Optional[operations.SearchSurface] = None, arena: Optional[operations.Arena] = None, category: Optional[str] = None, max_results: Optional[int] = None, @@ -30,7 +34,7 @@ def get_benchmarks( ) -> components.UnifiedBenchmarksResponse: r"""List Benchmarks - Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter's own tau-bench and GPQA evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account. + Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter's own tau-bench, GPQA, and web-search evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Use task_type=search (or a search_* benchmark_type) for OpenRouter's search benchmarks, which publish each model's highest-scoring eligible evaluation configuration with same-configuration runs combined by task-weighted mean. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account. :param http_referer: The app identifier should be your app's URL and is used as the primary identifier for rankings. This is used to track API usage per application. @@ -40,7 +44,11 @@ def get_benchmarks( :param x_open_router_categories: Comma-separated list of app categories (e.g. \"cli-agent,cloud-agent\"). Used for marketplace rankings. :param source: Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources. - :param task_type: Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. + :param task_type: Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only. + :param benchmark_type: Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources' items as they are. + :param include_run_config: Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract. + :param search_engine: OpenRouter search benchmarks only: filter by the search engine used. + :param search_surface: OpenRouter search benchmarks only: filter by the request surface the lane ran on. :param arena: Design Arena only: arena to query. Defaults to `models` when source is `design-arena`. :param category: Design Arena only: category within the arena (e.g. `codecategories`, `uicomponent`, `gamedev`, `3d`, `dataviz`, `image`, `video`, `svg`). When omitted, returns all categories. :param max_results: Maximum number of items to return. When omitted, all matching results are returned. @@ -65,6 +73,10 @@ def get_benchmarks( x_open_router_categories=x_open_router_categories, source=source, task_type=task_type, + benchmark_type=benchmark_type, + include_run_config=include_run_config, + search_engine=search_engine, + search_surface=search_surface, arena=arena, category=category, max_results=max_results, @@ -167,6 +179,10 @@ async def get_benchmarks_async( x_open_router_categories: Optional[str] = None, source: Optional[operations.Source] = None, task_type: Optional[operations.TaskType] = None, + benchmark_type: Optional[operations.BenchmarkType] = None, + include_run_config: Optional[bool] = False, + search_engine: Optional[str] = None, + search_surface: Optional[operations.SearchSurface] = None, arena: Optional[operations.Arena] = None, category: Optional[str] = None, max_results: Optional[int] = None, @@ -177,7 +193,7 @@ async def get_benchmarks_async( ) -> components.UnifiedBenchmarksResponse: r"""List Benchmarks - Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter's own tau-bench and GPQA evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account. + Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter's own tau-bench, GPQA, and web-search evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Use task_type=search (or a search_* benchmark_type) for OpenRouter's search benchmarks, which publish each model's highest-scoring eligible evaluation configuration with same-configuration runs combined by task-weighted mean. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account. :param http_referer: The app identifier should be your app's URL and is used as the primary identifier for rankings. This is used to track API usage per application. @@ -187,7 +203,11 @@ async def get_benchmarks_async( :param x_open_router_categories: Comma-separated list of app categories (e.g. \"cli-agent,cloud-agent\"). Used for marketplace rankings. :param source: Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources. - :param task_type: Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. + :param task_type: Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only. + :param benchmark_type: Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources' items as they are. + :param include_run_config: Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract. + :param search_engine: OpenRouter search benchmarks only: filter by the search engine used. + :param search_surface: OpenRouter search benchmarks only: filter by the request surface the lane ran on. :param arena: Design Arena only: arena to query. Defaults to `models` when source is `design-arena`. :param category: Design Arena only: category within the arena (e.g. `codecategories`, `uicomponent`, `gamedev`, `3d`, `dataviz`, `image`, `video`, `svg`). When omitted, returns all categories. :param max_results: Maximum number of items to return. When omitted, all matching results are returned. @@ -212,6 +232,10 @@ async def get_benchmarks_async( x_open_router_categories=x_open_router_categories, source=source, task_type=task_type, + benchmark_type=benchmark_type, + include_run_config=include_run_config, + search_engine=search_engine, + search_surface=search_surface, arena=arena, category=category, max_results=max_results, diff --git a/src/openrouter/components/__init__.py b/src/openrouter/components/__init__.py index 670eeb2d..f714cc33 100644 --- a/src/openrouter/components/__init__.py +++ b/src/openrouter/components/__init__.py @@ -2890,8 +2890,8 @@ UnifiedBenchmarksMetaVersion, ) from .unifiedbenchmarksoritem import ( - BenchmarkType, UnifiedBenchmarksORItem, + UnifiedBenchmarksORItemBenchmarkType, UnifiedBenchmarksORItemSource, UnifiedBenchmarksORItemTypedDict, ) @@ -2900,7 +2900,18 @@ UnifiedBenchmarksResponseData, UnifiedBenchmarksResponseDataTypedDict, UnifiedBenchmarksResponseTypedDict, - UnknownUnifiedBenchmarksResponseData, + ) + from .unifiedbenchmarkssearchitem import ( + PrimaryMetric, + SearchSurface, + UnifiedBenchmarksSearchItem, + UnifiedBenchmarksSearchItemBenchmarkType, + UnifiedBenchmarksSearchItemSource, + UnifiedBenchmarksSearchItemTypedDict, + ) + from .unifiedbenchmarkssearchrunconfig import ( + UnifiedBenchmarksSearchRunConfig, + UnifiedBenchmarksSearchRunConfigTypedDict, ) from .unprocessableentityresponseerrordata import ( UnprocessableEntityResponseErrorData, @@ -3344,7 +3355,6 @@ "BashServerToolEnvironmentTypedDict", "BashServerToolType", "BashServerToolTypedDict", - "BenchmarkType", "Billable", "BooleanCapability", "BooleanCapabilityType", @@ -4741,6 +4751,7 @@ "PricingOverride", "PricingOverrideTypedDict", "PricingTypedDict", + "PrimaryMetric", "PromptCacheBreakpoint", "PromptCacheBreakpointMode", "PromptCacheBreakpointTypedDict", @@ -4921,6 +4932,7 @@ "SearchModelsServerToolOpenRouterType", "SearchModelsServerToolOpenRouterTypedDict", "SearchQualityLevel", + "SearchSurface", "Security", "SecurityTypedDict", "ServerToolUse", @@ -5165,12 +5177,19 @@ "UnifiedBenchmarksMetaTypedDict", "UnifiedBenchmarksMetaVersion", "UnifiedBenchmarksORItem", + "UnifiedBenchmarksORItemBenchmarkType", "UnifiedBenchmarksORItemSource", "UnifiedBenchmarksORItemTypedDict", "UnifiedBenchmarksResponse", "UnifiedBenchmarksResponseData", "UnifiedBenchmarksResponseDataTypedDict", "UnifiedBenchmarksResponseTypedDict", + "UnifiedBenchmarksSearchItem", + "UnifiedBenchmarksSearchItemBenchmarkType", + "UnifiedBenchmarksSearchItemSource", + "UnifiedBenchmarksSearchItemTypedDict", + "UnifiedBenchmarksSearchRunConfig", + "UnifiedBenchmarksSearchRunConfigTypedDict", "UniqueInsight", "UniqueInsightTypedDict", "Unit", @@ -5201,7 +5220,6 @@ "UnknownOutputMessageItemContent", "UnknownReasoningDetailUnion", "UnknownStreamEvents", - "UnknownUnifiedBenchmarksResponseData", "UnprocessableEntityResponseErrorData", "UnprocessableEntityResponseErrorDataTypedDict", "UpdateBYOKKeyRequest", @@ -7432,15 +7450,22 @@ "UnifiedBenchmarksMetaSource": ".unifiedbenchmarksmeta", "UnifiedBenchmarksMetaTypedDict": ".unifiedbenchmarksmeta", "UnifiedBenchmarksMetaVersion": ".unifiedbenchmarksmeta", - "BenchmarkType": ".unifiedbenchmarksoritem", "UnifiedBenchmarksORItem": ".unifiedbenchmarksoritem", + "UnifiedBenchmarksORItemBenchmarkType": ".unifiedbenchmarksoritem", "UnifiedBenchmarksORItemSource": ".unifiedbenchmarksoritem", "UnifiedBenchmarksORItemTypedDict": ".unifiedbenchmarksoritem", "UnifiedBenchmarksResponse": ".unifiedbenchmarksresponse", "UnifiedBenchmarksResponseData": ".unifiedbenchmarksresponse", "UnifiedBenchmarksResponseDataTypedDict": ".unifiedbenchmarksresponse", "UnifiedBenchmarksResponseTypedDict": ".unifiedbenchmarksresponse", - "UnknownUnifiedBenchmarksResponseData": ".unifiedbenchmarksresponse", + "PrimaryMetric": ".unifiedbenchmarkssearchitem", + "SearchSurface": ".unifiedbenchmarkssearchitem", + "UnifiedBenchmarksSearchItem": ".unifiedbenchmarkssearchitem", + "UnifiedBenchmarksSearchItemBenchmarkType": ".unifiedbenchmarkssearchitem", + "UnifiedBenchmarksSearchItemSource": ".unifiedbenchmarkssearchitem", + "UnifiedBenchmarksSearchItemTypedDict": ".unifiedbenchmarkssearchitem", + "UnifiedBenchmarksSearchRunConfig": ".unifiedbenchmarkssearchrunconfig", + "UnifiedBenchmarksSearchRunConfigTypedDict": ".unifiedbenchmarkssearchrunconfig", "UnprocessableEntityResponseErrorData": ".unprocessableentityresponseerrordata", "UnprocessableEntityResponseErrorDataTypedDict": ".unprocessableentityresponseerrordata", "UpdateBYOKKeyRequest": ".updatebyokkeyrequest", diff --git a/src/openrouter/components/unifiedbenchmarksoritem.py b/src/openrouter/components/unifiedbenchmarksoritem.py index c58ba941..c1671f3e 100644 --- a/src/openrouter/components/unifiedbenchmarksoritem.py +++ b/src/openrouter/components/unifiedbenchmarksoritem.py @@ -7,7 +7,7 @@ from typing_extensions import TypedDict -BenchmarkType = Union[ +UnifiedBenchmarksORItemBenchmarkType = Union[ Literal[ "gpqa_diamond", "tau_bench_verified_airline", @@ -28,7 +28,7 @@ class UnifiedBenchmarksORItemTypedDict(TypedDict): r"""Standard deviation of run accuracy, or null for a single run.""" avg_cost_per_task: Nullable[float] r"""Average cost per task in USD, or null if unavailable.""" - benchmark_type: BenchmarkType + benchmark_type: UnifiedBenchmarksORItemBenchmarkType r"""OpenRouter benchmark evaluation type.""" display_name: str r"""Human-readable model name.""" @@ -52,7 +52,7 @@ class UnifiedBenchmarksORItem(BaseModel): avg_cost_per_task: Nullable[float] r"""Average cost per task in USD, or null if unavailable.""" - benchmark_type: BenchmarkType + benchmark_type: UnifiedBenchmarksORItemBenchmarkType r"""OpenRouter benchmark evaluation type.""" display_name: str diff --git a/src/openrouter/components/unifiedbenchmarksresponse.py b/src/openrouter/components/unifiedbenchmarksresponse.py index 311cb86d..e2ed5fb9 100644 --- a/src/openrouter/components/unifiedbenchmarksresponse.py +++ b/src/openrouter/components/unifiedbenchmarksresponse.py @@ -14,13 +14,13 @@ UnifiedBenchmarksORItem, UnifiedBenchmarksORItemTypedDict, ) -from functools import partial +from .unifiedbenchmarkssearchitem import ( + UnifiedBenchmarksSearchItem, + UnifiedBenchmarksSearchItemTypedDict, +) from openrouter.types import BaseModel -from openrouter.utils.unions import parse_open_union -from pydantic import ConfigDict -from pydantic.functional_validators import BeforeValidator -from typing import Any, List, Literal, Union -from typing_extensions import Annotated, TypeAliasType, TypedDict +from typing import List, Union +from typing_extensions import TypeAliasType, TypedDict UnifiedBenchmarksResponseDataTypedDict = TypeAliasType( @@ -29,44 +29,20 @@ UnifiedBenchmarksAAItemTypedDict, UnifiedBenchmarksORItemTypedDict, UnifiedBenchmarksDAItemTypedDict, + UnifiedBenchmarksSearchItemTypedDict, ], ) -class UnknownUnifiedBenchmarksResponseData(BaseModel): - r"""A UnifiedBenchmarksResponseData variant the SDK doesn't recognize. Preserves the raw payload.""" - - source: Literal["UNKNOWN"] = "UNKNOWN" - raw: Any - is_unknown: Literal[True] = True - - model_config = ConfigDict(frozen=True) - - -_UNIFIED_BENCHMARKS_RESPONSE_DATA_VARIANTS: dict[str, Any] = { - "artificial-analysis": UnifiedBenchmarksAAItem, - "design-arena": UnifiedBenchmarksDAItem, - "openrouter": UnifiedBenchmarksORItem, -} - - -UnifiedBenchmarksResponseData = Annotated[ +UnifiedBenchmarksResponseData = TypeAliasType( + "UnifiedBenchmarksResponseData", Union[ UnifiedBenchmarksAAItem, - UnifiedBenchmarksDAItem, UnifiedBenchmarksORItem, - UnknownUnifiedBenchmarksResponseData, + UnifiedBenchmarksDAItem, + UnifiedBenchmarksSearchItem, ], - BeforeValidator( - partial( - parse_open_union, - disc_key="source", - variants=_UNIFIED_BENCHMARKS_RESPONSE_DATA_VARIANTS, - unknown_cls=UnknownUnifiedBenchmarksResponseData, - union_name="UnifiedBenchmarksResponseData", - ) - ), -] +) class UnifiedBenchmarksResponseTypedDict(TypedDict): diff --git a/src/openrouter/components/unifiedbenchmarkssearchitem.py b/src/openrouter/components/unifiedbenchmarkssearchitem.py new file mode 100644 index 00000000..f0b74435 --- /dev/null +++ b/src/openrouter/components/unifiedbenchmarkssearchitem.py @@ -0,0 +1,142 @@ +"""Code generated by Speakeasy (https://speakeasy.com). DO NOT EDIT.""" + +from __future__ import annotations +from .unifiedbenchmarkssearchrunconfig import ( + UnifiedBenchmarksSearchRunConfig, + UnifiedBenchmarksSearchRunConfigTypedDict, +) +from openrouter.types import BaseModel, Nullable, UNSET_SENTINEL, UnrecognizedStr +from pydantic import model_serializer +from typing import Literal, Optional, Union +from typing_extensions import NotRequired, TypedDict + + +UnifiedBenchmarksSearchItemBenchmarkType = Union[ + Literal[ + "search_browsecomp", + "search_hle", + "search_dsqa", + "search_widesearch", + ], + UnrecognizedStr, +] +r"""OpenRouter search benchmark.""" + + +PrimaryMetric = Union[ + Literal[ + "accuracy", + "f1_by_item", + ], + UnrecognizedStr, +] +r"""Identifies the meaning of `primary_score`: `f1_by_item` for WideSearch, `accuracy` for all other search benchmarks.""" + + +SearchSurface = Union[ + Literal[ + "server-tool", + "plugin", + ], + UnrecognizedStr, +] +r"""Request surface the published configuration went through.""" + + +UnifiedBenchmarksSearchItemSource = Literal["openrouter",] +r"""Benchmark source discriminator.""" + + +class UnifiedBenchmarksSearchItemTypedDict(TypedDict): + avg_cost_per_task: Nullable[float] + r"""Average cost per task in USD, or null if unavailable.""" + avg_latency_per_task_ms: Nullable[float] + r"""Average wall-clock latency per task in milliseconds, or null if unavailable.""" + benchmark_type: UnifiedBenchmarksSearchItemBenchmarkType + r"""OpenRouter search benchmark.""" + display_name: str + r"""Human-readable model name.""" + last_run_timestamp: str + r"""Timestamp of the newest qualifying run in the published configuration's lane.""" + model_permaslug: str + r"""Stable OpenRouter model identifier.""" + primary_metric: PrimaryMetric + r"""Identifies the meaning of `primary_score`: `f1_by_item` for WideSearch, `accuracy` for all other search benchmarks.""" + primary_score: float + r"""The benchmark's headline score from 0 to 1. Its meaning is identified by `primary_metric`: item-weighted F1 for WideSearch or strict accuracy for the other search benchmarks. Higher is better.""" + search_engine: str + r"""Search engine the published configuration used.""" + search_surface: SearchSurface + r"""Request surface the published configuration went through.""" + source: UnifiedBenchmarksSearchItemSource + r"""Benchmark source discriminator.""" + total_tasks: int + r"""Tasks evaluated across the published configuration's runs.""" + run_config: NotRequired[UnifiedBenchmarksSearchRunConfigTypedDict] + r"""Published lane configuration, included only when include_run_config=true. Only the agent turn count, reasoning effort, and temperature are exposed; other harness settings are intentionally not part of the public contract.""" + + +class UnifiedBenchmarksSearchItem(BaseModel): + avg_cost_per_task: Nullable[float] + r"""Average cost per task in USD, or null if unavailable.""" + + avg_latency_per_task_ms: Nullable[float] + r"""Average wall-clock latency per task in milliseconds, or null if unavailable.""" + + benchmark_type: UnifiedBenchmarksSearchItemBenchmarkType + r"""OpenRouter search benchmark.""" + + display_name: str + r"""Human-readable model name.""" + + last_run_timestamp: str + r"""Timestamp of the newest qualifying run in the published configuration's lane.""" + + model_permaslug: str + r"""Stable OpenRouter model identifier.""" + + primary_metric: PrimaryMetric + r"""Identifies the meaning of `primary_score`: `f1_by_item` for WideSearch, `accuracy` for all other search benchmarks.""" + + primary_score: float + r"""The benchmark's headline score from 0 to 1. Its meaning is identified by `primary_metric`: item-weighted F1 for WideSearch or strict accuracy for the other search benchmarks. Higher is better.""" + + search_engine: str + r"""Search engine the published configuration used.""" + + search_surface: SearchSurface + r"""Request surface the published configuration went through.""" + + source: UnifiedBenchmarksSearchItemSource + r"""Benchmark source discriminator.""" + + total_tasks: int + r"""Tasks evaluated across the published configuration's runs.""" + + run_config: Optional[UnifiedBenchmarksSearchRunConfig] = None + r"""Published lane configuration, included only when include_run_config=true. Only the agent turn count, reasoning effort, and temperature are exposed; other harness settings are intentionally not part of the public contract.""" + + @model_serializer(mode="wrap") + def serialize_model(self, handler): + optional_fields = set(["run_config"]) + nullable_fields = set(["avg_cost_per_task", "avg_latency_per_task_ms"]) + serialized = handler(self) + m = {} + + for n, f in type(self).model_fields.items(): + k = f.alias or n + val = serialized.get(k, serialized.get(n)) + is_nullable_and_explicitly_set = ( + k in nullable_fields + and (self.__pydantic_fields_set__.intersection({n})) # pylint: disable=no-member + ) + + if val != UNSET_SENTINEL: + if ( + val is not None + or k not in optional_fields + or is_nullable_and_explicitly_set + ): + m[k] = val + + return m diff --git a/src/openrouter/components/unifiedbenchmarkssearchrunconfig.py b/src/openrouter/components/unifiedbenchmarkssearchrunconfig.py new file mode 100644 index 00000000..e1b94da1 --- /dev/null +++ b/src/openrouter/components/unifiedbenchmarkssearchrunconfig.py @@ -0,0 +1,44 @@ +"""Code generated by Speakeasy (https://speakeasy.com). DO NOT EDIT.""" + +from __future__ import annotations +from openrouter.types import BaseModel, Nullable, UNSET_SENTINEL +from pydantic import model_serializer +from typing_extensions import TypedDict + + +class UnifiedBenchmarksSearchRunConfigTypedDict(TypedDict): + r"""Published lane configuration, included only when include_run_config=true. Only the agent turn count, reasoning effort, and temperature are exposed; other harness settings are intentionally not part of the public contract.""" + + max_agent_turns: Nullable[int] + r"""Agent-turn count for the published lane, or null for plugin lanes.""" + reasoning_effort: Nullable[str] + r"""Reasoning effort configured for the published lane, or null when omitted.""" + temperature: Nullable[float] + r"""Sampling temperature configured for the published lane, or null when omitted.""" + + +class UnifiedBenchmarksSearchRunConfig(BaseModel): + r"""Published lane configuration, included only when include_run_config=true. Only the agent turn count, reasoning effort, and temperature are exposed; other harness settings are intentionally not part of the public contract.""" + + max_agent_turns: Nullable[int] + r"""Agent-turn count for the published lane, or null for plugin lanes.""" + + reasoning_effort: Nullable[str] + r"""Reasoning effort configured for the published lane, or null when omitted.""" + + temperature: Nullable[float] + r"""Sampling temperature configured for the published lane, or null when omitted.""" + + @model_serializer(mode="wrap") + def serialize_model(self, handler): + serialized = handler(self) + m = {} + + for n, f in type(self).model_fields.items(): + k = f.alias or n + val = serialized.get(k, serialized.get(n)) + + if val != UNSET_SENTINEL: + m[k] = val + + return m diff --git a/src/openrouter/operations/__init__.py b/src/openrouter/operations/__init__.py index 37c3c8ce..50ebc6f5 100644 --- a/src/openrouter/operations/__init__.py +++ b/src/openrouter/operations/__init__.py @@ -326,10 +326,12 @@ ) from .getbenchmarks import ( Arena, + BenchmarkType, GetBenchmarksGlobals, GetBenchmarksGlobalsTypedDict, GetBenchmarksRequest, GetBenchmarksRequestTypedDict, + SearchSurface, Source, TaskType, ) @@ -813,6 +815,7 @@ __all__ = [ "Arena", + "BenchmarkType", "BulkAddWorkspaceMembersGlobals", "BulkAddWorkspaceMembersGlobalsTypedDict", "BulkAddWorkspaceMembersRequest", @@ -1356,6 +1359,7 @@ "Result", "ResultTypedDict", "Role", + "SearchSurface", "SendChatCompletionRequestGlobals", "SendChatCompletionRequestGlobalsTypedDict", "SendChatCompletionRequestRequest", @@ -1678,10 +1682,12 @@ "GetAppRankingsSort": ".getapprankings", "Subcategory": ".getapprankings", "Arena": ".getbenchmarks", + "BenchmarkType": ".getbenchmarks", "GetBenchmarksGlobals": ".getbenchmarks", "GetBenchmarksGlobalsTypedDict": ".getbenchmarks", "GetBenchmarksRequest": ".getbenchmarks", "GetBenchmarksRequestTypedDict": ".getbenchmarks", + "SearchSurface": ".getbenchmarks", "Source": ".getbenchmarks", "TaskType": ".getbenchmarks", "GetBYOKKeyGlobals": ".getbyokkey", diff --git a/src/openrouter/operations/getbenchmarks.py b/src/openrouter/operations/getbenchmarks.py index 62573095..51f1f39f 100644 --- a/src/openrouter/operations/getbenchmarks.py +++ b/src/openrouter/operations/getbenchmarks.py @@ -89,10 +89,35 @@ def serialize_model(self, handler): "coding", "intelligence", "agentic", + "search", ], UnrecognizedStr, ] -r"""Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category.""" +r"""Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only.""" + + +BenchmarkType = Union[ + Literal[ + "gpqa_diamond", + "tau_bench_verified_airline", + "search_browsecomp", + "search_hle", + "search_dsqa", + "search_widesearch", + ], + UnrecognizedStr, +] +r"""Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources' items as they are.""" + + +SearchSurface = Union[ + Literal[ + "server-tool", + "plugin", + ], + UnrecognizedStr, +] +r"""OpenRouter search benchmarks only: filter by the request surface the lane ran on.""" Arena = Union[ @@ -123,7 +148,15 @@ class GetBenchmarksRequestTypedDict(TypedDict): source: NotRequired[Source] r"""Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources.""" task_type: NotRequired[TaskType] - r"""Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category.""" + r"""Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only.""" + benchmark_type: NotRequired[BenchmarkType] + r"""Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources' items as they are.""" + include_run_config: NotRequired[bool] + r"""Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract.""" + search_engine: NotRequired[str] + r"""OpenRouter search benchmarks only: filter by the search engine used.""" + search_surface: NotRequired[SearchSurface] + r"""OpenRouter search benchmarks only: filter by the request surface the lane ran on.""" arena: NotRequired[Arena] r"""Design Arena only: arena to query. Defaults to `models` when source is `design-arena`.""" category: NotRequired[str] @@ -171,7 +204,31 @@ class GetBenchmarksRequest(BaseModel): Optional[TaskType], FieldMetadata(query=QueryParamMetadata(style="form", explode=True)), ] = None - r"""Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category.""" + r"""Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only.""" + + benchmark_type: Annotated[ + Optional[BenchmarkType], + FieldMetadata(query=QueryParamMetadata(style="form", explode=True)), + ] = None + r"""Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources' items as they are.""" + + include_run_config: Annotated[ + Optional[bool], + FieldMetadata(query=QueryParamMetadata(style="form", explode=True)), + ] = False + r"""Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract.""" + + search_engine: Annotated[ + Optional[str], + FieldMetadata(query=QueryParamMetadata(style="form", explode=True)), + ] = None + r"""OpenRouter search benchmarks only: filter by the search engine used.""" + + search_surface: Annotated[ + Optional[SearchSurface], + FieldMetadata(query=QueryParamMetadata(style="form", explode=True)), + ] = None + r"""OpenRouter search benchmarks only: filter by the request surface the lane ran on.""" arena: Annotated[ Optional[Arena], @@ -200,6 +257,10 @@ def serialize_model(self, handler): "X-OpenRouter-Categories", "source", "task_type", + "benchmark_type", + "include_run_config", + "search_engine", + "search_surface", "arena", "category", "max_results", diff --git a/uv.lock b/uv.lock index d823e181..3ac0d57c 100644 --- a/uv.lock +++ b/uv.lock @@ -213,7 +213,7 @@ wheels = [ [[package]] name = "openrouter" -version = "1.1.43" +version = "1.1.44" source = { editable = "." } dependencies = [ { name = "httpcore" }, From b385938fd59bd214ecfbe08c6e811acc09249b9c Mon Sep 17 00:00:00 2001 From: "speakeasy-github[bot]" <128539517+speakeasy-github[bot]@users.noreply.github.com> Date: Tue, 11 Aug 2026 14:05:55 +0000 Subject: [PATCH 2/2] empty commit to trigger [run-tests] workflow