diff --git a/.speakeasy/gen.lock b/.speakeasy/gen.lock
index a243234c..9a815164 100644
--- a/.speakeasy/gen.lock
+++ b/.speakeasy/gen.lock
@@ -1,19 +1,19 @@
lockVersion: 2.0.0
id: c48cf606-fb42-4a45-9c23-8f0555307828
management:
- docChecksum: 75d736c349cb105e920d684fd2506fca
+ docChecksum: e408687b6809c0eb28ce2608b3c9d155
docVersion: 1.0.0
speakeasyVersion: 1.787.0
generationVersion: 2.914.0
- releaseVersion: 1.1.43
- configChecksum: cccf7b2204d0f401df2776c4b1e3a363
+ releaseVersion: 1.1.44
+ configChecksum: 58e13146341fef6b0ea6ae8882827d37
repoURL: https://github.com/OpenRouterTeam/python-sdk.git
installationURL: https://github.com/OpenRouterTeam/python-sdk.git
published: true
persistentEdits:
- generation_id: 229674de-e3e7-4794-8be0-367148eb3a30
- pristine_commit_hash: af2265f9d619fb2150d3848f261d729d66277467
- pristine_tree_hash: 663823946e41b46acc3200e982caa55cb1d0eafa
+ generation_id: f961c45e-5096-4334-bf45-ef26d2b749e8
+ pristine_commit_hash: ba9a7f326ed5f0c2a84c6b81ced43935892e8629
+ pristine_tree_hash: e87ac6459017910906d4f2566ca55e1cebef6340
features:
python:
acceptHeaders: 3.0.0
@@ -760,10 +760,6 @@ trackedFiles:
id: 73f42e4bc945
last_write_checksum: sha1:599705c7ef520285d8da9e5510109846aa7fd332
pristine_git_object: 539a7c32200bef803260226e81f3225e9529688e
- docs/components/benchmarktype.mdx:
- id: c191e897ed61
- last_write_checksum: sha1:819054a8e4d3096134355828f732c73e90fa2d67
- pristine_git_object: 2a473fdfeeccc38689147c863d1f857560a97ab4
docs/components/billable.mdx:
id: b983fa757056
last_write_checksum: sha1:80d142a13e98a8f13da2850753282adf2ff59521
@@ -4260,6 +4256,10 @@ trackedFiles:
id: 74675f38c1c6
last_write_checksum: sha1:ca4bf606df26dd9e2b636d7d9e9f0fd183b21145
pristine_git_object: 5fb1b765b2b0e46d40042f341bbcc91aa0d92762
+ docs/components/primarymetric.mdx:
+ id: c6ec097bacaf
+ last_write_checksum: sha1:fc7d89ac02ff585424b5bc8cc9b0ac82338c7b6f
+ pristine_git_object: 6a99a2737d18a3c227bf7fcabf58b481d3c7d94b
docs/components/promptcachebreakpoint.mdx:
id: c135956dc623
last_write_checksum: sha1:6df345ea99da1fdcae22895faa2ee36339ba7332
@@ -4688,6 +4688,10 @@ trackedFiles:
id: 0cbced049b6f
last_write_checksum: sha1:106b7467c7ab738c7ce04ab44d83a91388dc0808
pristine_git_object: 09c661196e0bbf039ff1e117c55aa1981ca1ef4a
+ docs/components/searchsurface.mdx:
+ id: 4bcafd219190
+ last_write_checksum: sha1:f0c602a940edebf01020c33cb012245082040338
+ pristine_git_object: 9ef3f90e268fa74f1f6081afbdae17cc178fb850
docs/components/security.mdx:
id: 126e702b7c97
last_write_checksum: sha1:3fbb5c3145b96472437de6e6384ae222b0ae5210
@@ -5358,8 +5362,12 @@ trackedFiles:
pristine_git_object: 23ef4c9a45aea5d918e554f7868f772429ef364c
docs/components/unifiedbenchmarksoritem.mdx:
id: 642fc71c5d26
- last_write_checksum: sha1:5c136345f1f69e6f8db7269a31050aac5ada0894
- pristine_git_object: 99228c0beefaf20bb7b0134946a75404c9d1d0b6
+ last_write_checksum: sha1:2543b293158a67f93f19ee23331483cb56893c5b
+ pristine_git_object: ec2f59459e9fb089bb24c122a6a205da6c37de28
+ docs/components/unifiedbenchmarksoritembenchmarktype.mdx:
+ id: dad93e1758c9
+ last_write_checksum: sha1:7d0430ad93a25a95c8de517749ec1f89d0ca4e1b
+ pristine_git_object: 18316442f3346d326cf4ccbc453cf26f36a5fd74
docs/components/unifiedbenchmarksoritemsource.mdx:
id: b89557fa192a
last_write_checksum: sha1:37dbe5e1fc82f2dd6983383ad74f82048c367062
@@ -5370,8 +5378,24 @@ trackedFiles:
pristine_git_object: ee48e645716b05d6473f5ba0199fb84264a44d12
docs/components/unifiedbenchmarksresponsedata.mdx:
id: fd372e1c57e8
- last_write_checksum: sha1:303a04665c6586ab8cd3662ff5640d16db8e6c53
- pristine_git_object: ff4c06716df56052ed05ba3a5ac7e4c01d5ea27f
+ last_write_checksum: sha1:4606f3c0a2f1be4ddb0558960b8da50657255e23
+ pristine_git_object: 50ec3eb6b4bc4c414a1fc79603a00939ae71f976
+ docs/components/unifiedbenchmarkssearchitem.mdx:
+ id: 426474b17840
+ last_write_checksum: sha1:5e2b83ba24cf9c6073771e0a99fc98100dc613df
+ pristine_git_object: 40b4f0bea4b0b07b3ce2e5b7801f106eacc18d2a
+ docs/components/unifiedbenchmarkssearchitembenchmarktype.mdx:
+ id: 12cfb7df44cb
+ last_write_checksum: sha1:0555a969885d6080df39854a57950edcbfc3661d
+ pristine_git_object: dfd7cc2da5eb13fb9153fd9cbef2c8ec2642a4a9
+ docs/components/unifiedbenchmarkssearchitemsource.mdx:
+ id: 2be77d920124
+ last_write_checksum: sha1:e00a6c233f2b3bcc3787c3a8943611d340ddef1d
+ pristine_git_object: cf189d077669409efa26fcc6d1a6570298153978
+ docs/components/unifiedbenchmarkssearchrunconfig.mdx:
+ id: 0646f31a49fb
+ last_write_checksum: sha1:bedb0fdd46bfdda0deacab2042b304e06cd49296
+ pristine_git_object: 21f463fd8c660165b0533c8a883f43dd2fd3e961
docs/components/uniqueinsight.mdx:
id: 02fd71bcb47a
last_write_checksum: sha1:793e90c78ab450e36533710d3fad6f4a75432e86
@@ -5728,6 +5752,10 @@ trackedFiles:
id: fd0c961ea26d
last_write_checksum: sha1:48d0112803ed464cb9f9fe06d4b58f982f936522
pristine_git_object: 455ab607b237f29dd06d6548926f8ee08be410aa
+ docs/operations/benchmarktype.mdx:
+ id: 7de894bb7280
+ last_write_checksum: sha1:12f34fb51ed9e31b78aa69307fcf42e6f99b4f31
+ pristine_git_object: 390b7aca29e54383b436ea5a6e0201c8765001f2
docs/operations/bulkaddworkspacemembersglobals.mdx:
id: f198abbe4de0
last_write_checksum: sha1:ca25e8bae0fa1b09844c8ed66e174d7a8c06542d
@@ -6218,8 +6246,8 @@ trackedFiles:
pristine_git_object: 679923bfbff14ca3bc63ce5d451089b1996e5c6e
docs/operations/getbenchmarksrequest.mdx:
id: 704fb5d64a7c
- last_write_checksum: sha1:506d8cf7a2eed421ae52c60703e3400cdfa3a652
- pristine_git_object: c5acba5bb9a6a2ae287d03f0aef8da1bc49df8e6
+ last_write_checksum: sha1:a657060a69596b6542cf6fb972ac336d756d9afc
+ pristine_git_object: e45dab262f662cd66690cfe35511957422db2e7c
docs/operations/getbyokkeyglobals.mdx:
id: ac592fd40012
last_write_checksum: sha1:1eac912896162f772846e2fe5e6f3b50ae8d20f2
@@ -6872,6 +6900,10 @@ trackedFiles:
id: ee0d3c3547d7
last_write_checksum: sha1:11ebdfcfbf9a1197a42516d3daa0eaa3e2955f81
pristine_git_object: 9e1219b8c4da58c9ee1d0707aa809d3d3b54f372
+ docs/operations/searchsurface.mdx:
+ id: 2d463024ea29
+ last_write_checksum: sha1:f612f65bd78b46bd424a0bbbbfb6260331738acc
+ pristine_git_object: cf8fe4f0a66b46ea069704ab9f791e3c2a4adb22
docs/operations/sendchatcompletionrequestglobals.mdx:
id: 8ad074be0768
last_write_checksum: sha1:93462ef7b25e039e770050019663f6b41dbf4187
@@ -6902,8 +6934,8 @@ trackedFiles:
pristine_git_object: 18f5cc3e2fe30014037f8f68d5513a4981d513b9
docs/operations/tasktype.mdx:
id: d3da8106d2d1
- last_write_checksum: sha1:06aede3448490e1fb77b5c1b758238a11ce1dceb
- pristine_git_object: 7cdaa5db9c5d84013c3aeeaf1f7493c1768c1fef
+ last_write_checksum: sha1:0955731aba6ddddeeaee3b34ca7e353dc2dd9700
+ pristine_git_object: ca10bf29d177ce5724fe59c84d929b20e40781c6
docs/operations/timerange.mdx:
id: 03a218211f06
last_write_checksum: sha1:3cdcfbfcdb71dcc17a0c3b4cd5704eac3718fa0f
@@ -7046,8 +7078,8 @@ trackedFiles:
pristine_git_object: c8541c5c189bae01e80db73fd88767391d7dbb9b
docs/sdks/benchmarks/README.mdx:
id: 5b483a6770ba
- last_write_checksum: sha1:2d935127bdbf18d1237b3e3af23f9145216806b2
- pristine_git_object: da27ed830fab524c5b9bf2c72031749bb37ed365
+ last_write_checksum: sha1:885b70fcbd2be60f51a2f58e133fd6452eb21600
+ pristine_git_object: 5026ac2a283240480026d80d2f77d5528ae8bdc8
docs/sdks/betaanalytics/README.mdx:
id: 239279ebf01b
last_write_checksum: sha1:5a76d28d77f68efe34ebbcb72850e6a87c74d3b1
@@ -7158,8 +7190,8 @@ trackedFiles:
pristine_git_object: 3e38f1a929f7d6b1d6de74604aa87e3d8f010544
pyproject.toml:
id: 5d07e7d72637
- last_write_checksum: sha1:5fc67580b2031b06ee7e749fe9f41554b646b425
- pristine_git_object: dd8146be33b81f978739b06319b61584fc37cabe
+ last_write_checksum: sha1:18256d8b9cc609a66b10b73bc29e1f89e71a5ce1
+ pristine_git_object: 2633c255582f9f232999f764a31f1a5d170c712f
scripts/prepare_readme.py:
id: e0c5957a6035
last_write_checksum: sha1:77f44b60b98bc126557ec27391f91dfba764bb54
@@ -7186,8 +7218,8 @@ trackedFiles:
pristine_git_object: 86713cfea633e09d33b3d4e65281071fe20e6137
src/openrouter/_version.py:
id: d8d15ad6c586
- last_write_checksum: sha1:37abc296599992615a7d4aa8c880825df579140b
- pristine_git_object: 5f253b27c28788047892ef19fc7f069c3799b795
+ last_write_checksum: sha1:bf92de0c3824a4280f6bb520e285a00d20188ed7
+ pristine_git_object: 055275608059735910d646e0ada351d41f4a6017
src/openrouter/analytics.py:
id: cb406b5aaabb
last_write_checksum: sha1:3ea0f1c73fb9c101b7bc5719da4dd7bf62ff469a
@@ -7202,8 +7234,8 @@ trackedFiles:
pristine_git_object: 7a562e21c7e66c9db666ead83748b73adefe4278
src/openrouter/benchmarks.py:
id: 178d2ad8d706
- last_write_checksum: sha1:c5a8d0fa780efff4f9f81a12e102cb82d825eb57
- pristine_git_object: 5dcb0e0427dd7e820662fb44bed1099420be25dd
+ last_write_checksum: sha1:15b8a7ea05a7550032afb15981958aa73b756881
+ pristine_git_object: f163fcbb9c21fc1327d6c9a1eefcc60cbf9091ea
src/openrouter/beta.py:
id: fffdf54fd8f5
last_write_checksum: sha1:4e34fb96ffe38673ca72e0f8576a8eddd4134d4b
@@ -7230,8 +7262,8 @@ trackedFiles:
pristine_git_object: ad3d247954547814054c01989a2dff3d12b3e4e1
src/openrouter/components/__init__.py:
id: 81754e97b3f4
- last_write_checksum: sha1:f60cd4f5bc011539964e984662bf4de255c22493
- pristine_git_object: 670eeb2dbe8c686babcc1bf5dd23cd0d572fedff
+ last_write_checksum: sha1:7fc8366c0da24c31337bded97df0f1c8f9cdb173
+ pristine_git_object: f714cc3337bfe431ca2007e9511a81d970dd4e62
src/openrouter/components/aabenchmarkentry.py:
id: e2e0f0b48c82
last_write_checksum: sha1:fab4d9a24d2cea937bb749d46c5f83941e99d65c
@@ -9470,12 +9502,20 @@ trackedFiles:
pristine_git_object: 68d76b7b6dbd1817f825e38902556719b179015d
src/openrouter/components/unifiedbenchmarksoritem.py:
id: 8da361fd40c8
- last_write_checksum: sha1:2f375462570405680ccf18a7e265c3e25fc67dd8
- pristine_git_object: c58ba941246a96890970e6eb1b53588600b2addd
+ last_write_checksum: sha1:f946862e244121ae6f27017a5f8652fc3f964d67
+ pristine_git_object: c1671f3ef45eaaf719dc7d29b2f6a3824fb98934
src/openrouter/components/unifiedbenchmarksresponse.py:
id: 4f7bbccaba03
- last_write_checksum: sha1:2eedc8ad728768805d1bc504caa30989c751ce1f
- pristine_git_object: 311cb86d0cc7f42bd9f9842e3783517c182ffe4f
+ last_write_checksum: sha1:a7f33321910d7e05e31f9533da54ee2706250aec
+ pristine_git_object: e2ed5fb98fd78b6e9a6184f467bd80617ca9371f
+ src/openrouter/components/unifiedbenchmarkssearchitem.py:
+ id: cc77d4dc2470
+ last_write_checksum: sha1:7dadc353af8a9c3b803e69da3d4c7fd840403eae
+ pristine_git_object: f0b74435764621178fa70a138da1419aa1eaf245
+ src/openrouter/components/unifiedbenchmarkssearchrunconfig.py:
+ id: 189239ce6b4f
+ last_write_checksum: sha1:9d85897b7857707fb6a759aaa938f4b5bf5662c2
+ pristine_git_object: e1b94da161012480960c726a459b3fbd7c1a9bf4
src/openrouter/components/unprocessableentityresponseerrordata.py:
id: e8ca4a51f994
last_write_checksum: sha1:e97c277b49f4bb6bcfad8eef8a46b382730729a0
@@ -9790,8 +9830,8 @@ trackedFiles:
pristine_git_object: 5c553b76ab24d9eaef51c3bf6d85aa74391e335e
src/openrouter/operations/__init__.py:
id: 9afcea1e7161
- last_write_checksum: sha1:7462feea63bad287e4382464c6a5bf24b401eaa2
- pristine_git_object: 37c3c8ce47455812f19b303f0b4d34738ee1237a
+ last_write_checksum: sha1:1a29771015fee7983b677eb476aeb281fad9c2dc
+ pristine_git_object: 50ebc6f58e44ee099607fa6140232132a7990767
src/openrouter/operations/bulkaddworkspacemembers.py:
id: e0ed56117619
last_write_checksum: sha1:5c44eb0d40fdece3ac084615f6c6082be4cf1d5a
@@ -9938,8 +9978,8 @@ trackedFiles:
pristine_git_object: a8e00f731e950d64ec8ca7ab10f6fca96884c8e7
src/openrouter/operations/getbenchmarks.py:
id: 5fb88644491e
- last_write_checksum: sha1:2df9c948624b4126685e5347e5c9442d2a05945d
- pristine_git_object: 62573095f6d816050964d68fb35f61b570acd5d5
+ last_write_checksum: sha1:c059365dfa135f5d6d929464c06af3421cf89c62
+ pristine_git_object: 51f1f39f8f61ebb2edc9770ad79e09bd7886fd2d
src/openrouter/operations/getbyokkey.py:
id: d141452bd88a
last_write_checksum: sha1:d6d391b230d90f51945144e63e9c53772650bd7b
@@ -11796,6 +11836,9 @@ examples:
application/json: {"error": {"code": 500, "message": "Internal Server Error"}}
getBenchmarks:
speakeasy-default-get-benchmarks:
+ parameters:
+ query:
+ include_run_config: true
responses:
"200":
application/json: {"data": [{"agentic_index": 58.3, "coding_index": 65.8, "display_name": "GPT-4o", "intelligence_index": 71.2, "model_permaslug": "openai/gpt-4o", "pricing": {"completion": "0.00001", "prompt": "0.0000025"}, "source": "artificial-analysis"}, {"accuracy": 0.72, "accuracy_stddev": 0.03, "avg_cost_per_task": 0.002, "benchmark_type": "gpqa_diamond", "display_name": "GPT-4o", "last_run_timestamp": "2026-06-03T12:00:00Z", "model_permaslug": "openai/gpt-4o", "source": "openrouter", "total_tasks": 300}], "meta": {"as_of": "2026-06-03T12:00:00Z", "citation": null, "model_count": 1, "source": null, "source_url": null, "task_type": null, "version": "v1"}}
@@ -12023,3 +12066,4 @@ examples:
"500":
application/json: {"error": {"code": 500, "message": "Internal Server Error"}}
examplesVersion: 1.0.2
+releaseNotes: "## Python SDK Changes:\n* `open_router.benchmarks.get_benchmarks()`: \n * `request` **Changed**\n * `response.data[]` **Changed** (Breaking ⚠️)\n"
diff --git a/.speakeasy/gen.yaml b/.speakeasy/gen.yaml
index b0b0cd37..7066bde3 100644
--- a/.speakeasy/gen.yaml
+++ b/.speakeasy/gen.yaml
@@ -36,7 +36,7 @@ generation:
documentation: mintlify
preApplyUnionDiscriminators: true
python:
- version: 1.1.43
+ version: 1.1.44
additionalDependencies:
dev: {}
main: {}
diff --git a/.speakeasy/out.openapi.yaml b/.speakeasy/out.openapi.yaml
index 23377f0b..6244780a 100644
--- a/.speakeasy/out.openapi.yaml
+++ b/.speakeasy/out.openapi.yaml
@@ -24472,16 +24472,11 @@ components:
properties:
data:
items:
- discriminator:
- mapping:
- artificial-analysis: '#/components/schemas/UnifiedBenchmarksAAItem'
- design-arena: '#/components/schemas/UnifiedBenchmarksDAItem'
- openrouter: '#/components/schemas/UnifiedBenchmarksORItem'
- propertyName: 'source'
oneOf:
- $ref: '#/components/schemas/UnifiedBenchmarksAAItem'
- $ref: '#/components/schemas/UnifiedBenchmarksDAItem'
- $ref: '#/components/schemas/UnifiedBenchmarksORItem'
+ - $ref: '#/components/schemas/UnifiedBenchmarksSearchItem'
type: 'array'
meta:
$ref: '#/components/schemas/UnifiedBenchmarksMeta'
@@ -24489,6 +24484,121 @@ components:
- 'data'
- 'meta'
type: 'object'
+ UnifiedBenchmarksSearchItem:
+ properties:
+ avg_cost_per_task:
+ description: 'Average cost per task in USD, or null if unavailable.'
+ example: 0.031
+ format: 'double'
+ type:
+ - 'number'
+ - 'null'
+ avg_latency_per_task_ms:
+ description: 'Average wall-clock latency per task in milliseconds, or null if unavailable.'
+ example: 45210
+ format: 'double'
+ type:
+ - 'number'
+ - 'null'
+ benchmark_type:
+ description: 'OpenRouter search benchmark.'
+ enum:
+ - 'search_browsecomp'
+ - 'search_hle'
+ - 'search_dsqa'
+ - 'search_widesearch'
+ example: 'search_browsecomp'
+ type: 'string'
+ x-speakeasy-unknown-values: allow
+ display_name:
+ description: 'Human-readable model name.'
+ example: 'GPT-4o'
+ type: 'string'
+ last_run_timestamp:
+ description: 'Timestamp of the newest qualifying run in the published configuration''s lane.'
+ example: '2026-07-28T13:38:18Z'
+ type: 'string'
+ model_permaslug:
+ description: 'Stable OpenRouter model identifier.'
+ example: 'openai/gpt-4o'
+ type: 'string'
+ primary_metric:
+ description: 'Identifies the meaning of `primary_score`: `f1_by_item` for WideSearch, `accuracy` for all other search benchmarks.'
+ enum:
+ - 'accuracy'
+ - 'f1_by_item'
+ example: 'accuracy'
+ type: 'string'
+ x-speakeasy-unknown-values: allow
+ primary_score:
+ description: 'The benchmark''s headline score from 0 to 1. Its meaning is identified by `primary_metric`: item-weighted F1 for WideSearch or strict accuracy for the other search benchmarks. Higher is better.'
+ example: 0.72
+ format: 'double'
+ type: 'number'
+ run_config:
+ $ref: '#/components/schemas/UnifiedBenchmarksSearchRunConfig'
+ search_engine:
+ description: 'Search engine the published configuration used.'
+ example: 'exa'
+ type: 'string'
+ search_surface:
+ description: 'Request surface the published configuration went through.'
+ enum:
+ - 'server-tool'
+ - 'plugin'
+ example: 'server-tool'
+ type: 'string'
+ x-speakeasy-unknown-values: allow
+ source:
+ description: 'Benchmark source discriminator.'
+ enum:
+ - 'openrouter'
+ type: 'string'
+ total_tasks:
+ description: 'Tasks evaluated across the published configuration''s runs.'
+ example: 100
+ type: 'integer'
+ required:
+ - 'source'
+ - 'model_permaslug'
+ - 'display_name'
+ - 'benchmark_type'
+ - 'primary_metric'
+ - 'primary_score'
+ - 'total_tasks'
+ - 'avg_cost_per_task'
+ - 'avg_latency_per_task_ms'
+ - 'search_engine'
+ - 'search_surface'
+ - 'last_run_timestamp'
+ type: 'object'
+ UnifiedBenchmarksSearchRunConfig:
+ description: 'Published lane configuration, included only when include_run_config=true. Only the agent turn count, reasoning effort, and temperature are exposed; other harness settings are intentionally not part of the public contract.'
+ properties:
+ max_agent_turns:
+ description: 'Agent-turn count for the published lane, or null for plugin lanes.'
+ example: 25
+ type:
+ - 'integer'
+ - 'null'
+ reasoning_effort:
+ description: 'Reasoning effort configured for the published lane, or null when omitted.'
+ example: 'high'
+ type:
+ - 'string'
+ - 'null'
+ temperature:
+ description: 'Sampling temperature configured for the published lane, or null when omitted.'
+ example: 0.2
+ format: 'double'
+ type:
+ - 'number'
+ - 'null'
+ required:
+ - 'max_agent_turns'
+ - 'reasoning_effort'
+ - 'temperature'
+ type: 'object'
UnprocessableEntityResponse:
description: 'Unprocessable Entity - Semantic validation failure'
example:
@@ -27379,7 +27489,7 @@ paths:
- $ref: "#/components/parameters/AppCategories"
/benchmarks:
get:
- description: 'Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter''s own tau-bench and GPQA evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account.'
+ description: 'Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter''s own tau-bench, GPQA, and web-search evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Use task_type=search (or a search_* benchmark_type) for OpenRouter''s search benchmarks, which publish each model''s highest-scoring eligible evaluation configuration with same-configuration runs combined by task-weighted mean. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account.'
operationId: 'getBenchmarks'
parameters:
- description: 'Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources.'
@@ -27395,19 +27505,65 @@ paths:
example: 'artificial-analysis'
type: 'string'
x-speakeasy-unknown-values: allow
- - description: 'Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category.'
+ - description: 'Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only.'
in: 'query'
name: 'task_type'
required: false
schema:
- description: 'Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category.'
+ description: 'Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only.'
enum:
- 'coding'
- 'intelligence'
- 'agentic'
+ - 'search'
example: 'coding'
type: 'string'
x-speakeasy-unknown-values: allow
+ - description: 'Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources'' items as they are.'
+ in: 'query'
+ name: 'benchmark_type'
+ required: false
+ schema:
+ description: 'Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources'' items as they are.'
+ enum:
+ - 'gpqa_diamond'
+ - 'tau_bench_verified_airline'
+ - 'search_browsecomp'
+ - 'search_hle'
+ - 'search_dsqa'
+ - 'search_widesearch'
+ example: 'search_widesearch'
+ type: 'string'
+ x-speakeasy-unknown-values: allow
+ - description: 'Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract.'
+ in: 'query'
+ name: 'include_run_config'
+ required: false
+ schema:
+ default: false
+ description: 'Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract.'
+ example: true
+ type: 'boolean'
+ - description: 'OpenRouter search benchmarks only: filter by the search engine used.'
+ in: 'query'
+ name: 'search_engine'
+ required: false
+ schema:
+ description: 'OpenRouter search benchmarks only: filter by the search engine used.'
+ example: 'exa'
+ type: 'string'
+ - description: 'OpenRouter search benchmarks only: filter by the request surface the lane ran on.'
+ in: 'query'
+ name: 'search_surface'
+ required: false
+ schema:
+ description: 'OpenRouter search benchmarks only: filter by the request surface the lane ran on.'
+ enum:
+ - 'server-tool'
+ - 'plugin'
+ example: 'server-tool'
+ type: 'string'
+ x-speakeasy-unknown-values: allow
- description: 'Design Arena only: arena to query. Defaults to `models` when source is `design-arena`.'
in: 'query'
name: 'arena'
diff --git a/.speakeasy/workflow.lock b/.speakeasy/workflow.lock
index a3ebe171..757af635 100644
--- a/.speakeasy/workflow.lock
+++ b/.speakeasy/workflow.lock
@@ -2,8 +2,8 @@ speakeasyVersion: 1.787.0
sources:
OpenRouter API:
sourceNamespace: open-router-chat-completions-api
- sourceRevisionDigest: sha256:c4947c72f13c5189bdfaf2524d748e70ddad983f99c52a513e72a0cf92981788
- sourceBlobDigest: sha256:d38579d85cc5d56410b05042748d7d0e0a1dd730ca2e0af8ef3fe13b67c00372
+ sourceRevisionDigest: sha256:0f0480623513ea8eb6966a24c5e7ea07750f656ec7a181a95c05b0688a23e203
+ sourceBlobDigest: sha256:86789908c3b9752a073bf734375a2597651a71927911c6b9470089ada3d6d255
tags:
- latest
- 1.0.0
@@ -11,10 +11,10 @@ targets:
open-router:
source: OpenRouter API
sourceNamespace: open-router-chat-completions-api
- sourceRevisionDigest: sha256:c4947c72f13c5189bdfaf2524d748e70ddad983f99c52a513e72a0cf92981788
- sourceBlobDigest: sha256:d38579d85cc5d56410b05042748d7d0e0a1dd730ca2e0af8ef3fe13b67c00372
+ sourceRevisionDigest: sha256:0f0480623513ea8eb6966a24c5e7ea07750f656ec7a181a95c05b0688a23e203
+ sourceBlobDigest: sha256:86789908c3b9752a073bf734375a2597651a71927911c6b9470089ada3d6d255
codeSamplesNamespace: open-router-python-code-samples
- codeSamplesRevisionDigest: sha256:431abf36979e4683c5acf6341ee055a3632f3ec81a916d560a120295beab9107
+ codeSamplesRevisionDigest: sha256:96ade224222d016e3a3e25dcf397a3b3a191917caa45946af874d0f9e9cb4d91
workflow:
workflowVersion: 1.0.0
speakeasyVersion: 1.787.0
diff --git a/RELEASES.md b/RELEASES.md
index 1009d3ec..72c3e00f 100644
--- a/RELEASES.md
+++ b/RELEASES.md
@@ -1219,4 +1219,14 @@ Based on:
### Generated
- [python v1.1.43] .
### Releases
-- [PyPI v1.1.43] https://pypi.org/project/openrouter/1.1.43 - .
\ No newline at end of file
+- [PyPI v1.1.43] https://pypi.org/project/openrouter/1.1.43 - .
+
+## 2026-08-11 14:03:44
+### Changes
+Based on:
+- OpenAPI Doc
+- Speakeasy CLI 1.787.0 (2.914.0) https://github.com/speakeasy-api/speakeasy
+### Generated
+- [python v1.1.44] .
+### Releases
+- [PyPI v1.1.44] https://pypi.org/project/openrouter/1.1.44 - .
\ No newline at end of file
diff --git a/docs/components/primarymetric.mdx b/docs/components/primarymetric.mdx
new file mode 100644
index 00000000..6a99a273
--- /dev/null
+++ b/docs/components/primarymetric.mdx
@@ -0,0 +1,22 @@
+---
+title: "PrimaryMetric"
+---
+
+Identifies the meaning of `primary_score`: `f1_by_item` for WideSearch, `accuracy` for all other search benchmarks.
+
+## Example Usage
+
+```python
+from openrouter.components import PrimaryMetric
+
+# Open enum: unrecognized values are captured as UnrecognizedStr
+value: PrimaryMetric = "accuracy"
+```
+
+
+## Values
+
+This is an open enum. Unrecognized values will not fail type checks.
+
+- `"accuracy"`
+- `"f1_by_item"`
diff --git a/docs/components/searchsurface.mdx b/docs/components/searchsurface.mdx
new file mode 100644
index 00000000..9ef3f90e
--- /dev/null
+++ b/docs/components/searchsurface.mdx
@@ -0,0 +1,22 @@
+---
+title: "SearchSurface"
+---
+
+Request surface the published configuration went through.
+
+## Example Usage
+
+```python
+from openrouter.components import SearchSurface
+
+# Open enum: unrecognized values are captured as UnrecognizedStr
+value: SearchSurface = "server-tool"
+```
+
+
+## Values
+
+This is an open enum. Unrecognized values will not fail type checks.
+
+- `"server-tool"`
+- `"plugin"`
diff --git a/docs/components/unifiedbenchmarksoritem.mdx b/docs/components/unifiedbenchmarksoritem.mdx
index 99228c0b..ec2f5945 100644
--- a/docs/components/unifiedbenchmarksoritem.mdx
+++ b/docs/components/unifiedbenchmarksoritem.mdx
@@ -4,14 +4,14 @@ title: "UnifiedBenchmarksORItem"
## Fields
-| Field | Type | Required | Description | Example |
-| ------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------ |
-| `accuracy` | *float* | :heavy_check_mark: | Aggregate accuracy score from 0 to 1. Higher is better. | 0.72 |
-| `accuracy_stddev` | *Nullable[float]* | :heavy_check_mark: | Standard deviation of run accuracy, or null for a single run. | 0.03 |
-| `avg_cost_per_task` | *Nullable[float]* | :heavy_check_mark: | Average cost per task in USD, or null if unavailable. | 0.002 |
-| `benchmark_type` | [components.BenchmarkType](../components/benchmarktype.mdx) | :heavy_check_mark: | OpenRouter benchmark evaluation type. | gpqa_diamond |
-| `display_name` | *str* | :heavy_check_mark: | Human-readable model name. | GPT-4o |
-| `last_run_timestamp` | *str* | :heavy_check_mark: | Timestamp of the most recent public benchmark run. | 2026-06-03T12:00:00Z |
-| `model_permaslug` | *str* | :heavy_check_mark: | Stable OpenRouter model identifier. | openai/gpt-4o |
-| `source` | [components.UnifiedBenchmarksORItemSource](../components/unifiedbenchmarksoritemsource.mdx) | :heavy_check_mark: | Benchmark source discriminator. | |
-| `total_tasks` | *int* | :heavy_check_mark: | Total benchmark tasks across runs. | 300 |
\ No newline at end of file
+| Field | Type | Required | Description | Example |
+| -------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------- |
+| `accuracy` | *float* | :heavy_check_mark: | Aggregate accuracy score from 0 to 1. Higher is better. | 0.72 |
+| `accuracy_stddev` | *Nullable[float]* | :heavy_check_mark: | Standard deviation of run accuracy, or null for a single run. | 0.03 |
+| `avg_cost_per_task` | *Nullable[float]* | :heavy_check_mark: | Average cost per task in USD, or null if unavailable. | 0.002 |
+| `benchmark_type` | [components.UnifiedBenchmarksORItemBenchmarkType](../components/unifiedbenchmarksoritembenchmarktype.mdx) | :heavy_check_mark: | OpenRouter benchmark evaluation type. | gpqa_diamond |
+| `display_name` | *str* | :heavy_check_mark: | Human-readable model name. | GPT-4o |
+| `last_run_timestamp` | *str* | :heavy_check_mark: | Timestamp of the most recent public benchmark run. | 2026-06-03T12:00:00Z |
+| `model_permaslug` | *str* | :heavy_check_mark: | Stable OpenRouter model identifier. | openai/gpt-4o |
+| `source` | [components.UnifiedBenchmarksORItemSource](../components/unifiedbenchmarksoritemsource.mdx) | :heavy_check_mark: | Benchmark source discriminator. | |
+| `total_tasks` | *int* | :heavy_check_mark: | Total benchmark tasks across runs. | 300 |
\ No newline at end of file
diff --git a/docs/components/benchmarktype.mdx b/docs/components/unifiedbenchmarksoritembenchmarktype.mdx
similarity index 61%
rename from docs/components/benchmarktype.mdx
rename to docs/components/unifiedbenchmarksoritembenchmarktype.mdx
index 2a473fdf..18316442 100644
--- a/docs/components/benchmarktype.mdx
+++ b/docs/components/unifiedbenchmarksoritembenchmarktype.mdx
@@ -1,5 +1,5 @@
---
-title: "BenchmarkType"
+title: "UnifiedBenchmarksORItemBenchmarkType"
---
OpenRouter benchmark evaluation type.
@@ -7,10 +7,10 @@ OpenRouter benchmark evaluation type.
## Example Usage
```python
-from openrouter.components import BenchmarkType
+from openrouter.components import UnifiedBenchmarksORItemBenchmarkType
# Open enum: unrecognized values are captured as UnrecognizedStr
-value: BenchmarkType = "gpqa_diamond"
+value: UnifiedBenchmarksORItemBenchmarkType = "gpqa_diamond"
```
diff --git a/docs/components/unifiedbenchmarksresponsedata.mdx b/docs/components/unifiedbenchmarksresponsedata.mdx
index ff4c0671..50ec3eb6 100644
--- a/docs/components/unifiedbenchmarksresponsedata.mdx
+++ b/docs/components/unifiedbenchmarksresponsedata.mdx
@@ -22,3 +22,9 @@ value: components.UnifiedBenchmarksDAItem = /* values here */
value: components.UnifiedBenchmarksORItem = /* values here */
```
+### `components.UnifiedBenchmarksSearchItem`
+
+```python
+value: components.UnifiedBenchmarksSearchItem = /* values here */
+```
+
diff --git a/docs/components/unifiedbenchmarkssearchitem.mdx b/docs/components/unifiedbenchmarkssearchitem.mdx
new file mode 100644
index 00000000..40b4f0be
--- /dev/null
+++ b/docs/components/unifiedbenchmarkssearchitem.mdx
@@ -0,0 +1,21 @@
+---
+title: "UnifiedBenchmarksSearchItem"
+---
+
+## Fields
+
+| Field | Type | Required | Description | Example |
+| ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `avg_cost_per_task` | *Nullable[float]* | :heavy_check_mark: | Average cost per task in USD, or null if unavailable. | 0.031 |
+| `avg_latency_per_task_ms` | *Nullable[float]* | :heavy_check_mark: | Average wall-clock latency per task in milliseconds, or null if unavailable. | 45210 |
+| `benchmark_type` | [components.UnifiedBenchmarksSearchItemBenchmarkType](../components/unifiedbenchmarkssearchitembenchmarktype.mdx) | :heavy_check_mark: | OpenRouter search benchmark. | search_browsecomp |
+| `display_name` | *str* | :heavy_check_mark: | Human-readable model name. | GPT-4o |
+| `last_run_timestamp` | *str* | :heavy_check_mark: | Timestamp of the newest qualifying run in the published configuration's lane. | 2026-07-28T13:38:18Z |
+| `model_permaslug` | *str* | :heavy_check_mark: | Stable OpenRouter model identifier. | openai/gpt-4o |
+| `primary_metric` | [components.PrimaryMetric](../components/primarymetric.mdx) | :heavy_check_mark: | Identifies the meaning of `primary_score`: `f1_by_item` for WideSearch, `accuracy` for all other search benchmarks. | accuracy |
+| `primary_score` | *float* | :heavy_check_mark: | The benchmark's headline score from 0 to 1. Its meaning is identified by `primary_metric`: item-weighted F1 for WideSearch or strict accuracy for the other search benchmarks. Higher is better. | 0.72 |
+| `run_config` | [Optional[components.UnifiedBenchmarksSearchRunConfig]](../components/unifiedbenchmarkssearchrunconfig.mdx) | :heavy_minus_sign: | Published lane configuration, included only when include_run_config=true. Only the agent turn count, reasoning effort, and temperature are exposed; other harness settings are intentionally not part of the public contract. | |
+| `search_engine` | *str* | :heavy_check_mark: | Search engine the published configuration used. | exa |
+| `search_surface` | [components.SearchSurface](../components/searchsurface.mdx) | :heavy_check_mark: | Request surface the published configuration went through. | server-tool |
+| `source` | [components.UnifiedBenchmarksSearchItemSource](../components/unifiedbenchmarkssearchitemsource.mdx) | :heavy_check_mark: | Benchmark source discriminator. | |
+| `total_tasks` | *int* | :heavy_check_mark: | Tasks evaluated across the published configuration's runs. | 100 |
\ No newline at end of file
diff --git a/docs/components/unifiedbenchmarkssearchitembenchmarktype.mdx b/docs/components/unifiedbenchmarkssearchitembenchmarktype.mdx
new file mode 100644
index 00000000..dfd7cc2d
--- /dev/null
+++ b/docs/components/unifiedbenchmarkssearchitembenchmarktype.mdx
@@ -0,0 +1,24 @@
+---
+title: "UnifiedBenchmarksSearchItemBenchmarkType"
+---
+
+OpenRouter search benchmark.
+
+## Example Usage
+
+```python
+from openrouter.components import UnifiedBenchmarksSearchItemBenchmarkType
+
+# Open enum: unrecognized values are captured as UnrecognizedStr
+value: UnifiedBenchmarksSearchItemBenchmarkType = "search_browsecomp"
+```
+
+
+## Values
+
+This is an open enum. Unrecognized values will not fail type checks.
+
+- `"search_browsecomp"`
+- `"search_hle"`
+- `"search_dsqa"`
+- `"search_widesearch"`
diff --git a/docs/components/unifiedbenchmarkssearchitemsource.mdx b/docs/components/unifiedbenchmarkssearchitemsource.mdx
new file mode 100644
index 00000000..cf189d07
--- /dev/null
+++ b/docs/components/unifiedbenchmarkssearchitemsource.mdx
@@ -0,0 +1,17 @@
+---
+title: "UnifiedBenchmarksSearchItemSource"
+---
+
+Benchmark source discriminator.
+
+## Example Usage
+
+```python
+from openrouter.components import UnifiedBenchmarksSearchItemSource
+value: UnifiedBenchmarksSearchItemSource = "openrouter"
+```
+
+
+## Values
+
+- `"openrouter"`
diff --git a/docs/components/unifiedbenchmarkssearchrunconfig.mdx b/docs/components/unifiedbenchmarkssearchrunconfig.mdx
new file mode 100644
index 00000000..21f463fd
--- /dev/null
+++ b/docs/components/unifiedbenchmarkssearchrunconfig.mdx
@@ -0,0 +1,14 @@
+---
+title: "UnifiedBenchmarksSearchRunConfig"
+---
+
+Published lane configuration, included only when include_run_config=true. Only the agent turn count, reasoning effort, and temperature are exposed; other harness settings are intentionally not part of the public contract.
+
+
+## Fields
+
+| Field | Type | Required | Description | Example |
+| ----------------------------------------------------------------------------- | ----------------------------------------------------------------------------- | ----------------------------------------------------------------------------- | ----------------------------------------------------------------------------- | ----------------------------------------------------------------------------- |
+| `max_agent_turns` | *Nullable[int]* | :heavy_check_mark: | Agent-turn count for the published lane, or null for plugin lanes. | 25 |
+| `reasoning_effort` | *Nullable[str]* | :heavy_check_mark: | Reasoning effort configured for the published lane, or null when omitted. | high |
+| `temperature` | *Nullable[float]* | :heavy_check_mark: | Sampling temperature configured for the published lane, or null when omitted. | 0.2 |
\ No newline at end of file
diff --git a/docs/operations/benchmarktype.mdx b/docs/operations/benchmarktype.mdx
new file mode 100644
index 00000000..390b7aca
--- /dev/null
+++ b/docs/operations/benchmarktype.mdx
@@ -0,0 +1,26 @@
+---
+title: "BenchmarkType"
+---
+
+Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources' items as they are.
+
+## Example Usage
+
+```python
+from openrouter.operations import BenchmarkType
+
+# Open enum: unrecognized values are captured as UnrecognizedStr
+value: BenchmarkType = "gpqa_diamond"
+```
+
+
+## Values
+
+This is an open enum. Unrecognized values will not fail type checks.
+
+- `"gpqa_diamond"`
+- `"tau_bench_verified_airline"`
+- `"search_browsecomp"`
+- `"search_hle"`
+- `"search_dsqa"`
+- `"search_widesearch"`
diff --git a/docs/operations/getbenchmarksrequest.mdx b/docs/operations/getbenchmarksrequest.mdx
index c5acba5b..e45dab26 100644
--- a/docs/operations/getbenchmarksrequest.mdx
+++ b/docs/operations/getbenchmarksrequest.mdx
@@ -4,13 +4,17 @@ title: "GetBenchmarksRequest"
## Fields
-| Field | Type | Required | Description | Example |
-| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| `http_referer` | *Optional[str]* | :heavy_minus_sign: | The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
| |
-| `x_open_router_title` | *Optional[str]* | :heavy_minus_sign: | The app display name allows you to customize how your app appears in OpenRouter's dashboard.
| |
-| `x_open_router_categories` | *Optional[str]* | :heavy_minus_sign: | Comma-separated list of app categories (e.g. "cli-agent,cloud-agent"). Used for marketplace rankings.
| |
-| `source` | [Optional[operations.Source]](../operations/source.mdx) | :heavy_minus_sign: | Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources. | artificial-analysis |
-| `task_type` | [Optional[operations.TaskType]](../operations/tasktype.mdx) | :heavy_minus_sign: | Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. | coding |
-| `arena` | [Optional[operations.Arena]](../operations/arena.mdx) | :heavy_minus_sign: | Design Arena only: arena to query. Defaults to `models` when source is `design-arena`. | models |
-| `category` | *Optional[str]* | :heavy_minus_sign: | Design Arena only: category within the arena (e.g. `codecategories`, `uicomponent`, `gamedev`, `3d`, `dataviz`, `image`, `video`, `svg`). When omitted, returns all categories. | codecategories |
-| `max_results` | *Optional[int]* | :heavy_minus_sign: | Maximum number of items to return. When omitted, all matching results are returned. | 50 |
\ No newline at end of file
+| Field | Type | Required | Description | Example |
+| ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `http_referer` | *Optional[str]* | :heavy_minus_sign: | The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
| |
+| `x_open_router_title` | *Optional[str]* | :heavy_minus_sign: | The app display name allows you to customize how your app appears in OpenRouter's dashboard.
| |
+| `x_open_router_categories` | *Optional[str]* | :heavy_minus_sign: | Comma-separated list of app categories (e.g. "cli-agent,cloud-agent"). Used for marketplace rankings.
| |
+| `source` | [Optional[operations.Source]](../operations/source.mdx) | :heavy_minus_sign: | Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources. | artificial-analysis |
+| `task_type` | [Optional[operations.TaskType]](../operations/tasktype.mdx) | :heavy_minus_sign: | Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only. | coding |
+| `benchmark_type` | [Optional[operations.BenchmarkType]](../operations/benchmarktype.mdx) | :heavy_minus_sign: | Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources' items as they are. | search_widesearch |
+| `include_run_config` | *Optional[bool]* | :heavy_minus_sign: | Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract. | true |
+| `search_engine` | *Optional[str]* | :heavy_minus_sign: | OpenRouter search benchmarks only: filter by the search engine used. | exa |
+| `search_surface` | [Optional[operations.SearchSurface]](../operations/searchsurface.mdx) | :heavy_minus_sign: | OpenRouter search benchmarks only: filter by the request surface the lane ran on. | server-tool |
+| `arena` | [Optional[operations.Arena]](../operations/arena.mdx) | :heavy_minus_sign: | Design Arena only: arena to query. Defaults to `models` when source is `design-arena`. | models |
+| `category` | *Optional[str]* | :heavy_minus_sign: | Design Arena only: category within the arena (e.g. `codecategories`, `uicomponent`, `gamedev`, `3d`, `dataviz`, `image`, `video`, `svg`). When omitted, returns all categories. | codecategories |
+| `max_results` | *Optional[int]* | :heavy_minus_sign: | Maximum number of items to return. When omitted, all matching results are returned. | 50 |
\ No newline at end of file
diff --git a/docs/operations/searchsurface.mdx b/docs/operations/searchsurface.mdx
new file mode 100644
index 00000000..cf8fe4f0
--- /dev/null
+++ b/docs/operations/searchsurface.mdx
@@ -0,0 +1,22 @@
+---
+title: "SearchSurface"
+---
+
+OpenRouter search benchmarks only: filter by the request surface the lane ran on.
+
+## Example Usage
+
+```python
+from openrouter.operations import SearchSurface
+
+# Open enum: unrecognized values are captured as UnrecognizedStr
+value: SearchSurface = "server-tool"
+```
+
+
+## Values
+
+This is an open enum. Unrecognized values will not fail type checks.
+
+- `"server-tool"`
+- `"plugin"`
diff --git a/docs/operations/tasktype.mdx b/docs/operations/tasktype.mdx
index 7cdaa5db..ca10bf29 100644
--- a/docs/operations/tasktype.mdx
+++ b/docs/operations/tasktype.mdx
@@ -2,7 +2,7 @@
title: "TaskType"
---
-Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category.
+Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only.
## Example Usage
@@ -21,3 +21,4 @@ This is an open enum. Unrecognized values will not fail type checks.
- `"coding"`
- `"intelligence"`
- `"agentic"`
+- `"search"`
diff --git a/docs/sdks/benchmarks/README.mdx b/docs/sdks/benchmarks/README.mdx
index da27ed83..5026ac2a 100644
--- a/docs/sdks/benchmarks/README.mdx
+++ b/docs/sdks/benchmarks/README.mdx
@@ -13,7 +13,7 @@ Benchmarks endpoints
## get_benchmarks
-Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter's own tau-bench and GPQA evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account.
+Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter's own tau-bench, GPQA, and web-search evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Use task_type=search (or a search_* benchmark_type) for OpenRouter's search benchmarks, which publish each model's highest-scoring eligible evaluation configuration with same-configuration runs combined by task-weighted mean. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account.
### Example Usage
@@ -29,7 +29,7 @@ with OpenRouter(
api_key=os.getenv("OPENROUTER_API_KEY", ""),
) as open_router:
- res = open_router.benchmarks.get_benchmarks()
+ res = open_router.benchmarks.get_benchmarks(include_run_config=True)
# Handle response
print(res)
@@ -38,17 +38,21 @@ with OpenRouter(
### Parameters
-| Parameter | Type | Required | Description | Example |
-| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| `http_referer` | *Optional[str]* | :heavy_minus_sign: | The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
| |
-| `x_open_router_title` | *Optional[str]* | :heavy_minus_sign: | The app display name allows you to customize how your app appears in OpenRouter's dashboard.
| |
-| `x_open_router_categories` | *Optional[str]* | :heavy_minus_sign: | Comma-separated list of app categories (e.g. "cli-agent,cloud-agent"). Used for marketplace rankings.
| |
-| `source` | [Optional[operations.Source]](../../operations/source.mdx) | :heavy_minus_sign: | Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources. | artificial-analysis |
-| `task_type` | [Optional[operations.TaskType]](../../operations/tasktype.mdx) | :heavy_minus_sign: | Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. | coding |
-| `arena` | [Optional[operations.Arena]](../../operations/arena.mdx) | :heavy_minus_sign: | Design Arena only: arena to query. Defaults to `models` when source is `design-arena`. | models |
-| `category` | *Optional[str]* | :heavy_minus_sign: | Design Arena only: category within the arena (e.g. `codecategories`, `uicomponent`, `gamedev`, `3d`, `dataviz`, `image`, `video`, `svg`). When omitted, returns all categories. | codecategories |
-| `max_results` | *Optional[int]* | :heavy_minus_sign: | Maximum number of items to return. When omitted, all matching results are returned. | 50 |
-| `retries` | [Optional[utils.RetryConfig]](../../models/utils/retryconfig.mdx) | :heavy_minus_sign: | Configuration to override the default retry behavior of the client. | |
+| Parameter | Type | Required | Description | Example |
+| ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `http_referer` | *Optional[str]* | :heavy_minus_sign: | The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
| |
+| `x_open_router_title` | *Optional[str]* | :heavy_minus_sign: | The app display name allows you to customize how your app appears in OpenRouter's dashboard.
| |
+| `x_open_router_categories` | *Optional[str]* | :heavy_minus_sign: | Comma-separated list of app categories (e.g. "cli-agent,cloud-agent"). Used for marketplace rankings.
| |
+| `source` | [Optional[operations.Source]](../../operations/source.mdx) | :heavy_minus_sign: | Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources. | artificial-analysis |
+| `task_type` | [Optional[operations.TaskType]](../../operations/tasktype.mdx) | :heavy_minus_sign: | Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only. | coding |
+| `benchmark_type` | [Optional[operations.BenchmarkType]](../../operations/benchmarktype.mdx) | :heavy_minus_sign: | Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources' items as they are. | search_widesearch |
+| `include_run_config` | *Optional[bool]* | :heavy_minus_sign: | Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract. | true |
+| `search_engine` | *Optional[str]* | :heavy_minus_sign: | OpenRouter search benchmarks only: filter by the search engine used. | exa |
+| `search_surface` | [Optional[operations.SearchSurface]](../../operations/searchsurface.mdx) | :heavy_minus_sign: | OpenRouter search benchmarks only: filter by the request surface the lane ran on. | server-tool |
+| `arena` | [Optional[operations.Arena]](../../operations/arena.mdx) | :heavy_minus_sign: | Design Arena only: arena to query. Defaults to `models` when source is `design-arena`. | models |
+| `category` | *Optional[str]* | :heavy_minus_sign: | Design Arena only: category within the arena (e.g. `codecategories`, `uicomponent`, `gamedev`, `3d`, `dataviz`, `image`, `video`, `svg`). When omitted, returns all categories. | codecategories |
+| `max_results` | *Optional[int]* | :heavy_minus_sign: | Maximum number of items to return. When omitted, all matching results are returned. | 50 |
+| `retries` | [Optional[utils.RetryConfig]](../../models/utils/retryconfig.mdx) | :heavy_minus_sign: | Configuration to override the default retry behavior of the client. | |
### Response
diff --git a/pyproject.toml b/pyproject.toml
index dd8146be..2633c255 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -1,6 +1,6 @@
[project]
name = "openrouter"
-version = "1.1.43"
+version = "1.1.44"
description = "Official Python Client SDK for OpenRouter."
authors = [{ name = "OpenRouter" },]
readme = "README-PYPI.md"
diff --git a/src/openrouter/_version.py b/src/openrouter/_version.py
index 5f253b27..05527560 100644
--- a/src/openrouter/_version.py
+++ b/src/openrouter/_version.py
@@ -3,10 +3,10 @@
import importlib.metadata
__title__: str = "openrouter"
-__version__: str = "1.1.43"
+__version__: str = "1.1.44"
__openapi_doc_version__: str = "1.0.0"
__gen_version__: str = "2.914.0"
-__user_agent__: str = "speakeasy-sdk/python 1.1.43 2.914.0 1.0.0 openrouter"
+__user_agent__: str = "speakeasy-sdk/python 1.1.44 2.914.0 1.0.0 openrouter"
try:
if __package__ is not None:
diff --git a/src/openrouter/benchmarks.py b/src/openrouter/benchmarks.py
index 5dcb0e04..f163fcbb 100644
--- a/src/openrouter/benchmarks.py
+++ b/src/openrouter/benchmarks.py
@@ -20,6 +20,10 @@ def get_benchmarks(
x_open_router_categories: Optional[str] = None,
source: Optional[operations.Source] = None,
task_type: Optional[operations.TaskType] = None,
+ benchmark_type: Optional[operations.BenchmarkType] = None,
+ include_run_config: Optional[bool] = False,
+ search_engine: Optional[str] = None,
+ search_surface: Optional[operations.SearchSurface] = None,
arena: Optional[operations.Arena] = None,
category: Optional[str] = None,
max_results: Optional[int] = None,
@@ -30,7 +34,7 @@ def get_benchmarks(
) -> components.UnifiedBenchmarksResponse:
r"""List Benchmarks
- Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter's own tau-bench and GPQA evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account.
+ Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter's own tau-bench, GPQA, and web-search evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Use task_type=search (or a search_* benchmark_type) for OpenRouter's search benchmarks, which publish each model's highest-scoring eligible evaluation configuration with same-configuration runs combined by task-weighted mean. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account.
:param http_referer: The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
@@ -40,7 +44,11 @@ def get_benchmarks(
:param x_open_router_categories: Comma-separated list of app categories (e.g. \"cli-agent,cloud-agent\"). Used for marketplace rankings.
:param source: Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources.
- :param task_type: Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category.
+ :param task_type: Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only.
+ :param benchmark_type: Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources' items as they are.
+ :param include_run_config: Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract.
+ :param search_engine: OpenRouter search benchmarks only: filter by the search engine used.
+ :param search_surface: OpenRouter search benchmarks only: filter by the request surface the lane ran on.
:param arena: Design Arena only: arena to query. Defaults to `models` when source is `design-arena`.
:param category: Design Arena only: category within the arena (e.g. `codecategories`, `uicomponent`, `gamedev`, `3d`, `dataviz`, `image`, `video`, `svg`). When omitted, returns all categories.
:param max_results: Maximum number of items to return. When omitted, all matching results are returned.
@@ -65,6 +73,10 @@ def get_benchmarks(
x_open_router_categories=x_open_router_categories,
source=source,
task_type=task_type,
+ benchmark_type=benchmark_type,
+ include_run_config=include_run_config,
+ search_engine=search_engine,
+ search_surface=search_surface,
arena=arena,
category=category,
max_results=max_results,
@@ -167,6 +179,10 @@ async def get_benchmarks_async(
x_open_router_categories: Optional[str] = None,
source: Optional[operations.Source] = None,
task_type: Optional[operations.TaskType] = None,
+ benchmark_type: Optional[operations.BenchmarkType] = None,
+ include_run_config: Optional[bool] = False,
+ search_engine: Optional[str] = None,
+ search_surface: Optional[operations.SearchSurface] = None,
arena: Optional[operations.Arena] = None,
category: Optional[str] = None,
max_results: Optional[int] = None,
@@ -177,7 +193,7 @@ async def get_benchmarks_async(
) -> components.UnifiedBenchmarksResponse:
r"""List Benchmarks
- Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter's own tau-bench and GPQA evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account.
+ Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter's own tau-bench, GPQA, and web-search evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Use task_type=search (or a search_* benchmark_type) for OpenRouter's search benchmarks, which publish each model's highest-scoring eligible evaluation configuration with same-configuration runs combined by task-weighted mean. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account.
:param http_referer: The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
@@ -187,7 +203,11 @@ async def get_benchmarks_async(
:param x_open_router_categories: Comma-separated list of app categories (e.g. \"cli-agent,cloud-agent\"). Used for marketplace rankings.
:param source: Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources.
- :param task_type: Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category.
+ :param task_type: Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only.
+ :param benchmark_type: Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources' items as they are.
+ :param include_run_config: Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract.
+ :param search_engine: OpenRouter search benchmarks only: filter by the search engine used.
+ :param search_surface: OpenRouter search benchmarks only: filter by the request surface the lane ran on.
:param arena: Design Arena only: arena to query. Defaults to `models` when source is `design-arena`.
:param category: Design Arena only: category within the arena (e.g. `codecategories`, `uicomponent`, `gamedev`, `3d`, `dataviz`, `image`, `video`, `svg`). When omitted, returns all categories.
:param max_results: Maximum number of items to return. When omitted, all matching results are returned.
@@ -212,6 +232,10 @@ async def get_benchmarks_async(
x_open_router_categories=x_open_router_categories,
source=source,
task_type=task_type,
+ benchmark_type=benchmark_type,
+ include_run_config=include_run_config,
+ search_engine=search_engine,
+ search_surface=search_surface,
arena=arena,
category=category,
max_results=max_results,
diff --git a/src/openrouter/components/__init__.py b/src/openrouter/components/__init__.py
index 670eeb2d..f714cc33 100644
--- a/src/openrouter/components/__init__.py
+++ b/src/openrouter/components/__init__.py
@@ -2890,8 +2890,8 @@
UnifiedBenchmarksMetaVersion,
)
from .unifiedbenchmarksoritem import (
- BenchmarkType,
UnifiedBenchmarksORItem,
+ UnifiedBenchmarksORItemBenchmarkType,
UnifiedBenchmarksORItemSource,
UnifiedBenchmarksORItemTypedDict,
)
@@ -2900,7 +2900,18 @@
UnifiedBenchmarksResponseData,
UnifiedBenchmarksResponseDataTypedDict,
UnifiedBenchmarksResponseTypedDict,
- UnknownUnifiedBenchmarksResponseData,
+ )
+ from .unifiedbenchmarkssearchitem import (
+ PrimaryMetric,
+ SearchSurface,
+ UnifiedBenchmarksSearchItem,
+ UnifiedBenchmarksSearchItemBenchmarkType,
+ UnifiedBenchmarksSearchItemSource,
+ UnifiedBenchmarksSearchItemTypedDict,
+ )
+ from .unifiedbenchmarkssearchrunconfig import (
+ UnifiedBenchmarksSearchRunConfig,
+ UnifiedBenchmarksSearchRunConfigTypedDict,
)
from .unprocessableentityresponseerrordata import (
UnprocessableEntityResponseErrorData,
@@ -3344,7 +3355,6 @@
"BashServerToolEnvironmentTypedDict",
"BashServerToolType",
"BashServerToolTypedDict",
- "BenchmarkType",
"Billable",
"BooleanCapability",
"BooleanCapabilityType",
@@ -4741,6 +4751,7 @@
"PricingOverride",
"PricingOverrideTypedDict",
"PricingTypedDict",
+ "PrimaryMetric",
"PromptCacheBreakpoint",
"PromptCacheBreakpointMode",
"PromptCacheBreakpointTypedDict",
@@ -4921,6 +4932,7 @@
"SearchModelsServerToolOpenRouterType",
"SearchModelsServerToolOpenRouterTypedDict",
"SearchQualityLevel",
+ "SearchSurface",
"Security",
"SecurityTypedDict",
"ServerToolUse",
@@ -5165,12 +5177,19 @@
"UnifiedBenchmarksMetaTypedDict",
"UnifiedBenchmarksMetaVersion",
"UnifiedBenchmarksORItem",
+ "UnifiedBenchmarksORItemBenchmarkType",
"UnifiedBenchmarksORItemSource",
"UnifiedBenchmarksORItemTypedDict",
"UnifiedBenchmarksResponse",
"UnifiedBenchmarksResponseData",
"UnifiedBenchmarksResponseDataTypedDict",
"UnifiedBenchmarksResponseTypedDict",
+ "UnifiedBenchmarksSearchItem",
+ "UnifiedBenchmarksSearchItemBenchmarkType",
+ "UnifiedBenchmarksSearchItemSource",
+ "UnifiedBenchmarksSearchItemTypedDict",
+ "UnifiedBenchmarksSearchRunConfig",
+ "UnifiedBenchmarksSearchRunConfigTypedDict",
"UniqueInsight",
"UniqueInsightTypedDict",
"Unit",
@@ -5201,7 +5220,6 @@
"UnknownOutputMessageItemContent",
"UnknownReasoningDetailUnion",
"UnknownStreamEvents",
- "UnknownUnifiedBenchmarksResponseData",
"UnprocessableEntityResponseErrorData",
"UnprocessableEntityResponseErrorDataTypedDict",
"UpdateBYOKKeyRequest",
@@ -7432,15 +7450,22 @@
"UnifiedBenchmarksMetaSource": ".unifiedbenchmarksmeta",
"UnifiedBenchmarksMetaTypedDict": ".unifiedbenchmarksmeta",
"UnifiedBenchmarksMetaVersion": ".unifiedbenchmarksmeta",
- "BenchmarkType": ".unifiedbenchmarksoritem",
"UnifiedBenchmarksORItem": ".unifiedbenchmarksoritem",
+ "UnifiedBenchmarksORItemBenchmarkType": ".unifiedbenchmarksoritem",
"UnifiedBenchmarksORItemSource": ".unifiedbenchmarksoritem",
"UnifiedBenchmarksORItemTypedDict": ".unifiedbenchmarksoritem",
"UnifiedBenchmarksResponse": ".unifiedbenchmarksresponse",
"UnifiedBenchmarksResponseData": ".unifiedbenchmarksresponse",
"UnifiedBenchmarksResponseDataTypedDict": ".unifiedbenchmarksresponse",
"UnifiedBenchmarksResponseTypedDict": ".unifiedbenchmarksresponse",
- "UnknownUnifiedBenchmarksResponseData": ".unifiedbenchmarksresponse",
+ "PrimaryMetric": ".unifiedbenchmarkssearchitem",
+ "SearchSurface": ".unifiedbenchmarkssearchitem",
+ "UnifiedBenchmarksSearchItem": ".unifiedbenchmarkssearchitem",
+ "UnifiedBenchmarksSearchItemBenchmarkType": ".unifiedbenchmarkssearchitem",
+ "UnifiedBenchmarksSearchItemSource": ".unifiedbenchmarkssearchitem",
+ "UnifiedBenchmarksSearchItemTypedDict": ".unifiedbenchmarkssearchitem",
+ "UnifiedBenchmarksSearchRunConfig": ".unifiedbenchmarkssearchrunconfig",
+ "UnifiedBenchmarksSearchRunConfigTypedDict": ".unifiedbenchmarkssearchrunconfig",
"UnprocessableEntityResponseErrorData": ".unprocessableentityresponseerrordata",
"UnprocessableEntityResponseErrorDataTypedDict": ".unprocessableentityresponseerrordata",
"UpdateBYOKKeyRequest": ".updatebyokkeyrequest",
diff --git a/src/openrouter/components/unifiedbenchmarksoritem.py b/src/openrouter/components/unifiedbenchmarksoritem.py
index c58ba941..c1671f3e 100644
--- a/src/openrouter/components/unifiedbenchmarksoritem.py
+++ b/src/openrouter/components/unifiedbenchmarksoritem.py
@@ -7,7 +7,7 @@
from typing_extensions import TypedDict
-BenchmarkType = Union[
+UnifiedBenchmarksORItemBenchmarkType = Union[
Literal[
"gpqa_diamond",
"tau_bench_verified_airline",
@@ -28,7 +28,7 @@ class UnifiedBenchmarksORItemTypedDict(TypedDict):
r"""Standard deviation of run accuracy, or null for a single run."""
avg_cost_per_task: Nullable[float]
r"""Average cost per task in USD, or null if unavailable."""
- benchmark_type: BenchmarkType
+ benchmark_type: UnifiedBenchmarksORItemBenchmarkType
r"""OpenRouter benchmark evaluation type."""
display_name: str
r"""Human-readable model name."""
@@ -52,7 +52,7 @@ class UnifiedBenchmarksORItem(BaseModel):
avg_cost_per_task: Nullable[float]
r"""Average cost per task in USD, or null if unavailable."""
- benchmark_type: BenchmarkType
+ benchmark_type: UnifiedBenchmarksORItemBenchmarkType
r"""OpenRouter benchmark evaluation type."""
display_name: str
diff --git a/src/openrouter/components/unifiedbenchmarksresponse.py b/src/openrouter/components/unifiedbenchmarksresponse.py
index 311cb86d..e2ed5fb9 100644
--- a/src/openrouter/components/unifiedbenchmarksresponse.py
+++ b/src/openrouter/components/unifiedbenchmarksresponse.py
@@ -14,13 +14,13 @@
UnifiedBenchmarksORItem,
UnifiedBenchmarksORItemTypedDict,
)
-from functools import partial
+from .unifiedbenchmarkssearchitem import (
+ UnifiedBenchmarksSearchItem,
+ UnifiedBenchmarksSearchItemTypedDict,
+)
from openrouter.types import BaseModel
-from openrouter.utils.unions import parse_open_union
-from pydantic import ConfigDict
-from pydantic.functional_validators import BeforeValidator
-from typing import Any, List, Literal, Union
-from typing_extensions import Annotated, TypeAliasType, TypedDict
+from typing import List, Union
+from typing_extensions import TypeAliasType, TypedDict
UnifiedBenchmarksResponseDataTypedDict = TypeAliasType(
@@ -29,44 +29,20 @@
UnifiedBenchmarksAAItemTypedDict,
UnifiedBenchmarksORItemTypedDict,
UnifiedBenchmarksDAItemTypedDict,
+ UnifiedBenchmarksSearchItemTypedDict,
],
)
-class UnknownUnifiedBenchmarksResponseData(BaseModel):
- r"""A UnifiedBenchmarksResponseData variant the SDK doesn't recognize. Preserves the raw payload."""
-
- source: Literal["UNKNOWN"] = "UNKNOWN"
- raw: Any
- is_unknown: Literal[True] = True
-
- model_config = ConfigDict(frozen=True)
-
-
-_UNIFIED_BENCHMARKS_RESPONSE_DATA_VARIANTS: dict[str, Any] = {
- "artificial-analysis": UnifiedBenchmarksAAItem,
- "design-arena": UnifiedBenchmarksDAItem,
- "openrouter": UnifiedBenchmarksORItem,
-}
-
-
-UnifiedBenchmarksResponseData = Annotated[
+UnifiedBenchmarksResponseData = TypeAliasType(
+ "UnifiedBenchmarksResponseData",
Union[
UnifiedBenchmarksAAItem,
- UnifiedBenchmarksDAItem,
UnifiedBenchmarksORItem,
- UnknownUnifiedBenchmarksResponseData,
+ UnifiedBenchmarksDAItem,
+ UnifiedBenchmarksSearchItem,
],
- BeforeValidator(
- partial(
- parse_open_union,
- disc_key="source",
- variants=_UNIFIED_BENCHMARKS_RESPONSE_DATA_VARIANTS,
- unknown_cls=UnknownUnifiedBenchmarksResponseData,
- union_name="UnifiedBenchmarksResponseData",
- )
- ),
-]
+)
class UnifiedBenchmarksResponseTypedDict(TypedDict):
diff --git a/src/openrouter/components/unifiedbenchmarkssearchitem.py b/src/openrouter/components/unifiedbenchmarkssearchitem.py
new file mode 100644
index 00000000..f0b74435
--- /dev/null
+++ b/src/openrouter/components/unifiedbenchmarkssearchitem.py
@@ -0,0 +1,142 @@
+"""Code generated by Speakeasy (https://speakeasy.com). DO NOT EDIT."""
+
+from __future__ import annotations
+from .unifiedbenchmarkssearchrunconfig import (
+ UnifiedBenchmarksSearchRunConfig,
+ UnifiedBenchmarksSearchRunConfigTypedDict,
+)
+from openrouter.types import BaseModel, Nullable, UNSET_SENTINEL, UnrecognizedStr
+from pydantic import model_serializer
+from typing import Literal, Optional, Union
+from typing_extensions import NotRequired, TypedDict
+
+
+UnifiedBenchmarksSearchItemBenchmarkType = Union[
+ Literal[
+ "search_browsecomp",
+ "search_hle",
+ "search_dsqa",
+ "search_widesearch",
+ ],
+ UnrecognizedStr,
+]
+r"""OpenRouter search benchmark."""
+
+
+PrimaryMetric = Union[
+ Literal[
+ "accuracy",
+ "f1_by_item",
+ ],
+ UnrecognizedStr,
+]
+r"""Identifies the meaning of `primary_score`: `f1_by_item` for WideSearch, `accuracy` for all other search benchmarks."""
+
+
+SearchSurface = Union[
+ Literal[
+ "server-tool",
+ "plugin",
+ ],
+ UnrecognizedStr,
+]
+r"""Request surface the published configuration went through."""
+
+
+UnifiedBenchmarksSearchItemSource = Literal["openrouter",]
+r"""Benchmark source discriminator."""
+
+
+class UnifiedBenchmarksSearchItemTypedDict(TypedDict):
+ avg_cost_per_task: Nullable[float]
+ r"""Average cost per task in USD, or null if unavailable."""
+ avg_latency_per_task_ms: Nullable[float]
+ r"""Average wall-clock latency per task in milliseconds, or null if unavailable."""
+ benchmark_type: UnifiedBenchmarksSearchItemBenchmarkType
+ r"""OpenRouter search benchmark."""
+ display_name: str
+ r"""Human-readable model name."""
+ last_run_timestamp: str
+ r"""Timestamp of the newest qualifying run in the published configuration's lane."""
+ model_permaslug: str
+ r"""Stable OpenRouter model identifier."""
+ primary_metric: PrimaryMetric
+ r"""Identifies the meaning of `primary_score`: `f1_by_item` for WideSearch, `accuracy` for all other search benchmarks."""
+ primary_score: float
+ r"""The benchmark's headline score from 0 to 1. Its meaning is identified by `primary_metric`: item-weighted F1 for WideSearch or strict accuracy for the other search benchmarks. Higher is better."""
+ search_engine: str
+ r"""Search engine the published configuration used."""
+ search_surface: SearchSurface
+ r"""Request surface the published configuration went through."""
+ source: UnifiedBenchmarksSearchItemSource
+ r"""Benchmark source discriminator."""
+ total_tasks: int
+ r"""Tasks evaluated across the published configuration's runs."""
+ run_config: NotRequired[UnifiedBenchmarksSearchRunConfigTypedDict]
+ r"""Published lane configuration, included only when include_run_config=true. Only the agent turn count, reasoning effort, and temperature are exposed; other harness settings are intentionally not part of the public contract."""
+
+
+class UnifiedBenchmarksSearchItem(BaseModel):
+ avg_cost_per_task: Nullable[float]
+ r"""Average cost per task in USD, or null if unavailable."""
+
+ avg_latency_per_task_ms: Nullable[float]
+ r"""Average wall-clock latency per task in milliseconds, or null if unavailable."""
+
+ benchmark_type: UnifiedBenchmarksSearchItemBenchmarkType
+ r"""OpenRouter search benchmark."""
+
+ display_name: str
+ r"""Human-readable model name."""
+
+ last_run_timestamp: str
+ r"""Timestamp of the newest qualifying run in the published configuration's lane."""
+
+ model_permaslug: str
+ r"""Stable OpenRouter model identifier."""
+
+ primary_metric: PrimaryMetric
+ r"""Identifies the meaning of `primary_score`: `f1_by_item` for WideSearch, `accuracy` for all other search benchmarks."""
+
+ primary_score: float
+ r"""The benchmark's headline score from 0 to 1. Its meaning is identified by `primary_metric`: item-weighted F1 for WideSearch or strict accuracy for the other search benchmarks. Higher is better."""
+
+ search_engine: str
+ r"""Search engine the published configuration used."""
+
+ search_surface: SearchSurface
+ r"""Request surface the published configuration went through."""
+
+ source: UnifiedBenchmarksSearchItemSource
+ r"""Benchmark source discriminator."""
+
+ total_tasks: int
+ r"""Tasks evaluated across the published configuration's runs."""
+
+ run_config: Optional[UnifiedBenchmarksSearchRunConfig] = None
+ r"""Published lane configuration, included only when include_run_config=true. Only the agent turn count, reasoning effort, and temperature are exposed; other harness settings are intentionally not part of the public contract."""
+
+ @model_serializer(mode="wrap")
+ def serialize_model(self, handler):
+ optional_fields = set(["run_config"])
+ nullable_fields = set(["avg_cost_per_task", "avg_latency_per_task_ms"])
+ serialized = handler(self)
+ m = {}
+
+ for n, f in type(self).model_fields.items():
+ k = f.alias or n
+ val = serialized.get(k, serialized.get(n))
+ is_nullable_and_explicitly_set = (
+ k in nullable_fields
+ and (self.__pydantic_fields_set__.intersection({n})) # pylint: disable=no-member
+ )
+
+ if val != UNSET_SENTINEL:
+ if (
+ val is not None
+ or k not in optional_fields
+ or is_nullable_and_explicitly_set
+ ):
+ m[k] = val
+
+ return m
diff --git a/src/openrouter/components/unifiedbenchmarkssearchrunconfig.py b/src/openrouter/components/unifiedbenchmarkssearchrunconfig.py
new file mode 100644
index 00000000..e1b94da1
--- /dev/null
+++ b/src/openrouter/components/unifiedbenchmarkssearchrunconfig.py
@@ -0,0 +1,44 @@
+"""Code generated by Speakeasy (https://speakeasy.com). DO NOT EDIT."""
+
+from __future__ import annotations
+from openrouter.types import BaseModel, Nullable, UNSET_SENTINEL
+from pydantic import model_serializer
+from typing_extensions import TypedDict
+
+
+class UnifiedBenchmarksSearchRunConfigTypedDict(TypedDict):
+ r"""Published lane configuration, included only when include_run_config=true. Only the agent turn count, reasoning effort, and temperature are exposed; other harness settings are intentionally not part of the public contract."""
+
+ max_agent_turns: Nullable[int]
+ r"""Agent-turn count for the published lane, or null for plugin lanes."""
+ reasoning_effort: Nullable[str]
+ r"""Reasoning effort configured for the published lane, or null when omitted."""
+ temperature: Nullable[float]
+ r"""Sampling temperature configured for the published lane, or null when omitted."""
+
+
+class UnifiedBenchmarksSearchRunConfig(BaseModel):
+ r"""Published lane configuration, included only when include_run_config=true. Only the agent turn count, reasoning effort, and temperature are exposed; other harness settings are intentionally not part of the public contract."""
+
+ max_agent_turns: Nullable[int]
+ r"""Agent-turn count for the published lane, or null for plugin lanes."""
+
+ reasoning_effort: Nullable[str]
+ r"""Reasoning effort configured for the published lane, or null when omitted."""
+
+ temperature: Nullable[float]
+ r"""Sampling temperature configured for the published lane, or null when omitted."""
+
+ @model_serializer(mode="wrap")
+ def serialize_model(self, handler):
+ serialized = handler(self)
+ m = {}
+
+ for n, f in type(self).model_fields.items():
+ k = f.alias or n
+ val = serialized.get(k, serialized.get(n))
+
+ if val != UNSET_SENTINEL:
+ m[k] = val
+
+ return m
diff --git a/src/openrouter/operations/__init__.py b/src/openrouter/operations/__init__.py
index 37c3c8ce..50ebc6f5 100644
--- a/src/openrouter/operations/__init__.py
+++ b/src/openrouter/operations/__init__.py
@@ -326,10 +326,12 @@
)
from .getbenchmarks import (
Arena,
+ BenchmarkType,
GetBenchmarksGlobals,
GetBenchmarksGlobalsTypedDict,
GetBenchmarksRequest,
GetBenchmarksRequestTypedDict,
+ SearchSurface,
Source,
TaskType,
)
@@ -813,6 +815,7 @@
__all__ = [
"Arena",
+ "BenchmarkType",
"BulkAddWorkspaceMembersGlobals",
"BulkAddWorkspaceMembersGlobalsTypedDict",
"BulkAddWorkspaceMembersRequest",
@@ -1356,6 +1359,7 @@
"Result",
"ResultTypedDict",
"Role",
+ "SearchSurface",
"SendChatCompletionRequestGlobals",
"SendChatCompletionRequestGlobalsTypedDict",
"SendChatCompletionRequestRequest",
@@ -1678,10 +1682,12 @@
"GetAppRankingsSort": ".getapprankings",
"Subcategory": ".getapprankings",
"Arena": ".getbenchmarks",
+ "BenchmarkType": ".getbenchmarks",
"GetBenchmarksGlobals": ".getbenchmarks",
"GetBenchmarksGlobalsTypedDict": ".getbenchmarks",
"GetBenchmarksRequest": ".getbenchmarks",
"GetBenchmarksRequestTypedDict": ".getbenchmarks",
+ "SearchSurface": ".getbenchmarks",
"Source": ".getbenchmarks",
"TaskType": ".getbenchmarks",
"GetBYOKKeyGlobals": ".getbyokkey",
diff --git a/src/openrouter/operations/getbenchmarks.py b/src/openrouter/operations/getbenchmarks.py
index 62573095..51f1f39f 100644
--- a/src/openrouter/operations/getbenchmarks.py
+++ b/src/openrouter/operations/getbenchmarks.py
@@ -89,10 +89,35 @@ def serialize_model(self, handler):
"coding",
"intelligence",
"agentic",
+ "search",
],
UnrecognizedStr,
]
-r"""Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category."""
+r"""Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only."""
+
+
+BenchmarkType = Union[
+ Literal[
+ "gpqa_diamond",
+ "tau_bench_verified_airline",
+ "search_browsecomp",
+ "search_hle",
+ "search_dsqa",
+ "search_widesearch",
+ ],
+ UnrecognizedStr,
+]
+r"""Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources' items as they are."""
+
+
+SearchSurface = Union[
+ Literal[
+ "server-tool",
+ "plugin",
+ ],
+ UnrecognizedStr,
+]
+r"""OpenRouter search benchmarks only: filter by the request surface the lane ran on."""
Arena = Union[
@@ -123,7 +148,15 @@ class GetBenchmarksRequestTypedDict(TypedDict):
source: NotRequired[Source]
r"""Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources."""
task_type: NotRequired[TaskType]
- r"""Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category."""
+ r"""Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only."""
+ benchmark_type: NotRequired[BenchmarkType]
+ r"""Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources' items as they are."""
+ include_run_config: NotRequired[bool]
+ r"""Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract."""
+ search_engine: NotRequired[str]
+ r"""OpenRouter search benchmarks only: filter by the search engine used."""
+ search_surface: NotRequired[SearchSurface]
+ r"""OpenRouter search benchmarks only: filter by the request surface the lane ran on."""
arena: NotRequired[Arena]
r"""Design Arena only: arena to query. Defaults to `models` when source is `design-arena`."""
category: NotRequired[str]
@@ -171,7 +204,31 @@ class GetBenchmarksRequest(BaseModel):
Optional[TaskType],
FieldMetadata(query=QueryParamMetadata(style="form", explode=True)),
] = None
- r"""Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category."""
+ r"""Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only."""
+
+ benchmark_type: Annotated[
+ Optional[BenchmarkType],
+ FieldMetadata(query=QueryParamMetadata(style="form", explode=True)),
+ ] = None
+ r"""Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources' items as they are."""
+
+ include_run_config: Annotated[
+ Optional[bool],
+ FieldMetadata(query=QueryParamMetadata(style="form", explode=True)),
+ ] = False
+ r"""Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract."""
+
+ search_engine: Annotated[
+ Optional[str],
+ FieldMetadata(query=QueryParamMetadata(style="form", explode=True)),
+ ] = None
+ r"""OpenRouter search benchmarks only: filter by the search engine used."""
+
+ search_surface: Annotated[
+ Optional[SearchSurface],
+ FieldMetadata(query=QueryParamMetadata(style="form", explode=True)),
+ ] = None
+ r"""OpenRouter search benchmarks only: filter by the request surface the lane ran on."""
arena: Annotated[
Optional[Arena],
@@ -200,6 +257,10 @@ def serialize_model(self, handler):
"X-OpenRouter-Categories",
"source",
"task_type",
+ "benchmark_type",
+ "include_run_config",
+ "search_engine",
+ "search_surface",
"arena",
"category",
"max_results",
diff --git a/uv.lock b/uv.lock
index d823e181..3ac0d57c 100644
--- a/uv.lock
+++ b/uv.lock
@@ -213,7 +213,7 @@ wheels = [
[[package]]
name = "openrouter"
-version = "1.1.43"
+version = "1.1.44"
source = { editable = "." }
dependencies = [
{ name = "httpcore" },